Compare commits
475 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| cf2560c9f8 | |||
| 800d3d7089 | |||
| b471207f7a | |||
| 25c59b52e3 | |||
| 4c72097478 | |||
| 1053cfa0f8 | |||
| 21a42cb6cd | |||
| 16c4a0d985 | |||
| 298253401a | |||
| 32aa980069 | |||
| 7b4d396eb1 | |||
| 36968fbb30 | |||
| 7474888801 | |||
| d46b7a43eb | |||
| 2376b32a8a | |||
| b8a88cc30c | |||
| be0b117711 | |||
| 449c8ea5af | |||
| 900da76652 | |||
| df0ee24084 | |||
| d3d8f53156 | |||
| 6eee303d7e | |||
| 053d71f5aa | |||
| 4fbbf5882c | |||
| 7012103b3b | |||
| 967024d20a | |||
| b966a8078b | |||
| f63b015284 | |||
| 0a9672d910 | |||
| 4ccccd0cd2 | |||
| 354fb672a3 | |||
| 73abf0740a | |||
| 81c82935ac | |||
| ac3ec56584 | |||
| 2cbe12d4d0 | |||
| 1475b0e986 | |||
| 8ceab0b9b1 | |||
| 1496b597d3 | |||
| 983519e629 | |||
| 58fa98ccd7 | |||
| a7bc5a22d4 | |||
| 23bbff0b56 | |||
| a47b58e2f4 | |||
| 8cf64ee93e | |||
| 630edeb3d6 | |||
| b1b216d4c8 | |||
| 55dd29b0ff | |||
| ea0568a38d | |||
| 298c86d0d4 | |||
| 859420809b | |||
| 0fe3a89768 | |||
| 5b892fbfc5 | |||
| fb4ed21a7f | |||
| 4209f3a9fe | |||
| ff2c80d5f1 | |||
| 3a07a08121 | |||
| 592da81a8c | |||
| def8c93711 | |||
| f73088da58 | |||
| 06a761ecbd | |||
| 97d737a30a | |||
| c7bffbf603 | |||
| 02c40ea612 | |||
| ce5ffb980a | |||
| d5ca98a982 | |||
| b0146782d6 | |||
| 5aaea8fc90 | |||
| 5d7752b657 | |||
| 9d94746479 | |||
| a87f2e1e07 | |||
| 7b7d814484 | |||
| 2daf67f31c | |||
| 62dd631379 | |||
| 2511192ecb | |||
| 22cb90de4a | |||
| 3951834df2 | |||
| 1ea0de106c | |||
| a9bf929ebd | |||
| 3336cd704b | |||
| a57a84f90e | |||
| ad546a8f1a | |||
| 1f447a8dfc | |||
| f574d896e9 | |||
| a561c2d6d4 | |||
| 79b8395be0 | |||
| d2c11a611f | |||
| b0903413b1 | |||
| 2135223490 | |||
| 6f86944a06 | |||
| 9861be2787 | |||
| c6fab363e6 | |||
| c32099705f | |||
| a1eac015dc | |||
| 64dfc65d51 | |||
| 70d5cbd0e0 | |||
| 5428d36d16 | |||
| ed4c15b7aa | |||
| 525b722c44 | |||
| 72bf210164 | |||
| 93e81c3bc8 | |||
| cb0f8e6411 | |||
| 848b2dc753 | |||
| 60f0396fe0 | |||
| 344e137a61 | |||
| 6977be7fb7 | |||
| a4721de8aa | |||
| 378ba122c8 | |||
| 287160622f | |||
| a5d788a2e9 | |||
| 7f651e1e45 | |||
| ad5bec114d | |||
| f71c62ad0f | |||
| 44f7dd2d8b | |||
| 15a9508a78 | |||
| bdea9a53e8 | |||
| 81f58667ff | |||
| 680a2e4c31 | |||
| 13993db542 | |||
| 59d595a4f0 | |||
| 4c562e15a4 | |||
| 5bcdf454ec | |||
| 4ccdda44ac | |||
| 5e46825474 | |||
| fcab5a9581 | |||
| e7f5e0ac23 | |||
| 9a3906831b | |||
| 195db43b9e | |||
| 332553ec1e | |||
| 646af3f652 | |||
| f24ac65fcb | |||
| 14312c93f9 | |||
| 4346d4155d | |||
| 7790ac5040 | |||
| 7b3ea0b59d | |||
| e2af710369 | |||
| 6b1b620f39 | |||
| 966524da16 | |||
| 83c999f298 | |||
| abff8a4f56 | |||
| 188575f3a2 | |||
| 9f4d8ca187 | |||
| 72cc8f3a5e | |||
| 915dcab597 | |||
| d82052b168 | |||
| dbb65bc71c | |||
| 2498bc595b | |||
| fee63d712f | |||
| 5965ea30e6 | |||
| 5c75495c03 | |||
| 98ae3c66cb | |||
| 942e15484a | |||
| 2349e66a97 | |||
| 5750e3224f | |||
| 558b4dd69e | |||
| 02ef49f67e | |||
| 7cdf8f1d03 | |||
| 7f894f784a | |||
| a770251151 | |||
| 7ccf743bb6 | |||
| 75bedfe06d | |||
| 4ea9885d44 | |||
| 0930def9ec | |||
| 6aa5af7640 | |||
| e74fe5edf7 | |||
| cbb5d0fcdf | |||
| d489bc40e4 | |||
| 499b2f183a | |||
| f57b926f66 | |||
| 733dca5aea | |||
| 1ab7e8ca87 | |||
| 53426f794c | |||
| f8189685e5 | |||
| e6bd200c85 | |||
| 4a4c33fd9d | |||
| dc44922b92 | |||
| 67896cfba7 | |||
| ee8663e6d9 | |||
| 652c05f052 | |||
| 193d129d9e | |||
| e995df4bc3 | |||
| f6a30c3626 | |||
| ee95420f8f | |||
| 2221cd0209 | |||
| 8368d831aa | |||
| 91a128b3ed | |||
| 83d8b63351 | |||
| 0881aa856f | |||
| 77abb02ada | |||
| 119c6cd0dc | |||
| cc649aa78e | |||
| 0b0c9fce0e | |||
| d845e2107f | |||
| 69992a1b39 | |||
| f351280f49 | |||
| c15065b044 | |||
| db41f16051 | |||
| e276de61be | |||
| 43c786ab18 | |||
| 1199f39d55 | |||
| 70aab86257 | |||
| e4b0af7a8e | |||
| a120d06f6b | |||
| e01ca37e45 | |||
| f5c3d898e2 | |||
| a5a3697b7c | |||
| 4be9874ba2 | |||
| 5647a2dc6a | |||
| a067a5b5f3 | |||
| 475b0ecfc1 | |||
| 2a9a605144 | |||
| eea7ca571b | |||
| 11c16610ee | |||
| 2f4124eb66 | |||
| db3e31d7ce | |||
| d917af21ab | |||
| 948c8c7cfd | |||
| dec8e1acf3 | |||
| 3c020526ce | |||
| b0da19b07f | |||
| 70bcd9b018 | |||
| d2604b609d | |||
| e960e0cacf | |||
| 77b114d434 | |||
| 2860092cef | |||
| c6119c4835 | |||
| 415d1927e0 | |||
| 3d1e21d242 | |||
| 7a75e1c3f4 | |||
| c3430e448a | |||
| ae49ea60d2 | |||
| 02dc0ce0c8 | |||
| 15697a18e8 | |||
| 481881e59d | |||
| 11a25b26a7 | |||
| bf87358f6c | |||
| 04a8242230 | |||
| ceb96c301c | |||
| 307df8fdc0 | |||
| c44875a2dd | |||
| 31f11990ca | |||
| b79b85856d | |||
| 6e861c6a19 | |||
| cf62296a51 | |||
| 387b132814 | |||
| de83de8624 | |||
| fe91b5a717 | |||
| a052506a5d | |||
| 6f2d6d0d69 | |||
| a8ae6025bd | |||
| 1b8332a609 | |||
| fe16ec7e57 | |||
| c5ce5e46bd | |||
| 99d66670bc | |||
| 072b42cac6 | |||
| acadb5b4c2 | |||
| 2e1de6c3af | |||
| 67de30908b | |||
| 9d398eff0e | |||
| 02b7312a00 | |||
| 08c35e84f3 | |||
| b92a5c1fc7 | |||
| 8997a587c5 | |||
| dc3d03d742 | |||
| 84df40715c | |||
| b639fb501a | |||
| 746ff47757 | |||
| 2d62db8118 | |||
| 7a2adcd9ba | |||
| 9ccf3ef0e8 | |||
| 0edab6d558 | |||
| 4b344f0cd8 | |||
| 17ddc4d5ba | |||
| 3e33860c47 | |||
| 0269a10833 | |||
| 7af3e9a334 | |||
| d666876be1 | |||
| 564fab7ec1 | |||
| 155c6c2a2a | |||
| d43cbe9344 | |||
| 57cc474c9f | |||
| 586603f8e1 | |||
| d57887d22a | |||
| 6183bcfc5f | |||
| 65f6113b4d | |||
| 8d88b89db1 | |||
| 62885e8963 | |||
| 9fc094a5da | |||
| f97383c17f | |||
| 4b892ec5e7 | |||
| 41035485db | |||
| 9696f4c917 | |||
| 55abf5f5ac | |||
| a1b2e41710 | |||
| dff4ab26e4 | |||
| 83b6e1cdf7 | |||
| 38dbaa15ea | |||
| 0e531b6061 | |||
| de94ef5537 | |||
| 6ef9d13877 | |||
| 1c7b94757d | |||
| 83486e0bef | |||
| 9787e8a53f | |||
| f59d6685ad | |||
| 8a986ef384 | |||
| 78f9f55e14 | |||
| c2e006f664 | |||
| 4b8e9737a5 | |||
| 722b09eaa4 | |||
| f5b4f5a1f2 | |||
| 4407d8da55 | |||
| 73b73527cd | |||
| 4df0e3a741 | |||
| 1a771e0172 | |||
| 4333c3c242 | |||
| b831c9ad57 | |||
| e5c08f7710 | |||
| 51c2968595 | |||
| 5993376322 | |||
| c4281622b9 | |||
| 9d26014de2 | |||
| f6c115d215 | |||
| 6bd102d778 | |||
| d3d6af5712 | |||
| 876093446f | |||
| 336f219f09 | |||
| 584251cbdc | |||
| a34995a788 | |||
| 998e5da227 | |||
| e04c15e367 | |||
| b81f69d407 | |||
| 0ac2064281 | |||
| 1d00bd244e | |||
| 8f5efc58c9 | |||
| a0c5ae1b5e | |||
| 99f48f9de1 | |||
| 1948b23f32 | |||
| c9eb572fc5 | |||
| 7d9895ff81 | |||
| d507210ef8 | |||
| db0a3d23d5 | |||
| f4f920f3cd | |||
| 75993ea276 | |||
| ee9bacb092 | |||
| b1e775c67b | |||
| d631e567aa | |||
| 9edf45be42 | |||
| 8b4b3c646a | |||
| 31bb0557d9 | |||
| 4593183cf9 | |||
| fdc45f1187 | |||
| b3d3c6d12c | |||
| f5f0794def | |||
| 55664fcca0 | |||
| afbf330f16 | |||
| 25aadf61bc | |||
| be47056467 | |||
| 8f623e0aea | |||
| 78025435e4 | |||
| 910455802e | |||
| b12914955c | |||
| dfe11eaf83 | |||
| 37fbe1a52b | |||
| af11bb2361 | |||
| 8da8697c1e | |||
| 944dc87531 | |||
| 5c4dd4644e | |||
| 9bbd172cfd | |||
| dbf9de77c3 | |||
| 8b790cd162 | |||
| 80219066e9 | |||
| 26fa5f098f | |||
| d75bb36131 | |||
| 30c5e8ca79 | |||
| c00e36fab6 | |||
| 80e60b9118 | |||
| e3a95a44bc | |||
| 86caf526f5 | |||
| e09f32b4b4 | |||
| 3803ab345d | |||
| 889b43136f | |||
| b517cf46af | |||
| ffd810fe00 | |||
| a1a0ed70a1 | |||
| 1554d9ede7 | |||
| 6f3190d83d | |||
| 3caaa32ec0 | |||
| 3facf3ccf5 | |||
| fad954e6e3 | |||
| c4809cb618 | |||
| a241423baf | |||
| 1e9421dd38 | |||
| b09ffda760 | |||
| aeff663a7c | |||
| fcd17fcd4c | |||
| feffb2711d | |||
| f76e5b592f | |||
| 71bdc900e3 | |||
| 96bbf5abc2 | |||
| df9424a228 | |||
| 3310acaf54 | |||
| e8722d462d | |||
| 14b2c13ed6 | |||
| 19af324096 | |||
| 0ca7b94e08 | |||
| 04ed310229 | |||
| 7ac33d521c | |||
| 08ada63da4 | |||
| cef3fd72fb | |||
| bca0b86e37 | |||
| bf973ff434 | |||
| a172d60b20 | |||
| 060e4395c2 | |||
| 03f1417454 | |||
| 6b787a4d1e | |||
| 095fac1de4 | |||
| 5ba9e353c7 | |||
| be117bfbb8 | |||
| c4f9bfe735 | |||
| cc96ef700f | |||
| cb41d8c94b | |||
| 8913b93703 | |||
| 174bf7b0ff | |||
| 7e44844cd5 | |||
| c0aa87af1b | |||
| f66f7a9290 | |||
| 378fa3499a | |||
| 01eeec498d | |||
| d422dfbddb | |||
| bd4a48d5a8 | |||
| aa91ccf68b | |||
| bda43565e1 | |||
| 8c9bae3ad1 | |||
| 87fc3bc073 | |||
| 9c101ffa86 | |||
| 5649d9e3c7 | |||
| 5ee96b6d04 | |||
| 6a7854919e | |||
| a6f6c0409b | |||
| 41446c980c | |||
| 7bed44a96e | |||
| 3f7bd2f586 | |||
| 2ba8466e0a | |||
| f80dcce5ed | |||
| f4aed3e9a4 | |||
| 5c4ef66a08 | |||
| 29d2564efd | |||
| 380fa9c7c8 | |||
| e55c6e09f8 | |||
| 355b58447a | |||
| fe35675eaa | |||
| ae9db8652a | |||
| ac3e52596c | |||
| 17bffb6739 | |||
| 8839771794 | |||
| 95ee23521b | |||
| 38672ad1ca | |||
| 79cee1cab6 | |||
| 3aece57b6e | |||
| 78e66149f8 | |||
| 6805077934 | |||
| 688178ba5f | |||
| cc13917903 | |||
| 65ebfb0dac | |||
| 3954d89329 | |||
| 41ebb41bd8 | |||
| 545800454b | |||
| edad7eba9e | |||
| bb8188afe6 | |||
| 305961e8dc | |||
| 06972ec3ad | |||
| 345dfde6fb | |||
| 1326d3820f | |||
| 9437f88ce2 | |||
| 67b5814108 | |||
| 5a49c26d9d |
@@ -10,7 +10,6 @@ add_library(vosk
|
||||
src/recognizer.cc
|
||||
src/spk_model.cc
|
||||
src/vosk_api.cc
|
||||
src/postprocessor.cc
|
||||
)
|
||||
|
||||
find_package(kaldi REQUIRED)
|
||||
|
||||
@@ -1,27 +0,0 @@
|
||||
# Vosk Speech Recognition Toolkit
|
||||
|
||||
Vosk is an offline open source speech recognition toolkit. It enables
|
||||
speech recognition for 20+ languages and dialects - English, Indian
|
||||
English, German, French, Spanish, Portuguese, Chinese, Russian, Turkish,
|
||||
Vietnamese, Italian, Dutch, Catalan, Arabic, Greek, Farsi, Filipino,
|
||||
Ukrainian, Kazakh, Swedish, Japanese, Esperanto, Hindi, Czech, Polish.
|
||||
More to come.
|
||||
|
||||
Vosk models are small (50 Mb) but provide continuous large vocabulary
|
||||
transcription, zero-latency response with streaming API, reconfigurable
|
||||
vocabulary and speaker identification.
|
||||
|
||||
Speech recognition bindings implemented for various programming languages
|
||||
like Python, Java, Node.JS, C#, C++, Rust, Go and others.
|
||||
|
||||
Vosk supplies speech recognition for chatbots, smart home appliances,
|
||||
virtual assistants. It can also create subtitles for movies,
|
||||
transcription for lectures and interviews.
|
||||
|
||||
Vosk scales from small devices like Raspberry Pi or Android smartphone to
|
||||
big clusters.
|
||||
|
||||
# Documentation
|
||||
|
||||
For installation instructions, examples and documentation visit [Vosk
|
||||
Website](https://alphacephei.com/vosk).
|
||||
@@ -1,21 +1,27 @@
|
||||
<!-- WEHUB_ZH_README -->
|
||||
> [!NOTE]
|
||||
> 本文档由 WeHub 基于上游 README 翻译整理,属于社区翻译,非官方中文文档。
|
||||
> [English](./README.en.md) · [原始项目](https://github.com/alphacep/vosk-api) · [上游 README](https://github.com/alphacep/vosk-api/blob/HEAD/README.md)
|
||||
> 原作者、版权与许可证归属以原始项目及本仓库 LICENSE 文件为准。
|
||||
|
||||
# Vosk Speech Recognition Toolkit
|
||||
|
||||
Vosk 是一款离线开源语音识别工具包。它支持 20 多种语言及方言的语音识别——英语、印度英语、德语、法语、西班牙语、葡萄牙语、中文、俄语、土耳其语、越南语、意大利语、荷兰语、加泰罗尼亚语、阿拉伯语、希腊语、波斯语、菲律宾语、乌克兰语、哈萨克语、瑞典语、日语、世界语、印地语、捷克语、波兰语。更多语言即将推出。
|
||||
Vosk is an offline open source speech recognition toolkit. It enables
|
||||
speech recognition for 20+ languages and dialects - English, Indian
|
||||
English, German, French, Spanish, Portuguese, Chinese, Russian, Turkish,
|
||||
Vietnamese, Italian, Dutch, Catalan, Arabic, Greek, Farsi, Filipino,
|
||||
Ukrainian, Kazakh, Swedish, Japanese, Esperanto, Hindi, Czech, Polish.
|
||||
More to come.
|
||||
|
||||
Vosk 模型体积小(50 Mb),但可提供连续大词汇量转写、通过流式 API(streaming API)实现零延迟响应、可重新配置的词汇表以及说话人识别。
|
||||
Vosk models are small (50 Mb) but provide continuous large vocabulary
|
||||
transcription, zero-latency response with streaming API, reconfigurable
|
||||
vocabulary and speaker identification.
|
||||
|
||||
已为 Python、Java、Node.JS、C#、C++、Rust、Go 等多种编程语言实现语音识别绑定。
|
||||
Speech recognition bindings implemented for various programming languages
|
||||
like Python, Java, Node.JS, C#, C++, Rust, Go and others.
|
||||
|
||||
Vosk 为聊天机器人、智能家居设备、虚拟助手提供语音识别,还可为电影生成字幕,为讲座和访谈提供转写。
|
||||
Vosk supplies speech recognition for chatbots, smart home appliances,
|
||||
virtual assistants. It can also create subtitles for movies,
|
||||
transcription for lectures and interviews.
|
||||
|
||||
Vosk 的部署规模可从小型设备(如 Raspberry Pi 或 Android 智能手机)扩展到大型集群。
|
||||
Vosk scales from small devices like Raspberry Pi or Android smartphone to
|
||||
big clusters.
|
||||
|
||||
# 文档
|
||||
# Documentation
|
||||
|
||||
有关安装说明、示例和文档,请访问 [Vosk 网站](https://alphacephei.com/vosk).
|
||||
For installation instructions, examples and documentation visit [Vosk
|
||||
Website](https://alphacephei.com/vosk).
|
||||
|
||||
@@ -1,7 +0,0 @@
|
||||
# WeHub 来源说明
|
||||
|
||||
- 原始项目:`alphacep/vosk-api`
|
||||
- 原始仓库:https://github.com/alphacep/vosk-api
|
||||
- 导入方式:上游默认分支的最新快照
|
||||
- 原作者、版权和许可证信息以原始仓库及本仓库 LICENSE 为准
|
||||
- 本文件仅用于记录来源,不代表 WeHub 是原项目作者
|
||||
+38
-26
@@ -4,50 +4,62 @@ buildscript {
|
||||
mavenCentral()
|
||||
}
|
||||
dependencies {
|
||||
classpath 'com.android.tools.build:gradle:8.13.0'
|
||||
classpath 'com.vanniktech:gradle-maven-publish-plugin:0.34.0'
|
||||
classpath 'com.android.tools.build:gradle:4.2.0'
|
||||
classpath 'com.vanniktech:gradle-maven-publish-plugin:0.18.0'
|
||||
}
|
||||
}
|
||||
|
||||
allprojects {
|
||||
version = '0.3.75'
|
||||
version = '0.3.45'
|
||||
}
|
||||
|
||||
subprojects {
|
||||
|
||||
apply plugin: 'com.android.library'
|
||||
apply plugin: 'maven-publish'
|
||||
apply plugin: 'com.vanniktech.maven.publish'
|
||||
|
||||
plugins.withId('com.vanniktech.maven.publish') {
|
||||
mavenPublish {
|
||||
group = 'com.alphacephei'
|
||||
version = version
|
||||
sonatypeHost = 's01'
|
||||
androidVariantToPublish = 'release'
|
||||
}
|
||||
}
|
||||
|
||||
repositories {
|
||||
google()
|
||||
mavenCentral()
|
||||
}
|
||||
|
||||
mavenPublishing {
|
||||
publishToMavenCentral()
|
||||
signAllPublications()
|
||||
}
|
||||
|
||||
mavenPublishing {
|
||||
pom {
|
||||
url = 'http://www.alphacephei.com.com/vosk/'
|
||||
licenses {
|
||||
license {
|
||||
name = 'The Apache License, Version 2.0'
|
||||
url = 'http://www.apache.org/licenses/LICENSE-2.0.txt'
|
||||
publishing {
|
||||
publications {
|
||||
aar(MavenPublication) {
|
||||
groupId 'com.alphacephei'
|
||||
version version
|
||||
pom {
|
||||
url = 'http://www.alphacephei.com.com/vosk/'
|
||||
licenses {
|
||||
license {
|
||||
name = 'The Apache License, Version 2.0'
|
||||
url = 'http://www.apache.org/licenses/LICENSE-2.0.txt'
|
||||
}
|
||||
}
|
||||
developers {
|
||||
developer {
|
||||
id = 'com.alphacephei'
|
||||
name = 'Alpha Cephei Inc'
|
||||
email = 'contact@alphacephei.com'
|
||||
}
|
||||
}
|
||||
scm {
|
||||
connection = 'scm:git:git://github.com/alphacep/vosk-api.git'
|
||||
url = 'https://github.com/alphacep/vosk-api/'
|
||||
}
|
||||
}
|
||||
}
|
||||
developers {
|
||||
developer {
|
||||
id = 'com.alphacephei'
|
||||
name = 'Alpha Cephei Inc'
|
||||
email = 'contact@alphacephei.com'
|
||||
}
|
||||
}
|
||||
scm {
|
||||
connection = 'scm:git:git://github.com/alphacep/vosk-api.git'
|
||||
url = 'https://github.com/alphacep/vosk-api/'
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -29,7 +29,7 @@ set -x
|
||||
OS_NAME=`echo $(uname -s) | tr '[:upper:]' '[:lower:]'`
|
||||
ANDROID_TOOLCHAIN_PATH=$ANDROID_NDK_HOME/toolchains/llvm/prebuilt/${OS_NAME}-x86_64
|
||||
WORKDIR_BASE=`pwd`/build
|
||||
PATH=$ANDROID_TOOLCHAIN_PATH/bin:$PATH
|
||||
PATH=$ANDROID_NDK_HOME/toolchains/llvm/prebuilt/${OS_NAME}-x86_64/bin:$PATH
|
||||
OPENFST_VERSION=1.8.0
|
||||
|
||||
for arch in armeabi-v7a arm64-v8a x86_64 x86; do
|
||||
@@ -45,7 +45,6 @@ case $arch in
|
||||
CC=armv7a-linux-androideabi21-clang
|
||||
CXX=armv7a-linux-androideabi21-clang++
|
||||
ARCHFLAGS="-mfloat-abi=softfp -mfpu=neon"
|
||||
PAGESIZE_LDFLAGS=""
|
||||
;;
|
||||
arm64-v8a)
|
||||
BLAS_ARCH=ARMV8
|
||||
@@ -55,8 +54,6 @@ case $arch in
|
||||
CC=aarch64-linux-android21-clang
|
||||
CXX=aarch64-linux-android21-clang++
|
||||
ARCHFLAGS=""
|
||||
# Ensure compatibility with 16KiB page size devices
|
||||
PAGESIZE_LDFLAGS="-Wl,-z,common-page-size=4096 -Wl,-z,max-page-size=16384"
|
||||
;;
|
||||
x86_64)
|
||||
BLAS_ARCH=ATOM
|
||||
@@ -66,7 +63,6 @@ case $arch in
|
||||
CC=x86_64-linux-android21-clang
|
||||
CXX=x86_64-linux-android21-clang++
|
||||
ARCHFLAGS=""
|
||||
PAGESIZE_LDFLAGS=""
|
||||
;;
|
||||
x86)
|
||||
BLAS_ARCH=ATOM
|
||||
@@ -76,7 +72,6 @@ case $arch in
|
||||
CC=i686-linux-android21-clang
|
||||
CXX=i686-linux-android21-clang++
|
||||
ARCHFLAGS=""
|
||||
PAGESIZE_LDFLAGS=""
|
||||
;;
|
||||
esac
|
||||
|
||||
@@ -84,16 +79,16 @@ mkdir -p $WORKDIR/local/lib
|
||||
|
||||
# openblas first
|
||||
cd $WORKDIR
|
||||
git clone -b v0.3.20 --single-branch https://github.com/xianyi/OpenBLAS
|
||||
make -C OpenBLAS TARGET=$BLAS_ARCH ONLY_CBLAS=1 AR=$AR CC=$CC HOSTCC=gcc ARM_SOFTFP_ABI=1 USE_THREAD=0 NUM_THREADS=1 -j 8
|
||||
git clone -b v0.3.13 --single-branch https://github.com/xianyi/OpenBLAS
|
||||
make -C OpenBLAS TARGET=$BLAS_ARCH ONLY_CBLAS=1 AR=$AR CC=$CC HOSTCC=gcc ARM_SOFTFP_ABI=1 USE_THREAD=0 NUM_THREADS=1 -j4
|
||||
make -C OpenBLAS install PREFIX=$WORKDIR/local
|
||||
|
||||
# CLAPACK
|
||||
cd $WORKDIR
|
||||
git clone -b v3.2.1 --single-branch https://github.com/alphacep/clapack
|
||||
mkdir -p clapack/BUILD && cd clapack/BUILD
|
||||
cmake -DCMAKE_C_FLAGS="$ARCHFLAGS" -DCMAKE_C_COMPILER_TARGET=$HOST \
|
||||
-DCMAKE_C_COMPILER=$CC -DCMAKE_SYSTEM_NAME=Generic -DCMAKE_AR=$ANDROID_TOOLCHAIN_PATH/bin/$AR \
|
||||
cmake -DCMAKE_C_FLAGS=$ARCHFLAGS -DCMAKE_C_COMPILER_TARGET=$HOST \
|
||||
-DCMAKE_C_COMPILER=$CC -DCMAKE_SYSTEM_NAME=Generic -DCMAKE_AR=$ANDROID_NDK_HOME/toolchains/llvm/prebuilt/${OS_NAME}-x86_64/bin/$AR \
|
||||
-DCMAKE_TRY_COMPILE_TARGET_TYPE=STATIC_LIBRARY \
|
||||
-DCMAKE_CROSSCOMPILING=True ..
|
||||
make -j 8 -C F2CLIBS/libf2c
|
||||
@@ -123,7 +118,7 @@ CXX=$CXX AR=$AR RANLIB=$RANLIB CXXFLAGS="$ARCHFLAGS -O3 -DFST_NO_DYNAMIC_LINKING
|
||||
--fst-root=${WORKDIR}/local --fst-version=${OPENFST_VERSION}
|
||||
make -j 8 depend
|
||||
cd $WORKDIR/kaldi/src
|
||||
make -j 8 online2 rnnlm
|
||||
make -j 8 online2 lm rnnlm
|
||||
|
||||
# Vosk-api
|
||||
cd $WORKDIR
|
||||
@@ -134,7 +129,7 @@ make -j 8 -C ${WORKDIR_BASE}/../../../src \
|
||||
OPENFST_ROOT=${WORKDIR}/local \
|
||||
OPENBLAS_ROOT=${WORKDIR}/local \
|
||||
CXX=$CXX \
|
||||
EXTRA_LDFLAGS="-llog -static-libstdc++ -Wl,-soname,libvosk.so ${PAGESIZE_LDFLAGS}"
|
||||
EXTRA_LDFLAGS="-llog -static-libstdc++ -Wl,-soname,libvosk.so"
|
||||
cp $WORKDIR/vosk/libvosk.so $WORKDIR/../../src/main/jniLibs/$arch/libvosk.so
|
||||
|
||||
done
|
||||
|
||||
+27
-11
@@ -3,15 +3,14 @@ def pomName = "Vosk Android"
|
||||
def pomDescription = "Vosk speech recognition library for Android"
|
||||
|
||||
android {
|
||||
namespace 'org.vosk'
|
||||
compileSdkVersion 36
|
||||
compileSdkVersion 29
|
||||
defaultConfig {
|
||||
minSdkVersion 21
|
||||
targetSdkVersion 36
|
||||
versionCode 10
|
||||
targetSdkVersion 29
|
||||
versionCode 6
|
||||
versionName = version
|
||||
archivesBaseName = archiveName
|
||||
ndkVersion = "28.2.13676358"
|
||||
ndkVersion = "22.1.7171670"
|
||||
}
|
||||
compileOptions {
|
||||
sourceCompatibility JavaVersion.VERSION_1_8
|
||||
@@ -25,15 +24,32 @@ task buildVosk(type: Exec) {
|
||||
}
|
||||
|
||||
dependencies {
|
||||
api 'net.java.dev.jna:jna:5.18.1@aar'
|
||||
api 'net.java.dev.jna:jna:4.4.0@aar'
|
||||
}
|
||||
|
||||
//preBuild.dependsOn buildVosk
|
||||
|
||||
mavenPublishing {
|
||||
coordinates("com.alphacephei", archiveName, version)
|
||||
pom {
|
||||
name = pomName
|
||||
description = pomDescription
|
||||
publishing {
|
||||
publications {
|
||||
aar(MavenPublication) {
|
||||
artifactId = archiveName
|
||||
artifact("$buildDir/outputs/aar/$archiveName-release.aar")
|
||||
pom {
|
||||
name = pomName
|
||||
description = pomDescription
|
||||
}
|
||||
//generate pom nodes for dependencies
|
||||
pom.withXml {
|
||||
def dependenciesNode = asNode().appendNode('dependencies')
|
||||
configurations.implementation.allDependencies.each { dependency ->
|
||||
if (dependency.name != 'unspecified') {
|
||||
def dependencyNode = dependenciesNode.appendNode('dependency')
|
||||
dependencyNode.appendNode('groupId', dependency.group)
|
||||
dependencyNode.appendNode('artifactId', dependency.name)
|
||||
dependencyNode.appendNode('version', dependency.version)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,3 +1,3 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<manifest>
|
||||
<manifest xmlns:android="http://schemas.android.com/apk/res/android" package="org.vosk">
|
||||
</manifest>
|
||||
@@ -52,30 +52,10 @@ public class LibVosk {
|
||||
|
||||
public static native String vosk_recognizer_partial_result(Pointer recognizer);
|
||||
|
||||
public static native void vosk_recognizer_set_grm(Pointer recognizer, String grammar);
|
||||
|
||||
public static native void vosk_recognizer_reset(Pointer recognizer);
|
||||
|
||||
public static native void vosk_recognizer_set_endpointer_mode(Pointer recognizer, int mode);
|
||||
|
||||
public static native void vosk_recognizer_set_endpointer_delays(Pointer recognizer, float t_start_max, float t_end, float t_max);
|
||||
|
||||
public static native void vosk_recognizer_free(Pointer recognizer);
|
||||
|
||||
public static native Pointer vosk_text_processor_new(String verbalizer, String tagger);
|
||||
|
||||
public static native void vosk_text_processor_free(Pointer processor);
|
||||
|
||||
public static native String vosk_text_processor_itn(Pointer processor, String input);
|
||||
|
||||
/**
|
||||
* Set log level for Kaldi messages.
|
||||
*
|
||||
* @param loglevel the level
|
||||
* 0 - default value to print info and error messages but no debug
|
||||
* less than 0 - don't print info messages
|
||||
* greater than 0 - more verbose mode
|
||||
*/
|
||||
public static void setLogLevel(LogLevel loglevel) {
|
||||
vosk_set_log_level(loglevel.getValue());
|
||||
}
|
||||
|
||||
@@ -1,18 +1,13 @@
|
||||
package org.vosk;
|
||||
|
||||
import java.io.IOException;
|
||||
import com.sun.jna.PointerType;
|
||||
|
||||
public class Model extends PointerType implements AutoCloseable {
|
||||
public Model() {
|
||||
}
|
||||
|
||||
public Model(String path) throws IOException {
|
||||
public Model(String path) {
|
||||
super(LibVosk.vosk_model_new(path));
|
||||
|
||||
if (getPointer() == null) {
|
||||
throw new IOException("Failed to create a model");
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
|
||||
@@ -1,163 +1,36 @@
|
||||
package org.vosk;
|
||||
|
||||
import com.sun.jna.PointerType;
|
||||
import java.io.IOException;
|
||||
|
||||
public class Recognizer extends PointerType implements AutoCloseable {
|
||||
/**
|
||||
* Creates the recognizer object.
|
||||
*
|
||||
* The recognizers process the speech and return text using shared model data
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The sample rate of the audio you are going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @throws IOException if the recognizer could not be created
|
||||
*/
|
||||
public Recognizer(Model model, float sampleRate) throws IOException {
|
||||
public Recognizer(Model model, float sampleRate) {
|
||||
super(LibVosk.vosk_recognizer_new(model, sampleRate));
|
||||
|
||||
if (getPointer() == null) {
|
||||
throw new IOException("Failed to create a recognizer");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates the recognizer object with speaker recognition.
|
||||
*
|
||||
* With the speaker recognition mode the recognizer not just recognize
|
||||
* text but also return speaker vectors one can use for speaker identification
|
||||
*
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The sample rate of the audio you are going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @param spkModel speaker model for speaker identification
|
||||
* @throws IOException if the recognizer could not be created
|
||||
*/
|
||||
public Recognizer(Model model, float sampleRate, SpeakerModel spkModel) throws IOException {
|
||||
public Recognizer(Model model, float sampleRate, SpeakerModel spkModel) {
|
||||
super(LibVosk.vosk_recognizer_new_spk(model.getPointer(), sampleRate, spkModel.getPointer()));
|
||||
|
||||
if (getPointer() == null) {
|
||||
throw new IOException("Failed to create a recognizer");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates the recognizer object with the phrase list.
|
||||
*
|
||||
* Sometimes when you want to improve recognition accuracy and when you don't need
|
||||
* to recognize large vocabulary you can specify a list of phrases to recognize. This
|
||||
* will improve recognizer speed and accuracy but might return [unk] if user said
|
||||
* something different.
|
||||
*
|
||||
* Only recognizers with lookahead models support this type of quick configuration.
|
||||
* Precompiled HCLG graph models are not supported.
|
||||
*
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The sample rate of the audio you are going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @param grammar The string with the list of phrases to recognize as JSON array of strings,
|
||||
* for example "["one two three four five", "[unk]"]".
|
||||
* @throws IOException if the recognizer could not be created
|
||||
*/
|
||||
public Recognizer(Model model, float sampleRate, String grammar) throws IOException {
|
||||
public Recognizer(Model model, float sampleRate, String grammar) {
|
||||
super(LibVosk.vosk_recognizer_new_grm(model.getPointer(), sampleRate, grammar));
|
||||
|
||||
if (getPointer() == null) {
|
||||
throw new IOException("Failed to create a recognizer");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Configures recognizer to output n-best results.
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "alternatives": [
|
||||
* { "text": "one two three four five", "confidence": 0.97 },
|
||||
* { "text": "one two three for five", "confidence": 0.03 },
|
||||
* ]
|
||||
* }
|
||||
* </pre>
|
||||
*
|
||||
* @param maxAlternatives - maximum alternatives to return from recognition results
|
||||
*/
|
||||
public void setMaxAlternatives(int maxAlternatives) {
|
||||
LibVosk.vosk_recognizer_set_max_alternatives(this.getPointer(), maxAlternatives);
|
||||
}
|
||||
|
||||
/** Enables words with times in the output
|
||||
*
|
||||
* <pre>
|
||||
* "result" : [{
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.110000,
|
||||
* "start" : 0.870000,
|
||||
* "word" : "what"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.530000,
|
||||
* "start" : 1.110000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.950000,
|
||||
* "start" : 1.530000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 2.340000,
|
||||
* "start" : 1.950000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 2.610000,
|
||||
* "start" : 2.340000,
|
||||
* "word" : "one"
|
||||
* }],
|
||||
* </pre>
|
||||
*
|
||||
* @param words - boolean value
|
||||
*/
|
||||
public void setWords(boolean words) {
|
||||
LibVosk.vosk_recognizer_set_words(this.getPointer(), words);
|
||||
}
|
||||
|
||||
/**
|
||||
* Like above return words and confidences in partial results.
|
||||
*
|
||||
* @param partial_words - boolean value
|
||||
*/
|
||||
public void setPartialWords(boolean partial_words) {
|
||||
LibVosk.vosk_recognizer_set_partial_words(this.getPointer(), partial_words);
|
||||
}
|
||||
|
||||
/**
|
||||
* Adds speaker model to already initialized recognizer.
|
||||
*
|
||||
* Can add speaker recognition model to already created recognizer.
|
||||
* Helps to initialize speaker recognition for grammar-based recognizer.
|
||||
*
|
||||
* @param spkModel Speaker recognition model
|
||||
*/
|
||||
public void setSpeakerModel(SpeakerModel spkModel) {
|
||||
LibVosk.vosk_recognizer_set_spk_model(this.getPointer(), spkModel.getPointer());
|
||||
}
|
||||
|
||||
/**
|
||||
* Accept and process new chunk of voice data.
|
||||
*
|
||||
* @param data - audio data in PCM 16-bit mono format
|
||||
* @param len - length of the audio data
|
||||
* @return 1 if silence is occurred and you can retrieve a new utterance with result method
|
||||
* 0 if decoding continues
|
||||
* -1 if exception occurred
|
||||
*/
|
||||
public boolean acceptWaveForm(byte[] data, int len) {
|
||||
return LibVosk.vosk_recognizer_accept_waveform(this.getPointer(), data, len);
|
||||
}
|
||||
@@ -170,104 +43,22 @@ public class Recognizer extends PointerType implements AutoCloseable {
|
||||
return LibVosk.vosk_recognizer_accept_waveform_f(this.getPointer(), data, len);
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns speech recognition result
|
||||
*
|
||||
* @return the result in JSON format which contains decoded line, decoded
|
||||
* words, times in seconds and confidences. You can parse this result
|
||||
* with any json parser
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "text" : "what zero zero zero one"
|
||||
* }
|
||||
* </pre>
|
||||
*
|
||||
* If alternatives enabled it returns result with alternatives, see also #setMaxAlternatives().
|
||||
*
|
||||
* If word times enabled returns word time, see also #setWordTimes().
|
||||
*/
|
||||
public String getResult() {
|
||||
return LibVosk.vosk_recognizer_result(this.getPointer());
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns partial speech recognition.
|
||||
*
|
||||
* @return partial speech recognition text which is not yet finalized.
|
||||
* result may change as recognizer process more data.
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "partial" : "cyril one eight zero"
|
||||
* }
|
||||
* </pre>
|
||||
*/
|
||||
public String getPartialResult() {
|
||||
return LibVosk.vosk_recognizer_partial_result(this.getPointer());
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns speech recognition result. Same as result, but doesn't wait for silence.
|
||||
* You usually call it in the end of the stream to get final bits of audio. It
|
||||
* flushes the feature pipeline, so all remaining audio chunks got processed.
|
||||
*
|
||||
* @return speech result in JSON format.
|
||||
*/
|
||||
public String getFinalResult() {
|
||||
return LibVosk.vosk_recognizer_final_result(this.getPointer());
|
||||
}
|
||||
|
||||
/**
|
||||
* Reconfigures recognizer to use grammar.
|
||||
*
|
||||
* @param grammar Set of phrases in JSON array of strings or "[]" to use default model graph.
|
||||
* @see #Recognizer(Model, float, String)
|
||||
*/
|
||||
public void setGrammar(String grammar) {
|
||||
LibVosk.vosk_recognizer_set_grm(this.getPointer(), grammar);
|
||||
}
|
||||
|
||||
/**
|
||||
* Resets the recognizer.
|
||||
* Resets current results so the recognition can continue from scratch.
|
||||
*/
|
||||
public void reset() {
|
||||
LibVosk.vosk_recognizer_reset(this.getPointer());
|
||||
}
|
||||
|
||||
/**
|
||||
* Endpointer delay mode
|
||||
*/
|
||||
public class EndpointerMode {
|
||||
public static final int DEFAULT = 0;
|
||||
public static final int SHORT = 1;
|
||||
public static final int LONG = 2;
|
||||
public static final int VERY_LONG = 3;
|
||||
}
|
||||
|
||||
/**
|
||||
* Configures endpointer mode for recognizer
|
||||
*/
|
||||
public void setEndpointerMode(int mode) {
|
||||
LibVosk.vosk_recognizer_set_endpointer_mode(this.getPointer(), mode);
|
||||
}
|
||||
|
||||
/**
|
||||
* Set endpointer delays
|
||||
*
|
||||
* @param t_start_max timeout for stopping recognition in case of initial silence (usually around 5.0)
|
||||
* @param t_end timeout for stopping recognition in milliseconds after we recognized something (usually around 0.5 - 1.0)
|
||||
* @param t_max timeout for forcing utterance end in milliseconds (usually around 20-30)
|
||||
**/
|
||||
public void setEndpointerDelays(float t_start_max, float t_end, float t_max) {
|
||||
LibVosk.vosk_recognizer_set_endpointer_delays(this.getPointer(), t_start_max, t_end, t_max);
|
||||
}
|
||||
|
||||
/**
|
||||
* Releases recognizer object.
|
||||
* Underlying model is also unreferenced and if needed, released.
|
||||
*/
|
||||
@Override
|
||||
public void close() {
|
||||
LibVosk.vosk_recognizer_free(this.getPointer());
|
||||
|
||||
@@ -1,36 +1,13 @@
|
||||
package org.vosk;
|
||||
|
||||
import com.sun.jna.PointerType;
|
||||
import java.io.IOException;
|
||||
|
||||
/**
|
||||
* Helps to initialize speaker recognition for grammar-based recognizer.
|
||||
*/
|
||||
public class SpeakerModel extends PointerType implements AutoCloseable {
|
||||
public SpeakerModel() {
|
||||
}
|
||||
|
||||
/**
|
||||
* Loads speaker model data from the file.
|
||||
*
|
||||
* The path must contain:
|
||||
* - a config file: mfcc.conf
|
||||
* - kaldi nnet: final.ext.raw
|
||||
* - mean.vec
|
||||
* - transform.mat
|
||||
*
|
||||
* @param path the path of the model on the filesystem
|
||||
* @throws IOException if the model could not be created
|
||||
*
|
||||
* @see <a href="http://kaldi-asr.org/doc/structkaldi_1_1MfccOptions.html">Kaldi MfccOptions</a>
|
||||
* @see <a href="http://kaldi-asr.org/doc/classkaldi_1_1nnet3_1_1Nnet.html">Kaldi Nnet</a>
|
||||
*/
|
||||
public SpeakerModel(String path) throws IOException {
|
||||
public SpeakerModel(String path) {
|
||||
super(LibVosk.vosk_spk_model_new(path));
|
||||
|
||||
if (getPointer() == null) {
|
||||
throw new IOException("Failed to create a speaker model");
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
|
||||
@@ -1,21 +0,0 @@
|
||||
package org.vosk;
|
||||
|
||||
import com.sun.jna.PointerType;
|
||||
|
||||
public class TextProcessor extends PointerType implements AutoCloseable {
|
||||
public TextProcessor() {
|
||||
}
|
||||
|
||||
public TextProcessor(String verbalizer, String tagger) {
|
||||
super(LibVosk.vosk_text_processor_new(verbalizer, tagger));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
LibVosk.vosk_text_processor_free(this.getPointer());
|
||||
}
|
||||
|
||||
public String itn(String input) {
|
||||
return LibVosk.vosk_text_processor_itn(this.getPointer(), input);
|
||||
}
|
||||
}
|
||||
@@ -19,9 +19,9 @@ import android.media.AudioRecord;
|
||||
import android.media.MediaRecorder.AudioSource;
|
||||
import android.os.Handler;
|
||||
import android.os.Looper;
|
||||
import android.annotation.SuppressLint;
|
||||
|
||||
import org.vosk.Recognizer;
|
||||
|
||||
import java.io.IOException;
|
||||
|
||||
/**
|
||||
@@ -48,7 +48,6 @@ public class SpeechService {
|
||||
*
|
||||
* @throws IOException thrown if audio recorder can not be created for some reason.
|
||||
*/
|
||||
@SuppressLint("MissingPermission")
|
||||
public SpeechService(Recognizer recognizer, float sampleRate) throws IOException {
|
||||
this.recognizer = recognizer;
|
||||
this.sampleRate = (int) sampleRate;
|
||||
@@ -66,51 +65,6 @@ public class SpeechService {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates speech service with a caller-supplied {@link AudioRecord}.
|
||||
* <p>
|
||||
* Use this when you need to control the audio input device - for example,
|
||||
* to pin recording to the built-in microphone when an external USB device
|
||||
* without a microphone is present:
|
||||
* <pre>
|
||||
* AudioRecord recorder = new AudioRecord.Builder()
|
||||
* .setAudioSource(MediaRecorder.AudioSource.VOICE_RECOGNITION)
|
||||
* .setAudioFormat(format)
|
||||
* .build();
|
||||
* if (Build.VERSION.SDK_INT >= 28) {
|
||||
* AudioManager am = (AudioManager) context.getSystemService(Context.AUDIO_SERVICE);
|
||||
* for (AudioDeviceInfo d : am.getDevices(AudioManager.GET_DEVICES_INPUTS)) {
|
||||
* if (d.getType() == AudioDeviceInfo.TYPE_BUILTIN_MIC) {
|
||||
* recorder.setPreferredDevice(d);
|
||||
* break;
|
||||
* }
|
||||
* }
|
||||
* }
|
||||
* SpeechService service = new SpeechService(recognizer, 16000f, recorder);
|
||||
* </pre>
|
||||
* <p>
|
||||
* The caller retains ownership of {@code recorder}: if this constructor
|
||||
* throws, the recorder is <em>not</em> released. Call
|
||||
* {@link AudioRecord#release()} yourself in that case.
|
||||
*
|
||||
* @param recognizer the Vosk recognizer
|
||||
* @param sampleRate sample rate in Hz; must match {@code recorder}'s configuration
|
||||
* @param recorder a fully-initialised {@link AudioRecord}
|
||||
* @throws IOException if {@code recorder} is in STATE_UNINITIALIZED
|
||||
*/
|
||||
public SpeechService(Recognizer recognizer, float sampleRate, AudioRecord recorder)
|
||||
throws IOException {
|
||||
this.recognizer = recognizer;
|
||||
this.sampleRate = (int) sampleRate;
|
||||
this.recorder = recorder;
|
||||
|
||||
bufferSize = Math.round(this.sampleRate * BUFFER_SIZE_SECONDS);
|
||||
|
||||
if (recorder.getState() == AudioRecord.STATE_UNINITIALIZED) {
|
||||
throw new IOException(
|
||||
"Failed to initialize recorder. Microphone might be already in use.");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Starts recognition. Does nothing if recognition is active.
|
||||
@@ -182,19 +136,6 @@ public class SpeechService {
|
||||
return stopRecognizerThread();
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Returns the audio session ID of the underlying {@link AudioRecord}.
|
||||
* <p>
|
||||
* The session ID can be used to attach audio effects such as
|
||||
* {@link android.media.audiofx.NoiseSuppressor} to the recording session.
|
||||
*
|
||||
* @return audio session ID, or {@link AudioRecord#ERROR} if unavailable
|
||||
*/
|
||||
public int getAudioSessionId() {
|
||||
return recorder.getAudioSessionId();
|
||||
}
|
||||
|
||||
/**
|
||||
* Shutdown the recognizer and release the recorder
|
||||
*/
|
||||
|
||||
@@ -3,12 +3,11 @@ def pomName = "Vosk English Model"
|
||||
def pomDescription = "Small English model for Android"
|
||||
|
||||
android {
|
||||
namespace "org.vosk"
|
||||
compileSdkVersion 36
|
||||
compileSdkVersion 29
|
||||
defaultConfig {
|
||||
minSdkVersion 21
|
||||
targetSdkVersion 36
|
||||
versionCode 10
|
||||
targetSdkVersion 29
|
||||
versionCode 6
|
||||
versionName = version
|
||||
archivesBaseName = archiveName
|
||||
}
|
||||
@@ -34,10 +33,15 @@ tasks.register('genUUID') {
|
||||
|
||||
preBuild.dependsOn(genUUID)
|
||||
|
||||
mavenPublishing {
|
||||
coordinates("com.alphacephei", archiveName, version)
|
||||
pom {
|
||||
name = pomName
|
||||
description = pomDescription
|
||||
publishing {
|
||||
publications {
|
||||
aar(MavenPublication) {
|
||||
artifactId = archiveName
|
||||
artifact("$buildDir/outputs/aar/$archiveName-release.aar")
|
||||
pom {
|
||||
name = pomName
|
||||
description = pomDescription
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,2 +1,3 @@
|
||||
<manifest>
|
||||
<manifest xmlns:android="http://schemas.android.com/apk/res/android"
|
||||
package="org.vosk.model.en">
|
||||
</manifest>
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
US English model for mobile Vosk applications
|
||||
|
||||
Copyright 2020 Alpha Cephei Inc
|
||||
|
||||
Accuracy: 10.38 (tedlium test) 9.85 (librispeech test-clean)
|
||||
Speed: 0.11xRT (desktop)
|
||||
Latency: 0.15s (right context)
|
||||
@@ -28,9 +28,6 @@ public class VoskDemo
|
||||
{
|
||||
// Demo float array
|
||||
VoskRecognizer rec = new VoskRecognizer(model, 16000.0f);
|
||||
|
||||
rec.SetEndpointerMode(EndpointerMode.LONG);
|
||||
|
||||
using(Stream source = File.OpenRead("test.wav")) {
|
||||
byte[] buffer = new byte[4096];
|
||||
int bytesRead;
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
<PropertyGroup>
|
||||
<OutputType>Exe</OutputType>
|
||||
<TargetFramework>net8.0</TargetFramework>
|
||||
<TargetFramework>net5.0</TargetFramework>
|
||||
<RootNamespace>VoskDemo</RootNamespace>
|
||||
</PropertyGroup>
|
||||
|
||||
@@ -11,7 +11,7 @@
|
||||
</PropertyGroup>
|
||||
|
||||
<ItemGroup>
|
||||
<PackageReference Include="Vosk" Version="0.3.75" />
|
||||
<PackageReference Include="Vosk" Version="0.3.45" />
|
||||
</ItemGroup>
|
||||
|
||||
</Project>
|
||||
|
||||
@@ -1,17 +0,0 @@
|
||||
<Project Sdk="Microsoft.NET.Sdk">
|
||||
|
||||
<PropertyGroup>
|
||||
<TargetFramework>net8.0</TargetFramework>
|
||||
<ImplicitUsings>enable</ImplicitUsings>
|
||||
<Nullable>enable</Nullable>
|
||||
<PackageId>Vosk</PackageId>
|
||||
<Version>0.3.75</Version>
|
||||
<authors>Alpha Cephei Inc</authors>
|
||||
<owners>Alpha Cephei Inc</owners>
|
||||
</PropertyGroup>
|
||||
|
||||
<Target Name="CopyFiles" AfterTargets="Build">
|
||||
<Copy SourceFiles="bin/Release/net8.0/Vosk.dll" DestinationFolder="lib/net8.0" />
|
||||
</Target>
|
||||
|
||||
</Project>
|
||||
@@ -2,7 +2,7 @@
|
||||
<package>
|
||||
<metadata>
|
||||
<id>Vosk</id>
|
||||
<version>0.3.75</version>
|
||||
<version>0.3.45</version>
|
||||
<authors>Alpha Cephei Inc</authors>
|
||||
<owners>Alpha Cephei Inc</owners>
|
||||
<license type="expression">Apache-2.0</license>
|
||||
@@ -23,10 +23,10 @@ Vosk scales from small devices like Raspberry Pi or Android smartphone to big cl
|
||||
<copyright>Copyright 2020-2050 Alpha Cephei Inc</copyright>
|
||||
<tags>speech recognition voice stt asr speech-to-text ai offline privacy</tags>
|
||||
<dependencies>
|
||||
<group targetFramework="net8.0"/>
|
||||
<group targetFramework=".NETStandard2.0"/>
|
||||
</dependencies>
|
||||
</metadata>
|
||||
<files>
|
||||
<file src="**" exclude="bin/**;obj/**;build.sh;src/*.cs;*.nupkg;**/.keep-me" />
|
||||
<file src="**" exclude="src/*.cs;build.sh;**/.keep-me;*.nupkg" />
|
||||
</files>
|
||||
</package>
|
||||
|
||||
@@ -1,2 +1,2 @@
|
||||
rm -rf bin lib obj
|
||||
/home/shmyrev/local/dotnet/dotnet pack Vosk.csproj -p:NuspecFile=Vosk.nuspec -o .
|
||||
mcs -out:lib/netstandard2.0/Vosk.dll -target:library src/*.cs
|
||||
nuget pack
|
||||
|
||||
@@ -1,55 +0,0 @@
|
||||
using System;
|
||||
|
||||
namespace Vosk
|
||||
{
|
||||
public class BatchModel : global::System.IDisposable
|
||||
{
|
||||
private global::System.Runtime.InteropServices.HandleRef handle;
|
||||
|
||||
internal BatchModel(global::System.IntPtr cPtr)
|
||||
{
|
||||
handle = new global::System.Runtime.InteropServices.HandleRef(this, cPtr);
|
||||
}
|
||||
|
||||
internal static global::System.Runtime.InteropServices.HandleRef getCPtr(BatchModel obj)
|
||||
{
|
||||
return (obj == null) ? new global::System.Runtime.InteropServices.HandleRef(null, global::System.IntPtr.Zero) : obj.handle;
|
||||
}
|
||||
|
||||
~BatchModel()
|
||||
{
|
||||
Dispose(false);
|
||||
}
|
||||
|
||||
public void Dispose()
|
||||
{
|
||||
Dispose(true);
|
||||
global::System.GC.SuppressFinalize(this);
|
||||
}
|
||||
|
||||
protected virtual void Dispose(bool disposing)
|
||||
{
|
||||
lock (this)
|
||||
{
|
||||
if (handle.Handle != global::System.IntPtr.Zero)
|
||||
{
|
||||
VoskPINVOKE.delete_BatchModel(handle);
|
||||
handle = new global::System.Runtime.InteropServices.HandleRef(null, global::System.IntPtr.Zero);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
public BatchModel(string model_path) : this(VoskPINVOKE.new_BatchModel(model_path))
|
||||
{
|
||||
}
|
||||
|
||||
public void WaitForCompletion()
|
||||
{
|
||||
if (handle.Handle != global::System.IntPtr.Zero)
|
||||
{
|
||||
VoskPINVOKE.wait_BatchModel(handle);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,88 +0,0 @@
|
||||
using System;
|
||||
|
||||
namespace Vosk
|
||||
{
|
||||
public class VoskBatchRecognizer : System.IDisposable
|
||||
{
|
||||
private System.Runtime.InteropServices.HandleRef handle;
|
||||
|
||||
internal VoskBatchRecognizer(System.IntPtr cPtr)
|
||||
{
|
||||
handle = new System.Runtime.InteropServices.HandleRef(this, cPtr);
|
||||
}
|
||||
|
||||
internal static System.Runtime.InteropServices.HandleRef getCPtr(VoskBatchRecognizer obj)
|
||||
{
|
||||
return (obj == null) ? new System.Runtime.InteropServices.HandleRef(null, System.IntPtr.Zero) : obj.handle;
|
||||
}
|
||||
|
||||
~VoskBatchRecognizer()
|
||||
{
|
||||
Dispose(false);
|
||||
}
|
||||
|
||||
public void Dispose()
|
||||
{
|
||||
Dispose(true);
|
||||
System.GC.SuppressFinalize(this);
|
||||
}
|
||||
|
||||
protected virtual void Dispose(bool disposing)
|
||||
{
|
||||
lock (this)
|
||||
{
|
||||
if (handle.Handle != System.IntPtr.Zero)
|
||||
{
|
||||
VoskPINVOKE.delete_VoskBatchRecognizer(handle);
|
||||
handle = new System.Runtime.InteropServices.HandleRef(null, System.IntPtr.Zero);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
public VoskBatchRecognizer(BatchModel model, float sample_rate) : this(VoskPINVOKE.new_VoskBatchRecognizer(BatchModel.getCPtr(model), sample_rate))
|
||||
{
|
||||
}
|
||||
|
||||
public bool AcceptWaveform(byte[] data, int len)
|
||||
{
|
||||
return VoskPINVOKE.VoskBatchRecognizer_AcceptWaveform(handle, data, len);
|
||||
}
|
||||
|
||||
private static string PtrToStringUTF8(System.IntPtr ptr)
|
||||
{
|
||||
int len = 0;
|
||||
while (System.Runtime.InteropServices.Marshal.ReadByte(ptr, len) != 0)
|
||||
len++;
|
||||
byte[] array = new byte[len];
|
||||
System.Runtime.InteropServices.Marshal.Copy(ptr, array, 0, len);
|
||||
return System.Text.Encoding.UTF8.GetString(array);
|
||||
}
|
||||
|
||||
public string FrontResult()
|
||||
{
|
||||
return PtrToStringUTF8(VoskPINVOKE.VoskBatchRecognizer_FrontResult(handle));
|
||||
}
|
||||
|
||||
public string Result()
|
||||
{
|
||||
string result = PtrToStringUTF8(VoskPINVOKE.VoskBatchRecognizer_FrontResult(handle));
|
||||
VoskPINVOKE.VoskBatchRecognizer_Pop(handle);
|
||||
return result;
|
||||
}
|
||||
|
||||
public int GetNumPendingChunks()
|
||||
{
|
||||
return VoskPINVOKE.VoskBatchRecognizer_GetPendingChunks(handle);
|
||||
}
|
||||
|
||||
public void FinishStream()
|
||||
{
|
||||
VoskPINVOKE.VoskBatchRecognizer_FinishStream(handle);
|
||||
}
|
||||
|
||||
public void SetNLSML(bool nlsml)
|
||||
{
|
||||
VoskPINVOKE.VoskBatchRecognizer_SetNLSML(handle, Convert.ToInt32(nlsml));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -65,12 +65,6 @@ class VoskPINVOKE {
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_reset")]
|
||||
public static extern void VoskRecognizer_Reset(global::System.Runtime.InteropServices.HandleRef jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_set_endpointer_mode")]
|
||||
public static extern void VoskRecognizer_SetEndpointerMode(global::System.Runtime.InteropServices.HandleRef jarg1, int jarg2);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_set_endpointer_delays")]
|
||||
public static extern void VoskRecognizer_SetEndpointerDelays(global::System.Runtime.InteropServices.HandleRef jarg1, float jarg2, float jarg3, float jarg4);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_set_log_level")]
|
||||
public static extern void SetLogLevel(int jarg1);
|
||||
|
||||
@@ -79,40 +73,6 @@ class VoskPINVOKE {
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_gpu_thread_init")]
|
||||
public static extern void GpuThreadInit();
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint = "vosk_batch_model_new")]
|
||||
public static extern global::System.IntPtr new_BatchModel(string jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint = "vosk_batch_model_free")]
|
||||
public static extern void delete_BatchModel(global::System.Runtime.InteropServices.HandleRef jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint = "vosk_batch_model_wait")]
|
||||
public static extern void wait_BatchModel(global::System.Runtime.InteropServices.HandleRef jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint = "vosk_batch_recognizer_new")]
|
||||
public static extern global::System.IntPtr new_VoskBatchRecognizer(global::System.Runtime.InteropServices.HandleRef jarg1, float jarg2);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint = "vosk_batch_recognizer_free")]
|
||||
public static extern void delete_VoskBatchRecognizer(global::System.Runtime.InteropServices.HandleRef jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint = "vosk_batch_recognizer_accept_waveform")]
|
||||
public static extern bool VoskBatchRecognizer_AcceptWaveform(global::System.Runtime.InteropServices.HandleRef jarg1, [global::System.Runtime.InteropServices.In, global::System.Runtime.InteropServices.MarshalAs(global::System.Runtime.InteropServices.UnmanagedType.LPArray)] byte[] jarg2, int jarg3);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint = "vosk_batch_recognizer_set_nlsml")]
|
||||
public static extern void VoskBatchRecognizer_SetNLSML(global::System.Runtime.InteropServices.HandleRef jarg1, int jarg2);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint = "vosk_batch_recognizer_finish_stream")]
|
||||
public static extern void VoskBatchRecognizer_FinishStream(global::System.Runtime.InteropServices.HandleRef jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint = "vosk_batch_recognizer_front_result")]
|
||||
public static extern global::System.IntPtr VoskBatchRecognizer_FrontResult(global::System.Runtime.InteropServices.HandleRef jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint = "vosk_batch_recognizer_pop")]
|
||||
public static extern void VoskBatchRecognizer_Pop(global::System.Runtime.InteropServices.HandleRef jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint = "vosk_batch_recognizer_get_pending_chunks")]
|
||||
public static extern int VoskBatchRecognizer_GetPendingChunks(global::System.Runtime.InteropServices.HandleRef jarg1);
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -1,12 +1,5 @@
|
||||
namespace Vosk {
|
||||
|
||||
public enum EndpointerMode {
|
||||
DEFAULT = 0,
|
||||
SHORT = 1,
|
||||
LONG = 2,
|
||||
VERY_LONG = 3
|
||||
}
|
||||
|
||||
public class VoskRecognizer : System.IDisposable {
|
||||
private System.Runtime.InteropServices.HandleRef handle;
|
||||
|
||||
@@ -98,14 +91,6 @@ public class VoskRecognizer : System.IDisposable {
|
||||
VoskPINVOKE.VoskRecognizer_Reset(handle);
|
||||
}
|
||||
|
||||
public void SetEndpointerMode(EndpointerMode mode) {
|
||||
VoskPINVOKE.VoskRecognizer_SetEndpointerMode(handle, (int) mode);
|
||||
}
|
||||
|
||||
public void SetEndpointerDelays(float t_start_max, float t_end, float t_max) {
|
||||
VoskPINVOKE.VoskRecognizer_SetEndpointerDelays(handle, t_start_max, t_end, t_max);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
-99
@@ -1,99 +0,0 @@
|
||||
package vosk
|
||||
|
||||
// #cgo CPPFLAGS: -I ${SRCDIR}/../src
|
||||
// #cgo !windows LDFLAGS: -L ${SRCDIR}/../src -lvosk -ldl -lpthread
|
||||
// #cgo windows LDFLAGS: -L ${SRCDIR}/../src -lvosk -lpthread
|
||||
// #include <stdlib.h>
|
||||
// #include <vosk_api.h>
|
||||
import "C"
|
||||
import "unsafe"
|
||||
|
||||
// VoskBatchModel contains a reference to the C VoskBatchModel
|
||||
type VoskBatchModel struct {
|
||||
model *C.struct_VoskBatchModel
|
||||
}
|
||||
|
||||
// NewBatchModel creates a new VoskBatchModel instance
|
||||
func NewBatchModel(modelPath string) (*VoskBatchModel, error) {
|
||||
cmodelPath := C.CString(modelPath)
|
||||
defer C.free(unsafe.Pointer(cmodelPath))
|
||||
internal := C.vosk_batch_model_new(cmodelPath)
|
||||
model := &VoskBatchModel{model: internal}
|
||||
return model, nil
|
||||
}
|
||||
|
||||
func (m *VoskBatchModel) Free() {
|
||||
C.vosk_batch_model_free(m.model)
|
||||
}
|
||||
|
||||
func (m *VoskBatchModel) Wait() {
|
||||
C.vosk_batch_model_wait(m.model);
|
||||
}
|
||||
|
||||
func freeBatchModel(model *VoskBatchModel) {
|
||||
C.vosk_batch_model_free(model.model)
|
||||
}
|
||||
|
||||
// VoskBatchRecognizer contains a reference to the C VoskBatchRecognizer
|
||||
type VoskBatchRecognizer struct {
|
||||
rec *C.struct_VoskBatchRecognizer
|
||||
}
|
||||
|
||||
func freeBatchRecognizer(recognizer *VoskBatchRecognizer) {
|
||||
C.vosk_batch_recognizer_free(recognizer.rec)
|
||||
}
|
||||
|
||||
func (r *VoskBatchRecognizer) Free() {
|
||||
C.vosk_batch_recognizer_free(r.rec)
|
||||
}
|
||||
|
||||
// NewBatchRecognizer creates a new VoskBatchRecognizer instance
|
||||
func NewBatchRecognizer(model *VoskBatchModel, sampleRate float64) (*VoskBatchRecognizer, error) {
|
||||
internal := C.vosk_batch_recognizer_new(model.model, C.float(sampleRate))
|
||||
rec := &VoskBatchRecognizer{rec: internal}
|
||||
return rec, nil
|
||||
}
|
||||
|
||||
// AcceptWaveform accepts and processes a new chunk of the voice data.
|
||||
func (r *VoskBatchRecognizer) AcceptWaveform(buffer []byte) {
|
||||
cbuf := C.CBytes(buffer)
|
||||
defer C.free(cbuf)
|
||||
C.vosk_batch_recognizer_accept_waveform(r.rec, (*C.char)(cbuf), C.int(len(buffer)))
|
||||
}
|
||||
|
||||
/** Set NLSML output
|
||||
* @param nlsml - boolean value
|
||||
*/
|
||||
//void vosk_batch_recognizer_set_nlsml(VoskBatchRecognizer *recognizer, int nlsml);
|
||||
|
||||
func (r *VoskBatchRecognizer) SetNlsml(nlsml int) {
|
||||
C.vosk_batch_recognizer_set_nlsml(r.rec, C.int(nlsml))
|
||||
}
|
||||
|
||||
/** Closes the stream */
|
||||
//void vosk_batch_recognizer_finish_stream(VoskBatchRecognizer *recognizer);
|
||||
|
||||
func (r *VoskBatchRecognizer) FinishStream() {
|
||||
C.vosk_batch_recognizer_finish_stream(r.rec)
|
||||
}
|
||||
|
||||
/** Return results */
|
||||
//const char *vosk_batch_recognizer_front_result(VoskBatchRecognizer *recognizer);
|
||||
|
||||
func (r *VoskBatchRecognizer) FrontResult() string {
|
||||
return C.GoString(C.vosk_batch_recognizer_front_result(r.rec))
|
||||
}
|
||||
|
||||
/** Release and free first retrieved result */
|
||||
//void vosk_batch_recognizer_pop(VoskBatchRecognizer *recognizer);
|
||||
|
||||
func (r *VoskBatchRecognizer) Pop() {
|
||||
C.vosk_batch_recognizer_pop(r.rec)
|
||||
}
|
||||
|
||||
/** Get amount of pending chunks for more intelligent waiting */
|
||||
//int vosk_batch_recognizer_get_pending_chunks(VoskBatchRecognizer *recognizer);
|
||||
func (r *VoskBatchRecognizer) GetPendingChunks() int {
|
||||
i := C.vosk_batch_recognizer_get_pending_chunks(r.rec)
|
||||
return int(i)
|
||||
}
|
||||
@@ -1,5 +0,0 @@
|
||||
This example expects a `s16le` converted audio file and converts it to text in a
|
||||
manner that imitates the Python example of [test_gpu_batch.py](../python/example/test_gpu_batch.py).
|
||||
|
||||
Note that the `libvosk.so` must be in the library path. This was successfully tested on
|
||||
Ubuntu 24.04 with Go 1.18, gcc-11, NVIDIA driver 570.172.08.
|
||||
@@ -1,54 +0,0 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
|
||||
vosk "github.com/alphacep/vosk-api/go"
|
||||
)
|
||||
|
||||
func main() {
|
||||
var filename string
|
||||
flag.StringVar(&filename, "f", "", "file to transcribe")
|
||||
flag.Parse()
|
||||
|
||||
vosk.GPUInit()
|
||||
|
||||
model, err := vosk.NewBatchModel("model")
|
||||
if err != nil {
|
||||
log.Fatal(err)
|
||||
}
|
||||
|
||||
rec, err := vosk.NewBatchRecognizer(model, 16000.0)
|
||||
if err != nil {
|
||||
log.Fatal(err)
|
||||
}
|
||||
|
||||
file, err := os.Open(filename)
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
defer file.Close()
|
||||
|
||||
buf := make([]byte, 8000)
|
||||
|
||||
for {
|
||||
if _, err := file.Read(buf); err != nil {
|
||||
if err != io.EOF {
|
||||
log.Fatal(err)
|
||||
}
|
||||
|
||||
break
|
||||
}
|
||||
rec.AcceptWaveform(buf)
|
||||
model.Wait()
|
||||
if rec.FrontResult() != "" {
|
||||
fmt.Println(rec.FrontResult())
|
||||
rec.Pop()
|
||||
}
|
||||
}
|
||||
// Is this needed? rec.FinishStream()
|
||||
}
|
||||
@@ -1,12 +1,13 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"bufio"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"encoding/json"
|
||||
|
||||
vosk "github.com/alphacep/vosk-api/go"
|
||||
)
|
||||
@@ -16,8 +17,6 @@ func main() {
|
||||
flag.StringVar(&filename, "f", "", "file to transcribe")
|
||||
flag.Parse()
|
||||
|
||||
vosk.GPUInit()
|
||||
|
||||
model, err := vosk.NewModel("model")
|
||||
if err != nil {
|
||||
log.Fatal(err)
|
||||
@@ -39,10 +38,11 @@ func main() {
|
||||
}
|
||||
defer file.Close()
|
||||
|
||||
reader := bufio.NewReader(file)
|
||||
buf := make([]byte, 4096)
|
||||
|
||||
for {
|
||||
_, err := file.Read(buf)
|
||||
_, err := reader.Read(buf)
|
||||
if err != nil {
|
||||
if err != io.EOF {
|
||||
log.Fatal(err)
|
||||
|
||||
+1
-19
@@ -1,8 +1,7 @@
|
||||
package vosk
|
||||
|
||||
// #cgo CPPFLAGS: -I ${SRCDIR}/../src
|
||||
// #cgo !windows LDFLAGS: -L ${SRCDIR}/../src -lvosk -ldl -lpthread
|
||||
// #cgo windows LDFLAGS: -L ${SRCDIR}/../src -lvosk -lpthread
|
||||
// #cgo LDFLAGS: -L ${SRCDIR}/../src -lvosk -ldl -lpthread
|
||||
// #include <stdlib.h>
|
||||
// #include <vosk_api.h>
|
||||
import "C"
|
||||
@@ -103,13 +102,6 @@ func (r *VoskRecognizer) SetSpkModel(spkModel *VoskSpkModel) {
|
||||
C.vosk_recognizer_set_spk_model(r.rec, spkModel.spkModel)
|
||||
}
|
||||
|
||||
// SetGrm sets which phrases to recognize on an already initialized recognizer.
|
||||
func (r *VoskRecognizer) SetGrm(grammar string) {
|
||||
cgrammar := C.CString(grammar)
|
||||
defer C.free(unsafe.Pointer(cgrammar))
|
||||
C.vosk_recognizer_set_grm(r.rec, cgrammar)
|
||||
}
|
||||
|
||||
// SetMaxAlternatives configures the recognizer to output n-best results.
|
||||
func (r *VoskRecognizer) SetMaxAlternatives(maxAlternatives int) {
|
||||
C.vosk_recognizer_set_max_alternatives(r.rec, C.int(maxAlternatives))
|
||||
@@ -125,16 +117,6 @@ func (r *VoskRecognizer) SetPartialWords(words int) {
|
||||
C.vosk_recognizer_set_partial_words(r.rec, C.int(words))
|
||||
}
|
||||
|
||||
// SetEndpointerDelays sets the recognition timeouts, where startMax
|
||||
// is the timeout for stopping recognition in case of initial silence
|
||||
// (usually around 5), end is the timeout for stopping recognition
|
||||
// in milliseconds after we recognized something (usually around 0.5-1.0),
|
||||
// and max is the timeout for forcing utterance end in milliseconds
|
||||
// (usually around 20-30).
|
||||
func (r *VoskRecognizer) SetEndpointerDelays(startMax, end, max float64) {
|
||||
C.vosk_recognizer_set_endpointer_delays(r.rec, C.float(startMax), C.float(end), C.float(max))
|
||||
}
|
||||
|
||||
// AcceptWaveform accepts and processes a new chunk of the voice data.
|
||||
func (r *VoskRecognizer) AcceptWaveform(buffer []byte) int {
|
||||
cbuf := C.CBytes(buffer)
|
||||
|
||||
@@ -11,5 +11,6 @@ repositories {
|
||||
}
|
||||
|
||||
dependencies {
|
||||
implementation group: 'com.alphacephei', name: 'vosk', version: '0.3.75'
|
||||
implementation group: 'net.java.dev.jna', name: 'jna', version: '5.7.0'
|
||||
implementation group: 'com.alphacephei', name: 'vosk', version: '0.3.40+'
|
||||
}
|
||||
|
||||
@@ -16,7 +16,7 @@ repositories {
|
||||
|
||||
archivesBaseName = 'vosk'
|
||||
group = 'com.alphacephei'
|
||||
version = '0.3.75'
|
||||
version = '0.3.45'
|
||||
|
||||
mavenPublish {
|
||||
group = 'com.alphacephei'
|
||||
@@ -25,7 +25,7 @@ mavenPublish {
|
||||
}
|
||||
|
||||
dependencies {
|
||||
api group: 'net.java.dev.jna', name: 'jna', version: '5.18.1'
|
||||
api group: 'net.java.dev.jna', name: 'jna', version: '5.7.0'
|
||||
testImplementation 'junit:junit:4.13'
|
||||
}
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package org.vosk;
|
||||
|
||||
import com.sun.jna.Native;
|
||||
import com.sun.jna.Library;
|
||||
import com.sun.jna.Platform;
|
||||
import com.sun.jna.Pointer;
|
||||
import java.io.File;
|
||||
@@ -12,9 +13,8 @@ import java.nio.file.StandardCopyOption;
|
||||
public class LibVosk {
|
||||
|
||||
private static void unpackDll(File targetDir, String lib) throws IOException {
|
||||
try (InputStream source = LibVosk.class.getResourceAsStream("/win32-x86-64/" + lib + ".dll")) {
|
||||
Files.copy(source, new File(targetDir, lib + ".dll").toPath(), StandardCopyOption.REPLACE_EXISTING);
|
||||
}
|
||||
InputStream source = LibVosk.class.getResourceAsStream("/win32-x86-64/" + lib + ".dll");
|
||||
Files.copy(source, new File(targetDir, lib + ".dll").toPath(), StandardCopyOption.REPLACE_EXISTING);
|
||||
}
|
||||
|
||||
static {
|
||||
@@ -23,7 +23,7 @@ public class LibVosk {
|
||||
// We have to unpack dependencies
|
||||
try {
|
||||
// To get a tmp folder we unpack small library and mark it for deletion
|
||||
File tmpFile = Native.extractFromResourcePath("/win32-x86-64/empty", LibVosk.class.getClassLoader());
|
||||
File tmpFile = Native.extractFromResourcePath("/win32-x86-64/empty");
|
||||
File tmpDir = tmpFile.getParentFile();
|
||||
new File(tmpDir, tmpFile.getName() + ".x").createNewFile();
|
||||
|
||||
@@ -78,30 +78,10 @@ public class LibVosk {
|
||||
|
||||
public static native String vosk_recognizer_partial_result(Pointer recognizer);
|
||||
|
||||
public static native void vosk_recognizer_set_grm(Pointer recognizer, String grammar);
|
||||
|
||||
public static native void vosk_recognizer_reset(Pointer recognizer);
|
||||
|
||||
public static native void vosk_recognizer_set_endpointer_mode(Pointer recognizer, int mode);
|
||||
|
||||
public static native void vosk_recognizer_set_endpointer_delays(Pointer recognizer, float t_start_max, float t_end, float t_max);
|
||||
|
||||
public static native void vosk_recognizer_free(Pointer recognizer);
|
||||
|
||||
public static native Pointer vosk_text_processor_new(String verbalizer, String tagger);
|
||||
|
||||
public static native void vosk_text_processor_free(Pointer processor);
|
||||
|
||||
public static native String vosk_text_processor_itn(Pointer processor, String input);
|
||||
|
||||
/**
|
||||
* Set log level for Kaldi messages.
|
||||
*
|
||||
* @param loglevel the level
|
||||
* 0 - default value to print info and error messages but no debug
|
||||
* less than 0 - don't print info messages
|
||||
* greater than 0 - more verbose mode
|
||||
*/
|
||||
public static void setLogLevel(LogLevel loglevel) {
|
||||
vosk_set_log_level(loglevel.getValue());
|
||||
}
|
||||
|
||||
@@ -4,17 +4,6 @@ import com.sun.jna.PointerType;
|
||||
import java.io.IOException;
|
||||
|
||||
public class Recognizer extends PointerType implements AutoCloseable {
|
||||
/**
|
||||
* Creates the recognizer object.
|
||||
*
|
||||
* The recognizers process the speech and return text using shared model data
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The sample rate of the audio you are going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @throws IOException if the recognizer could not be created
|
||||
*/
|
||||
public Recognizer(Model model, float sampleRate) throws IOException {
|
||||
super(LibVosk.vosk_recognizer_new(model, sampleRate));
|
||||
|
||||
@@ -23,141 +12,30 @@ public class Recognizer extends PointerType implements AutoCloseable {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates the recognizer object with speaker recognition.
|
||||
*
|
||||
* With the speaker recognition mode the recognizer not just recognize
|
||||
* text but also return speaker vectors one can use for speaker identification
|
||||
*
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The sample rate of the audio you are going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @param spkModel speaker model for speaker identification
|
||||
* @throws IOException if the recognizer could not be created
|
||||
*/
|
||||
public Recognizer(Model model, float sampleRate, SpeakerModel spkModel) throws IOException {
|
||||
public Recognizer(Model model, float sampleRate, SpeakerModel spkModel) {
|
||||
super(LibVosk.vosk_recognizer_new_spk(model.getPointer(), sampleRate, spkModel.getPointer()));
|
||||
|
||||
if (getPointer() == null) {
|
||||
throw new IOException("Failed to create a recognizer");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates the recognizer object with the phrase list.
|
||||
*
|
||||
* Sometimes when you want to improve recognition accuracy and when you don't need
|
||||
* to recognize large vocabulary you can specify a list of phrases to recognize. This
|
||||
* will improve recognizer speed and accuracy but might return [unk] if user said
|
||||
* something different.
|
||||
*
|
||||
* Only recognizers with lookahead models support this type of quick configuration.
|
||||
* Precompiled HCLG graph models are not supported.
|
||||
*
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The sample rate of the audio you are going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @param grammar The string with the list of phrases to recognize as JSON array of strings,
|
||||
* for example "["one two three four five", "[unk]"]".
|
||||
* @throws IOException if the recognizer could not be created
|
||||
*/
|
||||
public Recognizer(Model model, float sampleRate, String grammar) throws IOException {
|
||||
public Recognizer(Model model, float sampleRate, String grammar) {
|
||||
super(LibVosk.vosk_recognizer_new_grm(model.getPointer(), sampleRate, grammar));
|
||||
|
||||
if (getPointer() == null) {
|
||||
throw new IOException("Failed to create a recognizer");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Configures recognizer to output n-best results.
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "alternatives": [
|
||||
* { "text": "one two three four five", "confidence": 0.97 },
|
||||
* { "text": "one two three for five", "confidence": 0.03 },
|
||||
* ]
|
||||
* }
|
||||
* </pre>
|
||||
*
|
||||
* @param maxAlternatives - maximum alternatives to return from recognition results
|
||||
*/
|
||||
public void setMaxAlternatives(int maxAlternatives) {
|
||||
LibVosk.vosk_recognizer_set_max_alternatives(this.getPointer(), maxAlternatives);
|
||||
}
|
||||
|
||||
/** Enables words with times in the output
|
||||
*
|
||||
* <pre>
|
||||
* "result" : [{
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.110000,
|
||||
* "start" : 0.870000,
|
||||
* "word" : "what"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.530000,
|
||||
* "start" : 1.110000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.950000,
|
||||
* "start" : 1.530000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 2.340000,
|
||||
* "start" : 1.950000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 2.610000,
|
||||
* "start" : 2.340000,
|
||||
* "word" : "one"
|
||||
* }],
|
||||
* </pre>
|
||||
*
|
||||
* @param words - boolean value
|
||||
*/
|
||||
public void setWords(boolean words) {
|
||||
LibVosk.vosk_recognizer_set_words(this.getPointer(), words);
|
||||
}
|
||||
|
||||
/**
|
||||
* Like above return words and confidences in partial results.
|
||||
*
|
||||
* @param partial_words - boolean value
|
||||
*/
|
||||
public void setPartialWords(boolean partial_words) {
|
||||
LibVosk.vosk_recognizer_set_partial_words(this.getPointer(), partial_words);
|
||||
}
|
||||
|
||||
/**
|
||||
* Adds speaker model to already initialized recognizer.
|
||||
*
|
||||
* Can add speaker recognition model to already created recognizer.
|
||||
* Helps to initialize speaker recognition for grammar-based recognizer.
|
||||
*
|
||||
* @param spkModel Speaker recognition model
|
||||
*/
|
||||
public void setSpeakerModel(SpeakerModel spkModel) {
|
||||
LibVosk.vosk_recognizer_set_spk_model(this.getPointer(), spkModel.getPointer());
|
||||
}
|
||||
|
||||
/**
|
||||
* Accept and process new chunk of voice data.
|
||||
*
|
||||
* @param data - audio data in PCM 16-bit mono format
|
||||
* @param len - length of the audio data
|
||||
* @return 1 if silence is occurred and you can retrieve a new utterance with result method
|
||||
* 0 if decoding continues
|
||||
* -1 if exception occurred
|
||||
*/
|
||||
public boolean acceptWaveForm(byte[] data, int len) {
|
||||
return LibVosk.vosk_recognizer_accept_waveform(this.getPointer(), data, len);
|
||||
}
|
||||
@@ -170,104 +48,22 @@ public class Recognizer extends PointerType implements AutoCloseable {
|
||||
return LibVosk.vosk_recognizer_accept_waveform_f(this.getPointer(), data, len);
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns speech recognition result
|
||||
*
|
||||
* @return the result in JSON format which contains decoded line, decoded
|
||||
* words, times in seconds and confidences. You can parse this result
|
||||
* with any json parser
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "text" : "what zero zero zero one"
|
||||
* }
|
||||
* </pre>
|
||||
*
|
||||
* If alternatives enabled it returns result with alternatives, see also #setMaxAlternatives().
|
||||
*
|
||||
* If word times enabled returns word time, see also #setWordTimes().
|
||||
*/
|
||||
public String getResult() {
|
||||
return LibVosk.vosk_recognizer_result(this.getPointer());
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns partial speech recognition.
|
||||
*
|
||||
* @return partial speech recognition text which is not yet finalized.
|
||||
* result may change as recognizer process more data.
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "partial" : "cyril one eight zero"
|
||||
* }
|
||||
* </pre>
|
||||
*/
|
||||
public String getPartialResult() {
|
||||
return LibVosk.vosk_recognizer_partial_result(this.getPointer());
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns speech recognition result. Same as result, but doesn't wait for silence.
|
||||
* You usually call it in the end of the stream to get final bits of audio. It
|
||||
* flushes the feature pipeline, so all remaining audio chunks got processed.
|
||||
*
|
||||
* @return speech result in JSON format.
|
||||
*/
|
||||
public String getFinalResult() {
|
||||
return LibVosk.vosk_recognizer_final_result(this.getPointer());
|
||||
}
|
||||
|
||||
/**
|
||||
* Reconfigures recognizer to use grammar.
|
||||
*
|
||||
* @param grammar Set of phrases in JSON array of strings or "[]" to use default model graph.
|
||||
* @see #Recognizer(Model, float, String)
|
||||
*/
|
||||
public void setGrammar(String grammar) {
|
||||
LibVosk.vosk_recognizer_set_grm(this.getPointer(), grammar);
|
||||
}
|
||||
|
||||
/**
|
||||
* Resets the recognizer.
|
||||
* Resets current results so the recognition can continue from scratch.
|
||||
*/
|
||||
public void reset() {
|
||||
LibVosk.vosk_recognizer_reset(this.getPointer());
|
||||
}
|
||||
|
||||
/**
|
||||
* Endpointer delay mode
|
||||
*/
|
||||
public class EndpointerMode {
|
||||
public static final int DEFAULT = 0;
|
||||
public static final int SHORT = 1;
|
||||
public static final int LONG = 2;
|
||||
public static final int VERY_LONG = 3;
|
||||
}
|
||||
|
||||
/**
|
||||
* Configures endpointer mode for recognizer
|
||||
*/
|
||||
public void setEndpointerMode(int mode) {
|
||||
LibVosk.vosk_recognizer_set_endpointer_mode(this.getPointer(), mode);
|
||||
}
|
||||
|
||||
/**
|
||||
* Set endpointer delays
|
||||
*
|
||||
* @param t_start_max timeout for stopping recognition in case of initial silence (usually around 5.0)
|
||||
* @param t_end timeout for stopping recognition in milliseconds after we recognized something (usually around 0.5 - 1.0)
|
||||
* @param t_max timeout for forcing utterance end in milliseconds (usually around 20-30)
|
||||
**/
|
||||
public void setEndpointerDelays(float t_start_max, float t_end, float t_max) {
|
||||
LibVosk.vosk_recognizer_set_endpointer_delays(this.getPointer(), t_start_max, t_end, t_max);
|
||||
}
|
||||
|
||||
/**
|
||||
* Releases recognizer object.
|
||||
* Underlying model is also unreferenced and if needed, released.
|
||||
*/
|
||||
@Override
|
||||
public void close() {
|
||||
LibVosk.vosk_recognizer_free(this.getPointer());
|
||||
|
||||
@@ -3,28 +3,10 @@ package org.vosk;
|
||||
import com.sun.jna.PointerType;
|
||||
import java.io.IOException;
|
||||
|
||||
/**
|
||||
* Helps to initialize speaker recognition for grammar-based recognizer.
|
||||
*/
|
||||
public class SpeakerModel extends PointerType implements AutoCloseable {
|
||||
public SpeakerModel() {
|
||||
}
|
||||
|
||||
/**
|
||||
* Loads speaker model data from the file.
|
||||
*
|
||||
* The path must contain:
|
||||
* - a config file: mfcc.conf
|
||||
* - kaldi nnet: final.ext.raw
|
||||
* - mean.vec
|
||||
* - transform.mat
|
||||
*
|
||||
* @param path the path of the model on the filesystem
|
||||
* @throws IOException if the model could not be created
|
||||
*
|
||||
* @see <a href="http://kaldi-asr.org/doc/structkaldi_1_1MfccOptions.html">Kaldi MfccOptions</a>
|
||||
* @see <a href="http://kaldi-asr.org/doc/classkaldi_1_1nnet3_1_1Nnet.html">Kaldi Nnet</a>
|
||||
*/
|
||||
public SpeakerModel(String path) throws IOException {
|
||||
super(LibVosk.vosk_spk_model_new(path));
|
||||
|
||||
|
||||
@@ -15,10 +15,8 @@ import javax.sound.sampled.UnsupportedAudioFileException;
|
||||
|
||||
import org.vosk.LogLevel;
|
||||
import org.vosk.Recognizer;
|
||||
import org.vosk.Recognizer.EndpointerMode;
|
||||
import org.vosk.LibVosk;
|
||||
import org.vosk.Model;
|
||||
import org.vosk.TextProcessor;
|
||||
|
||||
public class DecoderTest {
|
||||
|
||||
@@ -97,24 +95,9 @@ public class DecoderTest {
|
||||
Assert.assertTrue(true);
|
||||
}
|
||||
|
||||
@Test
|
||||
public void decoderEndpointerDelays() throws IOException, UnsupportedAudioFileException {
|
||||
try (Model model = new Model("model");
|
||||
Recognizer recognizer = new Recognizer(model, 16000)) {
|
||||
recognizer.setEndpointerMode(EndpointerMode.VERY_LONG);
|
||||
recognizer.setEndpointerDelays(5.0f, 3.0f, 50.0f);
|
||||
}
|
||||
Assert.assertTrue(true);
|
||||
}
|
||||
|
||||
@Test(expected = IOException.class)
|
||||
public void decoderTestException() throws IOException {
|
||||
Model model = new Model("model_missing");
|
||||
}
|
||||
|
||||
@Test
|
||||
public void testItn() throws IOException {
|
||||
TextProcessor p = new TextProcessor("model/itn/en_itn_tagger.fst", "model/itn/en_itn_verbalizer.fst");
|
||||
System.out.println(p.itn("as easy as one two three"));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,3 +0,0 @@
|
||||
.gradle/
|
||||
.idea/
|
||||
build/
|
||||
@@ -1 +0,0 @@
|
||||
Doomsdayrs doomsdayrs@gmail.com
|
||||
@@ -1,8 +0,0 @@
|
||||
# Maintenance
|
||||
|
||||
To maintain this module, please ensure the following.
|
||||
1. Kotlin version is kept up to date.
|
||||
This will be found in the plugins block of the [build.gradle.kts](./build.gradle.kts).
|
||||
Ensure both multiplatform and serialization have the same value.
|
||||
2. Ensure dependencies are up to date
|
||||
3. Ensure that the android min & target sdks are up to date.
|
||||
@@ -1,63 +0,0 @@
|
||||
# vosk-api-kotlin
|
||||
|
||||
The vosk-api wrapped using Kotlin Multiplatform.
|
||||
|
||||
## Usage
|
||||
|
||||
The following are ways to use this wrapper.
|
||||
|
||||
### JVM
|
||||
|
||||
For Java & Android targets.
|
||||
|
||||
```kotlin
|
||||
dependencoes {
|
||||
val voskVersion = "0.4.0-alpha0"
|
||||
|
||||
// Generic
|
||||
implementation("com.alphacephei:vosk-api-kotlin:$voskVersion")
|
||||
|
||||
// Android
|
||||
implementation("com.alphacephei:vosk-api-kotlin-android:$voskVersion")
|
||||
}
|
||||
```
|
||||
|
||||
## Building
|
||||
|
||||
To build this project, follow the following steps.
|
||||
|
||||
1. Install `libvosk`.
|
||||
This can be done from [source][source install] (parent monorepo)
|
||||
or [downloaded][download].
|
||||
2. Download a [Vosk model](https://alphacephei.com/vosk/models) to use.
|
||||
- It is suggested to use a small model to speed up tests.
|
||||
3. Once both are downloaded and placed into a proper location (hopefully following UNIX specification).
|
||||
Set the following environment variables:
|
||||
- `VOSK_MODEL` to the path of the model.
|
||||
- `VOSK_PATH` to the path of `libvosk`.
|
||||
These are used by the various tests to operate.
|
||||
4. Now that the required steps are complete, one can run `./gradlew build`.
|
||||
|
||||
## Kotlin/Native
|
||||
|
||||
Currently, the native target is disabled due to a lack of vosk-api packaging on platforms.
|
||||
Further worsened by the fact that installing the vosk-api on Linux systems is a chore.
|
||||
|
||||
First, either install from [source][source install]
|
||||
or [download][download] and install into the proper unix directories as expected in
|
||||
[libvosk.def](./src/nativeInterop/cinterop/libvosk.def).
|
||||
|
||||
To enable development for Kotlin/Native, do either for the following.
|
||||
- Add `NATIVE_EXPERIMENT=true` to your environment.
|
||||
- Go into [build.gradle.kts](./build.gradle.kts) & find `enableNative` & set the right side to true.
|
||||
|
||||
Afterwards, when syncing the project, the native source sets will become available to work on.
|
||||
It is suggested to run `cinteropLibvoskNative` to generate the Kotlin C bindings to work with.
|
||||
|
||||
## Future
|
||||
|
||||
- Possibly target Kotlin/JS
|
||||
- Possibly target Kotlin/Objective-C (?)
|
||||
|
||||
[source install]: https://alphacephei.com/vosk/install
|
||||
[download]: https://github.com/alphacep/vosk-api/releases/latest
|
||||
@@ -1,204 +0,0 @@
|
||||
import org.jetbrains.dokka.gradle.DokkaTask
|
||||
import org.jetbrains.kotlin.gradle.ExperimentalKotlinGradlePluginApi
|
||||
import org.jetbrains.kotlin.gradle.dsl.JvmTarget
|
||||
|
||||
/*
|
||||
* Copyright 2020 Alpha Cephei Inc. & Doomsdayrs
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
plugins {
|
||||
kotlin("multiplatform") version "2.0.0"
|
||||
id("com.android.library")
|
||||
`maven-publish`
|
||||
id("org.jetbrains.dokka") version "1.9.20"
|
||||
kotlin("plugin.serialization") version "2.0.0"
|
||||
}
|
||||
|
||||
group = "com.alphacephei"
|
||||
version = "0.3.75"
|
||||
|
||||
repositories {
|
||||
google()
|
||||
mavenCentral()
|
||||
}
|
||||
|
||||
val dokkaOutputDir = "$buildDir/dokka"
|
||||
|
||||
tasks.getByName<DokkaTask>("dokkaHtml") {
|
||||
outputDirectory.set(file(dokkaOutputDir))
|
||||
}
|
||||
|
||||
val deleteDokkaOutputDir by tasks.register<Delete>("deleteDokkaOutputDirectory") {
|
||||
delete(dokkaOutputDir)
|
||||
}
|
||||
|
||||
val javadocJar = tasks.register<Jar>("javadocJar") {
|
||||
dependsOn(deleteDokkaOutputDir, tasks.dokkaHtml)
|
||||
archiveClassifier.set("javadoc")
|
||||
from(dokkaOutputDir)
|
||||
}
|
||||
|
||||
fun org.jetbrains.kotlin.gradle.dsl.KotlinMultiplatformExtension.native(
|
||||
configure: org.jetbrains.kotlin.gradle.plugin.mpp.KotlinNativeTargetWithHostTests.() -> Unit = {}
|
||||
) {
|
||||
when {
|
||||
org.jetbrains.kotlin.konan.target.HostManager.hostIsMingw -> mingwX64("native")
|
||||
org.jetbrains.kotlin.konan.target.HostManager.hostIsLinux -> linuxX64("native")
|
||||
org.jetbrains.kotlin.konan.target.HostManager.hostIsMac -> if (org.jetbrains.kotlin.konan.target.HostManager.hostArch() == "arm64") {
|
||||
macosArm64("native")
|
||||
} else {
|
||||
macosX64("native")
|
||||
}
|
||||
|
||||
else -> error("Unsupported Host OS: ${org.jetbrains.kotlin.konan.target.HostManager.hostOs()}")
|
||||
}.apply(configure)
|
||||
}
|
||||
|
||||
kotlin {
|
||||
jvm {
|
||||
@OptIn(ExperimentalKotlinGradlePluginApi::class)
|
||||
compilerOptions {
|
||||
jvmTarget.set(JvmTarget.JVM_17)
|
||||
}
|
||||
|
||||
testRuns["test"].executionTask.configure {
|
||||
useJUnitPlatform()
|
||||
environment("MODEL", "VOSK_MODEL")
|
||||
//environment("MODEL", "/home/doomsdayrs/Downloads/vosk-model-small-en-us-0.15/")
|
||||
environment("LIBRARY", "VOSK_PATH")
|
||||
//environment("LIBRARY", "/usr/local/lib64/libvosk/libvosk.so")
|
||||
environment("AUDIO", "$projectDir/../python/example/test.wav")
|
||||
}
|
||||
}
|
||||
|
||||
androidTarget {
|
||||
publishAllLibraryVariants()
|
||||
}
|
||||
|
||||
/**
|
||||
* If native target should be enabled or not.
|
||||
*
|
||||
* Currently disabled as there is no proper packaging distribution currently.
|
||||
*/
|
||||
@Suppress("SimplifyBooleanWithConstants") // Ignore, the false is for overrides
|
||||
val enableNative = System.getenv("NATIVE_EXPERIMENT") == "true" || false
|
||||
|
||||
if (enableNative)
|
||||
native {
|
||||
val main by compilations.getting
|
||||
val libvosk by main.cinterops.creating
|
||||
|
||||
binaries {
|
||||
sharedLib()
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@OptIn(ExperimentalKotlinGradlePluginApi::class)
|
||||
applyDefaultHierarchyTemplate {
|
||||
withJvm()
|
||||
withAndroidTarget()
|
||||
|
||||
if (enableNative)
|
||||
withNative()
|
||||
}
|
||||
|
||||
publishing {
|
||||
publications {
|
||||
withType<MavenPublication> {
|
||||
artifact(javadocJar)
|
||||
pom {
|
||||
url.set("http://www.alphacephei.com.com/vosk/")
|
||||
licenses {
|
||||
license {
|
||||
name.set("The Apache License, Version 2.0")
|
||||
url.set("http://www.apache.org/licenses/LICENSE-2.0.txt")
|
||||
}
|
||||
}
|
||||
developers {
|
||||
developer {
|
||||
id.set("com.alphacephei")
|
||||
name.set("Alpha Cephei Inc")
|
||||
email.set("contact@alphacephei.com")
|
||||
}
|
||||
}
|
||||
scm {
|
||||
connection.set("scm:git:git://github.com/alphacep/vosk-api.git")
|
||||
url.set("https://github.com/alphacep/vosk-api/")
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
val jna_version = "5.14.0"
|
||||
val coroutines_version = "1.7.3"
|
||||
|
||||
sourceSets {
|
||||
val commonMain by getting {
|
||||
dependencies {
|
||||
api("org.jetbrains.kotlinx:kotlinx-serialization-json:1.7.0")
|
||||
api("org.jetbrains.kotlinx:kotlinx-coroutines-core:$coroutines_version")
|
||||
}
|
||||
}
|
||||
val commonTest by getting {
|
||||
dependencies {
|
||||
implementation(kotlin("test"))
|
||||
implementation("org.jetbrains.kotlinx:kotlinx-coroutines-test:$coroutines_version")
|
||||
}
|
||||
}
|
||||
val jvmMain by getting {
|
||||
dependencies {
|
||||
api("net.java.dev.jna:jna:$jna_version")
|
||||
}
|
||||
}
|
||||
val jvmTest by getting
|
||||
if (enableNative) {
|
||||
val nativeMain by getting
|
||||
}
|
||||
val androidMain by getting {
|
||||
dependsOn(jvmMain)
|
||||
dependencies {
|
||||
api("net.java.dev.jna:jna:$jna_version@aar")
|
||||
}
|
||||
}
|
||||
val androidUnitTest by getting {
|
||||
dependencies {
|
||||
implementation("junit:junit:4.13.2")
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
android {
|
||||
namespace = "com.alphacephei.library"
|
||||
compileSdk = 34
|
||||
sourceSets["main"].manifest.srcFile("src/androidMain/AndroidManifest.xml")
|
||||
defaultConfig {
|
||||
minSdk = 24
|
||||
targetSdk = 34
|
||||
}
|
||||
compileOptions {
|
||||
sourceCompatibility = JavaVersion.VERSION_17
|
||||
targetCompatibility = JavaVersion.VERSION_17
|
||||
}
|
||||
publishing {
|
||||
multipleVariants {
|
||||
withSourcesJar()
|
||||
withJavadocJar()
|
||||
allVariants()
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,31 +0,0 @@
|
||||
/*
|
||||
* Copyright 2020 Alpha Cephei Inc. & Doomsdayrs
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
pluginManagement {
|
||||
repositories {
|
||||
google()
|
||||
gradlePluginPortal()
|
||||
mavenCentral()
|
||||
}
|
||||
resolutionStrategy {
|
||||
eachPlugin {
|
||||
if (requested.id.namespace == "com.android") {
|
||||
useModule("com.android.tools.build:gradle:8.3.0")
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
rootProject.name = "vosk-api-kotlin"
|
||||
@@ -1,6 +0,0 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<manifest xmlns:android="http://schemas.android.com/apk/res/android"
|
||||
package="com.alphacephei.library">
|
||||
|
||||
<uses-permission android:name="android.permission.RECORD_AUDIO" />
|
||||
</manifest>
|
||||
@@ -1,46 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package org.vosk.android
|
||||
|
||||
/**
|
||||
* Interface to receive recognition results
|
||||
*/
|
||||
interface RecognitionListener {
|
||||
/**
|
||||
* Called when partial recognition result is available.
|
||||
*/
|
||||
fun onPartialResult(hypothesis: String)
|
||||
|
||||
/**
|
||||
* Called after silence occured.
|
||||
*/
|
||||
fun onResult(hypothesis: String)
|
||||
|
||||
/**
|
||||
* Called after stream end.
|
||||
*/
|
||||
fun onFinalResult(hypothesis: String)
|
||||
|
||||
/**
|
||||
* Called when an error occurs.
|
||||
*/
|
||||
fun onError(exception: Exception)
|
||||
|
||||
/**
|
||||
* Called after timeout expired
|
||||
*/
|
||||
fun onTimeout()
|
||||
}
|
||||
@@ -1,241 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package org.vosk.android
|
||||
|
||||
import android.annotation.SuppressLint
|
||||
import android.media.AudioFormat
|
||||
import android.media.AudioRecord
|
||||
import android.media.MediaRecorder.AudioSource
|
||||
import android.os.Handler
|
||||
import android.os.Looper
|
||||
import org.vosk.Recognizer
|
||||
import java.io.IOException
|
||||
import kotlin.math.roundToInt
|
||||
|
||||
/**
|
||||
* Service that records audio in a thread, passes it to a recognizer and emits
|
||||
* recognition results. Recognition events are passed to a client using
|
||||
* [RecognitionListener]
|
||||
*/
|
||||
class SpeechService @Throws(IOException::class) constructor(
|
||||
private val recognizer: Recognizer,
|
||||
sampleRate: Float
|
||||
) {
|
||||
private val sampleRate: Int
|
||||
private val bufferSize: Int
|
||||
private val recorder: AudioRecord
|
||||
private var recognizerThread: RecognizerThread? = null
|
||||
private val mainHandler = Handler(Looper.getMainLooper())
|
||||
|
||||
/**
|
||||
* Creates speech service. Service holds the AudioRecord object, so you
|
||||
* need to call [.shutdown] in order to properly finalize it.
|
||||
*
|
||||
* @throws IOException thrown if audio recorder can not be created for some reason.
|
||||
*/
|
||||
init {
|
||||
this.sampleRate = sampleRate.toInt()
|
||||
bufferSize = (this.sampleRate * BUFFER_SIZE_SECONDS).roundToInt()
|
||||
@SuppressLint("MissingPermission")
|
||||
recorder = AudioRecord(
|
||||
AudioSource.VOICE_RECOGNITION, this.sampleRate,
|
||||
AudioFormat.CHANNEL_IN_MONO,
|
||||
AudioFormat.ENCODING_PCM_16BIT, bufferSize * 2
|
||||
)
|
||||
if (recorder.state == AudioRecord.STATE_UNINITIALIZED) {
|
||||
recorder.release()
|
||||
throw IOException(
|
||||
"Failed to initialize recorder. Microphone might be already in use."
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Starts recognition. Does nothing if recognition is active.
|
||||
*
|
||||
* @return true if recognition was actually started
|
||||
*/
|
||||
fun startListening(listener: RecognitionListener): Boolean {
|
||||
if (null != recognizerThread) return false
|
||||
recognizerThread = RecognizerThread(listener)
|
||||
recognizerThread!!.start()
|
||||
return true
|
||||
}
|
||||
|
||||
/**
|
||||
* Starts recognition. After specified timeout listening stops and the
|
||||
* endOfSpeech signals about that. Does nothing if recognition is active.
|
||||
*
|
||||
*
|
||||
* timeout - timeout in milliseconds to listen.
|
||||
*
|
||||
* @return true if recognition was actually started
|
||||
*/
|
||||
fun startListening(listener: RecognitionListener, timeout: Int): Boolean {
|
||||
if (null != recognizerThread) return false
|
||||
recognizerThread = RecognizerThread(listener, timeout)
|
||||
recognizerThread!!.start()
|
||||
return true
|
||||
}
|
||||
|
||||
private fun stopRecognizerThread(): Boolean {
|
||||
if (null == recognizerThread) return false
|
||||
try {
|
||||
recognizerThread!!.interrupt()
|
||||
recognizerThread!!.join()
|
||||
} catch (e: InterruptedException) {
|
||||
// Restore the interrupted status.
|
||||
Thread.currentThread().interrupt()
|
||||
}
|
||||
recognizerThread = null
|
||||
return true
|
||||
}
|
||||
|
||||
/**
|
||||
* Stops recognition. Listener should receive final result if there is
|
||||
* any. Does nothing if recognition is not active.
|
||||
*
|
||||
* @return true if recognition was actually stopped
|
||||
*/
|
||||
fun stop(): Boolean {
|
||||
return stopRecognizerThread()
|
||||
}
|
||||
|
||||
/**
|
||||
* Cancel recognition. Do not post any new events, simply cancel processing.
|
||||
* Does nothing if recognition is not active.
|
||||
*
|
||||
* @return true if recognition was actually stopped
|
||||
*/
|
||||
fun cancel(): Boolean {
|
||||
if (recognizerThread != null) {
|
||||
recognizerThread!!.setPause(true)
|
||||
}
|
||||
return stopRecognizerThread()
|
||||
}
|
||||
|
||||
/**
|
||||
* Shutdown the recognizer and release the recorder
|
||||
*/
|
||||
fun shutdown() {
|
||||
recorder.release()
|
||||
}
|
||||
|
||||
fun setPause(paused: Boolean) {
|
||||
if (recognizerThread != null) {
|
||||
recognizerThread!!.setPause(paused)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Resets recognizer in a thread, starts recognition over again
|
||||
*/
|
||||
fun reset() {
|
||||
if (recognizerThread != null) {
|
||||
recognizerThread!!.reset()
|
||||
}
|
||||
}
|
||||
|
||||
private inner class RecognizerThread @JvmOverloads constructor(
|
||||
var listener: RecognitionListener,
|
||||
timeout: Int = Companion.NO_TIMEOUT
|
||||
) : Thread() {
|
||||
private var remainingSamples: Int
|
||||
private val timeoutSamples: Int
|
||||
|
||||
@Volatile
|
||||
private var paused = false
|
||||
|
||||
@Volatile
|
||||
private var reset = false
|
||||
|
||||
init {
|
||||
timeoutSamples = if (timeout != Companion.NO_TIMEOUT) {
|
||||
timeout * sampleRate / 1000
|
||||
} else {
|
||||
Companion.NO_TIMEOUT
|
||||
}
|
||||
remainingSamples = timeoutSamples
|
||||
}
|
||||
|
||||
/**
|
||||
* When we are paused, don't process audio by the recognizer and don't emit
|
||||
* any listener results
|
||||
*
|
||||
* @param paused the status of pause
|
||||
*/
|
||||
fun setPause(paused: Boolean) {
|
||||
this.paused = paused
|
||||
}
|
||||
|
||||
/**
|
||||
* Set reset state to signal reset of the recognizer and start over
|
||||
*/
|
||||
fun reset() {
|
||||
reset = true
|
||||
}
|
||||
|
||||
override fun run() {
|
||||
recorder.startRecording()
|
||||
if (recorder.recordingState == AudioRecord.RECORDSTATE_STOPPED) {
|
||||
recorder.stop()
|
||||
val ioe = IOException(
|
||||
"Failed to start recording. Microphone might be already in use."
|
||||
)
|
||||
mainHandler.post { listener.onError(ioe) }
|
||||
}
|
||||
val buffer = ShortArray(bufferSize)
|
||||
while (!interrupted()
|
||||
&& (timeoutSamples == Companion.NO_TIMEOUT || remainingSamples > 0)
|
||||
) {
|
||||
val nread = recorder.read(buffer, 0, buffer.size)
|
||||
if (paused) {
|
||||
continue
|
||||
}
|
||||
if (reset) {
|
||||
recognizer.reset()
|
||||
reset = false
|
||||
}
|
||||
if (nread < 0) throw RuntimeException("error reading audio buffer")
|
||||
if (recognizer.acceptWaveform(buffer)) {
|
||||
val result = recognizer.result
|
||||
mainHandler.post { listener.onResult(result) }
|
||||
} else {
|
||||
val partialResult = recognizer.partialResult
|
||||
mainHandler.post { listener.onPartialResult(partialResult) }
|
||||
}
|
||||
if (timeoutSamples != NO_TIMEOUT) {
|
||||
remainingSamples -= nread
|
||||
}
|
||||
}
|
||||
recorder.stop()
|
||||
if (!paused) {
|
||||
// If we met timeout signal that speech ended
|
||||
if (timeoutSamples != NO_TIMEOUT && remainingSamples <= 0) {
|
||||
mainHandler.post { listener.onTimeout() }
|
||||
} else {
|
||||
val finalResult = recognizer.finalResult
|
||||
mainHandler.post { listener.onFinalResult(finalResult) }
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
companion object {
|
||||
private const val NO_TIMEOUT = -1
|
||||
private const val BUFFER_SIZE_SECONDS = 0.2f
|
||||
}
|
||||
}
|
||||
@@ -1,151 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package org.vosk.android
|
||||
|
||||
import android.os.Handler
|
||||
import android.os.Looper
|
||||
import org.vosk.Recognizer
|
||||
import java.io.IOException
|
||||
import java.io.InputStream
|
||||
import kotlin.math.roundToInt
|
||||
|
||||
/**
|
||||
* Service that recognizes stream audio in a thread, passes it to a recognizer and emits
|
||||
* recognition results. Recognition events are passed to a client using
|
||||
* [RecognitionListener]
|
||||
*/
|
||||
class SpeechStreamService(
|
||||
private val recognizer: Recognizer,
|
||||
inputStream: InputStream,
|
||||
sampleRate: Float
|
||||
) {
|
||||
private val inputStream: InputStream
|
||||
private val sampleRate: Int
|
||||
private val bufferSize: Int
|
||||
private var recognizerThread: Thread? = null
|
||||
private val mainHandler = Handler(Looper.getMainLooper())
|
||||
|
||||
/**
|
||||
* Creates speech service.
|
||||
*/
|
||||
init {
|
||||
this.sampleRate = sampleRate.toInt()
|
||||
this.inputStream = inputStream
|
||||
bufferSize = (this.sampleRate * BUFFER_SIZE_SECONDS * 2).roundToInt()
|
||||
}
|
||||
|
||||
/**
|
||||
* Starts recognition. Does nothing if recognition is active.
|
||||
*
|
||||
* @return true if recognition was actually started
|
||||
*/
|
||||
fun start(listener: RecognitionListener): Boolean {
|
||||
if (null != recognizerThread) return false
|
||||
recognizerThread = RecognizerThread(listener)
|
||||
recognizerThread!!.start()
|
||||
return true
|
||||
}
|
||||
|
||||
/**
|
||||
* Starts recognition. After specified timeout listening stops and the
|
||||
* endOfSpeech signals about that. Does nothing if recognition is active.
|
||||
*
|
||||
*
|
||||
* timeout - timeout in milliseconds to listen.
|
||||
*
|
||||
* @return true if recognition was actually started
|
||||
*/
|
||||
fun start(listener: RecognitionListener, timeout: Int): Boolean {
|
||||
if (null != recognizerThread) return false
|
||||
recognizerThread = RecognizerThread(listener, timeout)
|
||||
recognizerThread!!.start()
|
||||
return true
|
||||
}
|
||||
|
||||
/**
|
||||
* Stops recognition. All listeners should receive final result if there is
|
||||
* any. Does nothing if recognition is not active.
|
||||
*
|
||||
* @return true if recognition was actually stopped
|
||||
*/
|
||||
fun stop(): Boolean {
|
||||
if (null == recognizerThread) return false
|
||||
try {
|
||||
recognizerThread!!.interrupt()
|
||||
recognizerThread!!.join()
|
||||
} catch (e: InterruptedException) {
|
||||
// Restore the interrupted status.
|
||||
Thread.currentThread().interrupt()
|
||||
}
|
||||
recognizerThread = null
|
||||
return true
|
||||
}
|
||||
|
||||
private inner class RecognizerThread @JvmOverloads constructor(
|
||||
var listener: RecognitionListener,
|
||||
timeout: Int = Companion.NO_TIMEOUT
|
||||
) : Thread() {
|
||||
private var remainingSamples: Int
|
||||
private val timeoutSamples: Int
|
||||
|
||||
init {
|
||||
if (timeout != Companion.NO_TIMEOUT) timeoutSamples =
|
||||
timeout * sampleRate / 1000 else timeoutSamples = Companion.NO_TIMEOUT
|
||||
remainingSamples = timeoutSamples
|
||||
}
|
||||
|
||||
override fun run() {
|
||||
val buffer = ByteArray(bufferSize)
|
||||
while (!interrupted()
|
||||
&& (timeoutSamples == Companion.NO_TIMEOUT || remainingSamples > 0)
|
||||
) {
|
||||
try {
|
||||
val nread = inputStream.read(buffer, 0, buffer.size)
|
||||
if (nread < 0) {
|
||||
break
|
||||
} else {
|
||||
val isSilence: Boolean = recognizer.acceptWaveform(buffer)
|
||||
if (isSilence) {
|
||||
val result = recognizer.result
|
||||
mainHandler.post { listener.onResult(result) }
|
||||
} else {
|
||||
val partialResult = recognizer.partialResult
|
||||
mainHandler.post { listener.onPartialResult(partialResult) }
|
||||
}
|
||||
}
|
||||
if (timeoutSamples != NO_TIMEOUT) {
|
||||
remainingSamples -= nread
|
||||
}
|
||||
} catch (e: IOException) {
|
||||
mainHandler.post { listener.onError(e) }
|
||||
}
|
||||
}
|
||||
|
||||
// If we met timeout signal that speech ended
|
||||
if (timeoutSamples != NO_TIMEOUT && remainingSamples <= 0) {
|
||||
mainHandler.post { listener.onTimeout() }
|
||||
} else {
|
||||
val finalResult = recognizer.finalResult
|
||||
mainHandler.post { listener.onFinalResult(finalResult) }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
companion object {
|
||||
private const val NO_TIMEOUT = -1
|
||||
private const val BUFFER_SIZE_SECONDS = 0.2f
|
||||
}
|
||||
}
|
||||
@@ -1,138 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
package org.vosk.android
|
||||
|
||||
import android.content.Context
|
||||
import android.content.res.AssetManager
|
||||
import android.os.Environment
|
||||
import android.os.Handler
|
||||
import android.os.Looper
|
||||
import android.util.Log
|
||||
import org.vosk.Model
|
||||
import java.io.*
|
||||
import java.util.concurrent.Executor
|
||||
import java.util.concurrent.Executors
|
||||
import java.util.function.Consumer
|
||||
|
||||
/**
|
||||
* Provides utility methods to sync model files to external storage to allow
|
||||
* C++ code access them. Relies on file named "uuid" to track updates.
|
||||
*/
|
||||
object StorageService {
|
||||
private val TAG = StorageService::class.simpleName
|
||||
|
||||
@JvmStatic
|
||||
fun unpack(
|
||||
context: Context,
|
||||
sourcePath: String,
|
||||
targetPath: String,
|
||||
completeCallback: Consumer<Model>,
|
||||
errorCallback: Consumer<IOException>
|
||||
) {
|
||||
val executor: Executor =
|
||||
Executors.newSingleThreadExecutor() // change according to your requirements
|
||||
val handler = Handler(Looper.getMainLooper())
|
||||
executor.execute {
|
||||
try {
|
||||
val outputPath = sync(context, sourcePath, targetPath)
|
||||
val model = Model(outputPath)
|
||||
handler.post { completeCallback.accept(model) }
|
||||
} catch (e: IOException) {
|
||||
handler.post { errorCallback.accept(e) }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@JvmStatic
|
||||
@Throws(IOException::class)
|
||||
fun sync(context: Context, sourcePath: String, targetPath: String): String {
|
||||
val assetManager = context.assets
|
||||
val externalFilesDir = context.getExternalFilesDir(null)
|
||||
?: throw IOException(
|
||||
"cannot get external files dir, "
|
||||
+ "external storage state is " + Environment.getExternalStorageState()
|
||||
)
|
||||
val targetDir = File(externalFilesDir, targetPath)
|
||||
val resultPath = File(targetDir, sourcePath).absolutePath
|
||||
val sourceUUID = readLine(assetManager.open("$sourcePath/uuid"))
|
||||
try {
|
||||
val targetUUID = readLine(FileInputStream(File(targetDir, "$sourcePath/uuid")))
|
||||
if (targetUUID == sourceUUID) return resultPath
|
||||
} catch (e: FileNotFoundException) {
|
||||
// ignore
|
||||
}
|
||||
deleteContents(targetDir)
|
||||
copyAssets(assetManager, sourcePath, targetDir)
|
||||
|
||||
// Copy uuid
|
||||
copyFile(assetManager, "$sourcePath/uuid", targetDir)
|
||||
return resultPath
|
||||
}
|
||||
|
||||
@Throws(IOException::class)
|
||||
private fun readLine(inputStream: InputStream): String {
|
||||
return BufferedReader(InputStreamReader(inputStream)).use { it.readLine() }
|
||||
}
|
||||
|
||||
private fun deleteContents(dir: File): Boolean {
|
||||
val files = dir.listFiles()
|
||||
var success = true
|
||||
if (files != null) {
|
||||
for (file in files) {
|
||||
if (file.isDirectory) {
|
||||
success = success and deleteContents(file)
|
||||
}
|
||||
if (!file.delete()) {
|
||||
success = false
|
||||
}
|
||||
}
|
||||
}
|
||||
return success
|
||||
}
|
||||
|
||||
@Throws(IOException::class)
|
||||
private fun copyAssets(assetManager: AssetManager, path: String, outPath: File) {
|
||||
val assets = assetManager.list(path) ?: return
|
||||
if (assets.isEmpty()) {
|
||||
if (!path.endsWith("uuid")) copyFile(assetManager, path, outPath)
|
||||
} else {
|
||||
val dir = File(outPath, path)
|
||||
if (!dir.exists()) {
|
||||
Log.v(TAG, "Making directory " + dir.absolutePath)
|
||||
if (!dir.mkdirs()) {
|
||||
Log.v(TAG, "Failed to create directory " + dir.absolutePath)
|
||||
}
|
||||
}
|
||||
for (asset in assets) {
|
||||
copyAssets(assetManager, "$path/$asset", outPath)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Throws(IOException::class)
|
||||
private fun copyFile(assetManager: AssetManager, fileName: String, outPath: File) {
|
||||
Log.v(TAG, "Copy $fileName to $outPath")
|
||||
assetManager.open(fileName).use { inputStream ->
|
||||
FileOutputStream("$outPath/$fileName").use { out ->
|
||||
val buffer = ByteArray(4000)
|
||||
var read: Int
|
||||
while (inputStream.read(buffer).also { read = it } != -1) {
|
||||
out.write(buffer, 0, read)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,22 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
/**
|
||||
* Thrown when a [Recognizer] cannot accept a given waveform
|
||||
*/
|
||||
class AcceptWaveformException(data: Any) : Exception("Could not accept waveform: $data")
|
||||
@@ -1,38 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import org.vosk.exception.ModelException
|
||||
|
||||
/**
|
||||
* Batch model object
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
* @constructor Creates the batch recognizer object
|
||||
*/
|
||||
expect class BatchModel @Throws(ModelException::class) constructor(path: String) : Freeable {
|
||||
|
||||
/**
|
||||
* Releases batch model object
|
||||
*/
|
||||
override fun free()
|
||||
|
||||
/**
|
||||
* Wait for the processing
|
||||
*/
|
||||
fun await()
|
||||
}
|
||||
@@ -1,62 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
/**
|
||||
* Batch recognizer object
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
* @constructor Creates batch recognizer object
|
||||
*/
|
||||
expect class BatchRecognizer constructor(model: BatchModel, sampleRate: Float) : Freeable {
|
||||
|
||||
/**
|
||||
* Releases batch recognizer object
|
||||
*/
|
||||
override fun free()
|
||||
|
||||
/**
|
||||
* Accept batch voice data
|
||||
*/
|
||||
fun acceptWaveform(data: ByteArray)
|
||||
|
||||
/**
|
||||
* Set NLSML output
|
||||
* @param nlsml - boolean value
|
||||
*/
|
||||
fun setNLSML(nlsml: Boolean)
|
||||
|
||||
/**
|
||||
* Closes the stream
|
||||
*/
|
||||
fun finishStream()
|
||||
|
||||
/**
|
||||
* Return results
|
||||
*/
|
||||
val frontResult: String
|
||||
|
||||
/**
|
||||
* Release and free first retrieved result
|
||||
*/
|
||||
fun pop()
|
||||
|
||||
/**
|
||||
* Get amount of pending chunks for more intelligent waiting
|
||||
*/
|
||||
val pendingChunks: Int
|
||||
}
|
||||
@@ -1,27 +0,0 @@
|
||||
/*
|
||||
* Copyright 2024 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
/**
|
||||
* VoskEpMode
|
||||
*/
|
||||
enum class EndPointerMode {
|
||||
ANSWER_DEFAULT,
|
||||
ANSWER_SHORT,
|
||||
ANSWER_LONG,
|
||||
ANSWER_VERY_LONG
|
||||
}
|
||||
@@ -1,31 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
/**
|
||||
* Denotes an object that must be freed afterwards.
|
||||
*
|
||||
* On JVM, This is done via AutoClosable.
|
||||
*/
|
||||
@Suppress("SpellCheckingInspection")
|
||||
interface Freeable {
|
||||
|
||||
/**
|
||||
* Dereference the related object
|
||||
*/
|
||||
fun free()
|
||||
}
|
||||
@@ -1,40 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
/**
|
||||
* Log level for Kaldi messages.
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
*/
|
||||
enum class LogLevel(val value: Int) {
|
||||
|
||||
/**
|
||||
* Don't print info messages
|
||||
*/
|
||||
WARNINGS(-1),
|
||||
|
||||
/**
|
||||
* Default value to print info and error messages but no debug
|
||||
*/
|
||||
INFO(0),
|
||||
|
||||
/**
|
||||
* More verbose mode
|
||||
*/
|
||||
DEBUG(1);
|
||||
}
|
||||
@@ -1,52 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import org.vosk.exception.IOException
|
||||
|
||||
|
||||
/**
|
||||
* Model stores all the data required for recognition
|
||||
*
|
||||
* It contains static data and can be shared across processing
|
||||
* threads.
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
* @constructor Loads model data from the file and returns the model object
|
||||
* @param path the path of the model on the filesystem
|
||||
* @throws IOException if the path provided is invalid
|
||||
*/
|
||||
expect class Model @Throws(IOException::class) constructor(path: String) : Freeable {
|
||||
|
||||
/**
|
||||
* Check if a word can be recognized by the model
|
||||
* @param word: the word
|
||||
* @returns the word symbol if @param word exists inside the model
|
||||
* or -1 otherwise.
|
||||
* Reminding that word symbol 0 is for <epsilon>
|
||||
*/
|
||||
fun findWord(word: String): Int
|
||||
|
||||
/**
|
||||
* Releases the model memory
|
||||
*
|
||||
* The model object is reference-counted so if some recognizer
|
||||
* depends on this model, model might still stay alive. When
|
||||
* last recognizer is released, model will be released too.
|
||||
*/
|
||||
override fun free()
|
||||
}
|
||||
@@ -1,273 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import org.vosk.exception.RecognizerException
|
||||
|
||||
/**
|
||||
* Recognizer object is the main object which processes data.
|
||||
*
|
||||
* Each recognizer usually runs in own thread and takes audio as input.
|
||||
* Once audio is processed recognizer returns JSON object as a string
|
||||
* which represent decoded information - words, confidences, times, n-best lists,
|
||||
* speaker information and so on
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
*/
|
||||
expect class Recognizer : Freeable {
|
||||
/**
|
||||
* Creates the recognizer object
|
||||
*
|
||||
* The recognizers process the speech and return text using shared model data
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The sample rate of the audio you going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @throws RecognizerException if a problem occurred
|
||||
*/
|
||||
@Throws(RecognizerException::class)
|
||||
constructor(model: Model, sampleRate: Float)
|
||||
|
||||
/**
|
||||
* Creates the recognizer object with speaker recognition
|
||||
*
|
||||
* With the speaker recognition mode the recognizer not just recognize
|
||||
* text but also return speaker vectors one can use for speaker identification
|
||||
*
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The sample rate of the audio you going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @param speakerModel speaker model for speaker identification
|
||||
* @throws RecognizerException if a problem occurred
|
||||
*/
|
||||
@Throws(RecognizerException::class)
|
||||
constructor(model: Model, sampleRate: Float, speakerModel: SpeakerModel)
|
||||
|
||||
/**
|
||||
* Creates the recognizer object with the phrase list
|
||||
*
|
||||
* Sometimes when you want to improve recognition accuracy and when you don't need
|
||||
* to recognize large vocabulary you can specify a list of phrases to recognize. This
|
||||
* will improve recognizer speed and accuracy but might return [unk] if user said
|
||||
* something different.
|
||||
*
|
||||
* Only recognizers with lookahead models support this type of quick configuration.
|
||||
* Precompiled HCLG graph models are not supported.
|
||||
*
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The valuesample rate of the audio you going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @param grammar The string with the list of phrases to recognize as JSON array of strings,
|
||||
* for example "["one two three four five", "[unk]"]".
|
||||
*
|
||||
* @throws RecognizerException if a problem occurred
|
||||
*/
|
||||
@Throws(RecognizerException::class)
|
||||
constructor(model: Model, sampleRate: Float, grammar: String)
|
||||
|
||||
/**
|
||||
* Adds speaker model to already initialized recognizer
|
||||
*
|
||||
* Can add speaker recognition model to already created recognizer. Helps to initialize
|
||||
* speaker recognition for grammar-based recognizer.
|
||||
*
|
||||
* @param speakerModel Speaker recognition model
|
||||
*/
|
||||
fun setSpeakerModel(speakerModel: SpeakerModel)
|
||||
|
||||
|
||||
/**
|
||||
* Reconfigures recognizer to use grammar
|
||||
*
|
||||
* @param grammar Set of phrases in JSON array of strings or "[]" to use default model graph.
|
||||
* See also vosk_recognizer_new_grm
|
||||
*/
|
||||
fun setGrammar(grammar: String)
|
||||
|
||||
/**
|
||||
* Configures recognizer to output n-best results
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "alternatives": [
|
||||
* { "text": "one two three four five", "confidence": 0.97 },
|
||||
* { "text": "one two three for five", "confidence": 0.03 },
|
||||
* ]
|
||||
* }
|
||||
* </pre>
|
||||
*
|
||||
* @param maxAlternatives - maximum alternatives to return from recognition results
|
||||
*/
|
||||
fun setMaxAlternatives(maxAlternatives: Int)
|
||||
|
||||
/**
|
||||
* Enables words with times in the output
|
||||
*
|
||||
* <pre>
|
||||
* "result" : [{
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.110000,
|
||||
* "start" : 0.870000,
|
||||
* "word" : "what"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.530000,
|
||||
* "start" : 1.110000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.950000,
|
||||
* "start" : 1.530000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 2.340000,
|
||||
* "start" : 1.950000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 2.610000,
|
||||
* "start" : 2.340000,
|
||||
* "word" : "one"
|
||||
* }],
|
||||
* </pre>
|
||||
*
|
||||
* C equivalent = vosk_recognizer_set_words
|
||||
* @param words - boolean value
|
||||
*/
|
||||
fun setOutputWordTimes(words: Boolean)
|
||||
|
||||
/**
|
||||
* Like [setOutputWordTimes] return words and confidences in partial results
|
||||
*
|
||||
* @param partialWords - boolean value
|
||||
*/
|
||||
fun setPartialWords(partialWords: Boolean)
|
||||
|
||||
/**
|
||||
* Set NLSML output
|
||||
* @param nlsml - boolean value
|
||||
*/
|
||||
fun setNLSML(nlsml: Boolean)
|
||||
|
||||
|
||||
/**
|
||||
* Set endpointer scaling factor
|
||||
*
|
||||
* @param mode Endpointer mode
|
||||
**/
|
||||
fun setEndPointerMode(mode: EndPointerMode)
|
||||
|
||||
/**
|
||||
* Set endpointer delays
|
||||
*
|
||||
* @param tStartMax timeout for stopping recognition in case of initial silence (usually around 5.0)
|
||||
* @param tEnd timeout for stopping recognition in milliseconds after we recognized something (usually around 0.5 - 1.0)
|
||||
* @param tMax timeout for forcing utterance end in milliseconds (usually around 20-30)
|
||||
**/
|
||||
fun setEndPointerDelays(tStartMax: Float, tEnd: Float, tMax: Float)
|
||||
|
||||
/**
|
||||
* Accept voice data
|
||||
*
|
||||
* accept and process new chunk of voice data
|
||||
*
|
||||
* @param data Audio data in PCM 16-bit mono format.
|
||||
* @param length Length of the audio data.
|
||||
* @returns
|
||||
* 1 - If silence is occurred and you can retrieve a new utterance with result method
|
||||
* 0 - If decoding continues
|
||||
* -1 - If exception occurred
|
||||
*/
|
||||
@Throws(AcceptWaveformException::class)
|
||||
fun acceptWaveform(data: ByteArray): Boolean
|
||||
|
||||
/**
|
||||
* Same as [acceptWaveform] but the version with the short data for language bindings where you have
|
||||
* audio as array of shorts
|
||||
*/
|
||||
@Throws(AcceptWaveformException::class)
|
||||
fun acceptWaveform(data: ShortArray): Boolean
|
||||
|
||||
/**
|
||||
* Same as [acceptWaveform] but the version with the float data for language bindings where you have
|
||||
* audio as array of floats
|
||||
*/
|
||||
@Throws(AcceptWaveformException::class)
|
||||
fun acceptWaveform(data: FloatArray): Boolean
|
||||
|
||||
/**
|
||||
* Returns speech recognition result
|
||||
*
|
||||
* @returns the result in JSON format which contains decoded line, decoded
|
||||
* words, times in seconds and confidences. You can parse this result
|
||||
* with any json parser
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "text" : "what zero zero zero one"
|
||||
* }
|
||||
* </pre>
|
||||
*
|
||||
* If alternatives enabled it returns result with alternatives, see also [setMaxAlternatives].
|
||||
*
|
||||
* If word times enabled returns word time, see also [setOutputWordTimes].
|
||||
*/
|
||||
val result: String
|
||||
|
||||
/**
|
||||
* Returns partial speech recognition
|
||||
*
|
||||
* @returns partial speech recognition text which is not yet finalized.
|
||||
* result may change as recognizer process more data.
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "partial" : "cyril one eight zero"
|
||||
* }
|
||||
* </pre>
|
||||
*/
|
||||
val finalResult: String
|
||||
|
||||
/**
|
||||
* Returns speech recognition result. Same as result, but doesn't wait for silence
|
||||
* You usually call it in the end of the stream to get final bits of audio. It
|
||||
* flushes the feature pipeline, so all remaining audio chunks got processed.
|
||||
*
|
||||
* @returns speech result in JSON format.
|
||||
*/
|
||||
val partialResult: String
|
||||
|
||||
/**
|
||||
* Resets the recognizer
|
||||
*
|
||||
* Resets current results so the recognition can continue from scratch
|
||||
*/
|
||||
fun reset()
|
||||
|
||||
/**
|
||||
* Releases recognizer object
|
||||
*
|
||||
* Underlying model is also unreferenced and if needed released
|
||||
*/
|
||||
override fun free()
|
||||
}
|
||||
@@ -1,40 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import org.vosk.exception.ModelException
|
||||
|
||||
/**
|
||||
* Speaker model is the same as model but contains the data
|
||||
* for speaker identification.
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
* @constructor Loads speaker model data from the file and returns the model object
|
||||
* @param path the path of the model on the filesystem
|
||||
* @throws ModelException if the path provided is invalid
|
||||
*/
|
||||
expect class SpeakerModel @Throws(ModelException::class) constructor(path: String) : Freeable {
|
||||
|
||||
/**
|
||||
* Releases the model memory
|
||||
*
|
||||
* The model object is reference-counted so if some recognizer
|
||||
* depends on this model, model might still stay alive. When
|
||||
* last recognizer is released, model will be released too.
|
||||
*/
|
||||
override fun free()
|
||||
}
|
||||
@@ -1,32 +0,0 @@
|
||||
/*
|
||||
* Copyright 2024 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
/**
|
||||
* Inverse text normalization
|
||||
*
|
||||
* @since 2024/06/19
|
||||
* @constructor Create text processor
|
||||
*/
|
||||
expect class TextProcessor constructor(tagger: Char, verbalizer: Char) : Freeable {
|
||||
|
||||
/** Release text processor */
|
||||
override fun free()
|
||||
|
||||
/** Convert string */
|
||||
fun itn(input: Char): Char
|
||||
}
|
||||
@@ -1,46 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
/**
|
||||
* 26 / 12 / 2022
|
||||
*
|
||||
* Control overarching features of libvosk.
|
||||
*/
|
||||
expect object Vosk {
|
||||
/**
|
||||
* Set log level for Kaldi messages
|
||||
*
|
||||
* @param logLevel the level
|
||||
*/
|
||||
fun setLogLevel(logLevel: LogLevel)
|
||||
|
||||
/**
|
||||
* Init, automatically select a CUDA device and allow multithreading.
|
||||
* Must be called once from the main thread.
|
||||
* Has no effect if HAVE_CUDA flag is not set.
|
||||
*/
|
||||
fun gpuInit()
|
||||
|
||||
|
||||
/**
|
||||
* Init CUDA device in a multi-threaded environment.
|
||||
* Must be called for each thread.
|
||||
* Has no effect if HAVE_CUDA flag is not set.
|
||||
*/
|
||||
fun gpuThreadInit()
|
||||
}
|
||||
@@ -1,22 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk.exception
|
||||
|
||||
/**
|
||||
* Internal common IO exception. On JVM this is just a type alias.
|
||||
*/
|
||||
expect open class IOException(message: String?) : Exception
|
||||
@@ -1,22 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk.exception
|
||||
|
||||
/**
|
||||
* Thrown when there is an exception creating a model.
|
||||
*/
|
||||
class ModelException(path: String): IOException("Failed to find model: $path")
|
||||
@@ -1,22 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk.exception
|
||||
|
||||
/**
|
||||
* Thrown when the recognizer fails to be created
|
||||
*/
|
||||
class RecognizerException: IOException("Failed to create recognizer.")
|
||||
@@ -1,31 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk.json
|
||||
|
||||
import kotlinx.serialization.Serializable
|
||||
|
||||
/**
|
||||
* Represents an alternative result.
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
*/
|
||||
@Serializable
|
||||
data class Alternative(
|
||||
val confidence: Double,
|
||||
val result: List<Result> = emptyList(),
|
||||
val text: String
|
||||
)
|
||||
@@ -1,61 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk.json
|
||||
|
||||
import kotlinx.serialization.decodeFromString
|
||||
import kotlinx.serialization.json.Json
|
||||
import kotlinx.serialization.json.encodeToJsonElement
|
||||
import org.vosk.Recognizer
|
||||
import org.vosk.Model
|
||||
import org.vosk.exception.RecognizerException
|
||||
|
||||
/*
|
||||
* 26 / 12 / 2022
|
||||
*/
|
||||
|
||||
/**
|
||||
* Vosk JSON encoder
|
||||
*/
|
||||
val voskJson = Json { encodeDefaults = true }
|
||||
|
||||
/**
|
||||
* Get the result as a JSON object
|
||||
*/
|
||||
fun Recognizer.resultAsJson(): ResultOutput =
|
||||
voskJson.decodeFromString(result)
|
||||
|
||||
/**
|
||||
* Get the final result as a JSON object
|
||||
*/
|
||||
fun Recognizer.finalResultAsJson(): ResultOutput =
|
||||
voskJson.decodeFromString(finalResult)
|
||||
|
||||
/**
|
||||
* Get the partial result as a JSON object
|
||||
*/
|
||||
fun Recognizer.partialResultAsJson(): PartialResultOutput =
|
||||
voskJson.decodeFromString(partialResult)
|
||||
|
||||
|
||||
/**
|
||||
* Create a [Recognizer], but using a list for grammar instead.
|
||||
*
|
||||
* The grammar list is converted into a JSON array
|
||||
*/
|
||||
@Throws(RecognizerException::class)
|
||||
fun Recognizer(model: Model, sampleRate: Float, grammar: List<String>) =
|
||||
Recognizer(model, sampleRate, voskJson.encodeToJsonElement(grammar).toString())
|
||||
@@ -1,32 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk.json
|
||||
|
||||
import kotlinx.serialization.SerialName
|
||||
import kotlinx.serialization.Serializable
|
||||
|
||||
/**
|
||||
* Represents a partial result
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
*/
|
||||
@Serializable
|
||||
data class PartialResultOutput(
|
||||
val partial: String,
|
||||
@SerialName("partial_result")
|
||||
val partialResult: List<Result> = emptyList(),
|
||||
)
|
||||
@@ -1,32 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk.json
|
||||
|
||||
import kotlinx.serialization.Serializable
|
||||
|
||||
/**
|
||||
* Represents a result for any given word.
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
*/
|
||||
@Serializable
|
||||
data class Result(
|
||||
val conf: Double? = null,
|
||||
val end: Double,
|
||||
val start: Double,
|
||||
val word: String,
|
||||
)
|
||||
@@ -1,31 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk.json
|
||||
|
||||
import kotlinx.serialization.Serializable
|
||||
|
||||
/**
|
||||
* Result output
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
*/
|
||||
@Serializable
|
||||
data class ResultOutput(
|
||||
val alternatives: List<Alternative> = emptyList(),
|
||||
val result: List<Result> = emptyList(),
|
||||
val text: String? = null
|
||||
)
|
||||
@@ -1,38 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk.json
|
||||
|
||||
/**
|
||||
* For extension functions transforming (input streams->recognizers) into flows.
|
||||
*/
|
||||
sealed interface WaveformResult {
|
||||
|
||||
/**
|
||||
* A result made after a period of silence.
|
||||
*/
|
||||
data class Result(val result: ResultOutput) : WaveformResult
|
||||
|
||||
/**
|
||||
* A partial result that is being made in progress.
|
||||
*/
|
||||
data class PartialResult(val result: PartialResultOutput) : WaveformResult
|
||||
|
||||
/**
|
||||
* A final result made after there is no more content to feed.
|
||||
*/
|
||||
data class FinalResult(val result: ResultOutput) : WaveformResult
|
||||
}
|
||||
@@ -1,83 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import com.sun.jna.PointerType
|
||||
import org.vosk.exception.ModelException
|
||||
import java.io.File
|
||||
import java.nio.file.Path
|
||||
import kotlin.io.path.absolutePathString
|
||||
|
||||
|
||||
/**
|
||||
* Batch model object
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
*/
|
||||
actual class BatchModel : Freeable, PointerType, AutoCloseable {
|
||||
|
||||
/**
|
||||
* Empty constructor for JNA
|
||||
*/
|
||||
constructor()
|
||||
|
||||
/**
|
||||
* Creates the batch recognizer object
|
||||
*/
|
||||
@Throws(ModelException::class)
|
||||
actual constructor(path: String) : super(
|
||||
LibVosk.vosk_batch_model_new(path) ?: throw ModelException(path)
|
||||
)
|
||||
|
||||
/**
|
||||
* Constructor using a Path, will retrieve absolutePath
|
||||
*
|
||||
* @param path to batch model
|
||||
*/
|
||||
@Throws(ModelException::class)
|
||||
constructor(path: Path) : this(path.absolutePathString())
|
||||
|
||||
/**
|
||||
* Constructor using a File, will retrieve absolutePath
|
||||
*
|
||||
* @param file to batch model
|
||||
*/
|
||||
@Throws(ModelException::class)
|
||||
constructor(file: File) : this(file.absolutePath)
|
||||
|
||||
/**
|
||||
* Releases batch model object
|
||||
*/
|
||||
actual override fun free() {
|
||||
LibVosk.vosk_batch_model_free(this)
|
||||
}
|
||||
|
||||
/**
|
||||
* Wait for the processing
|
||||
*/
|
||||
actual fun await() {
|
||||
LibVosk.vosk_batch_model_wait(this)
|
||||
}
|
||||
|
||||
/**
|
||||
* @see free
|
||||
*/
|
||||
override fun close() {
|
||||
free()
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,95 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import com.sun.jna.PointerType
|
||||
|
||||
|
||||
/**
|
||||
* Batch recognizer object
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
* @constructor Creates batch recognizer object
|
||||
*/
|
||||
actual class BatchRecognizer : Freeable, PointerType, AutoCloseable {
|
||||
|
||||
/**
|
||||
* Empty constructor for JNA
|
||||
*/
|
||||
constructor()
|
||||
|
||||
/**
|
||||
* Creates batch recognizer object
|
||||
*/
|
||||
actual constructor(model: BatchModel, sampleRate: Float) :
|
||||
super(LibVosk.vosk_batch_recognizer_new(model, sampleRate))
|
||||
|
||||
/**
|
||||
* Releases batch recognizer object
|
||||
*/
|
||||
actual override fun free() {
|
||||
LibVosk.vosk_batch_recognizer_free(this);
|
||||
}
|
||||
|
||||
/**
|
||||
* Accept batch voice data
|
||||
*/
|
||||
actual fun acceptWaveform(data: ByteArray) {
|
||||
LibVosk.vosk_batch_recognizer_accept_waveform(this, data, data.size)
|
||||
}
|
||||
|
||||
/**
|
||||
* Set NLSML output
|
||||
* @param nlsml - boolean value
|
||||
*/
|
||||
actual fun setNLSML(nlsml: Boolean) {
|
||||
LibVosk.vosk_batch_recognizer_set_nlsml(this, nlsml)
|
||||
}
|
||||
|
||||
/**
|
||||
* Closes the stream
|
||||
*/
|
||||
actual fun finishStream() {
|
||||
LibVosk.vosk_batch_recognizer_finish_stream(this)
|
||||
}
|
||||
|
||||
/**
|
||||
* Return results
|
||||
*/
|
||||
actual val frontResult: String
|
||||
get() = LibVosk.vosk_batch_recognizer_front_result(this)
|
||||
|
||||
/**
|
||||
* Release and free first retrieved result
|
||||
*/
|
||||
actual fun pop() {
|
||||
LibVosk.vosk_batch_recognizer_pop(this)
|
||||
}
|
||||
|
||||
/**
|
||||
* Get amount of pending chunks for more intelligent waiting
|
||||
*/
|
||||
actual val pendingChunks: Int
|
||||
get() = LibVosk.vosk_batch_recognizer_get_pending_chunks(this)
|
||||
|
||||
/**
|
||||
* @see free
|
||||
*/
|
||||
override fun close() {
|
||||
free()
|
||||
}
|
||||
}
|
||||
@@ -1,213 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import com.sun.jna.Native
|
||||
import com.sun.jna.Platform
|
||||
import com.sun.jna.Pointer
|
||||
import java.io.File
|
||||
import java.io.IOException
|
||||
import java.io.InputStream
|
||||
import java.nio.file.Files
|
||||
import java.nio.file.StandardCopyOption
|
||||
|
||||
/**
|
||||
* 26 / 12 / 2022
|
||||
*/
|
||||
@Suppress("FunctionName")
|
||||
internal object LibVosk {
|
||||
|
||||
@JvmStatic
|
||||
@Deprecated(
|
||||
"LibVosk is now for internal JNA, use Vosk instead",
|
||||
ReplaceWith("Vosk.setLogLevel(logLevel)", "org.vosk.Vosk")
|
||||
)
|
||||
fun setLogLevel(logLevel: LogLevel) {
|
||||
Vosk.setLogLevel(logLevel)
|
||||
}
|
||||
|
||||
@Throws(IOException::class)
|
||||
private fun unpackDll(targetDir: File, lib: String) {
|
||||
Vosk::class.java.getResourceAsStream("/win32-x86-64/$lib.dll")!!.use {
|
||||
Files.copy(
|
||||
it,
|
||||
File(targetDir, "$lib.dll").toPath(),
|
||||
StandardCopyOption.REPLACE_EXISTING
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
init {
|
||||
when {
|
||||
Platform.isAndroid() -> {
|
||||
Native.register(LibVosk::class.java, "vosk")
|
||||
}
|
||||
|
||||
Platform.isWindows() -> {
|
||||
// We have to unpack dependencies
|
||||
try {
|
||||
// To get a tmp folder we unpack small library and mark it for deletion
|
||||
val tmpFile: File = Native.extractFromResourcePath(
|
||||
"/win32-x86-64/empty",
|
||||
LibVosk::class.java.classLoader
|
||||
)
|
||||
val tmpDir = tmpFile.parentFile!!
|
||||
File(tmpDir, tmpFile.name + ".x").createNewFile()
|
||||
|
||||
// Now unpack dependencies
|
||||
unpackDll(tmpDir, "libwinpthread-1");
|
||||
unpackDll(tmpDir, "libgcc_s_seh-1");
|
||||
unpackDll(tmpDir, "libstdc++-6");
|
||||
|
||||
} catch (e: IOException) {
|
||||
// Nothing for now, it will fail on next step
|
||||
} finally {
|
||||
Native.register(LibVosk::class.java, "libvosk");
|
||||
}
|
||||
}
|
||||
|
||||
else -> {
|
||||
Native.register(LibVosk::class.java, "vosk");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
external fun vosk_model_new(path: String): Pointer?
|
||||
|
||||
external fun vosk_model_free(model: Model)
|
||||
|
||||
external fun vosk_model_find_word(model: Model, word: String): Int
|
||||
|
||||
|
||||
external fun vosk_spk_model_new(path: String): Pointer?
|
||||
|
||||
external fun vosk_spk_model_free(model: SpeakerModel)
|
||||
|
||||
|
||||
external fun vosk_recognizer_new(model: Model, sampleRate: Float): Pointer?
|
||||
|
||||
external fun vosk_recognizer_new_spk(
|
||||
model: Model,
|
||||
sampleRate: Float,
|
||||
spkModel: SpeakerModel
|
||||
): Pointer?
|
||||
|
||||
external fun vosk_recognizer_new_grm(
|
||||
model: Model,
|
||||
sampleRate: Float,
|
||||
grammar: String?
|
||||
): Pointer?
|
||||
|
||||
external fun vosk_recognizer_set_spk_model(recognizer: Recognizer, spk_model: SpeakerModel)
|
||||
|
||||
external fun vosk_recognizer_set_grm(recognizer: Recognizer, grammar: String)
|
||||
|
||||
external fun vosk_recognizer_set_max_alternatives(recognizer: Recognizer, maxAlternatives: Int)
|
||||
|
||||
external fun vosk_recognizer_set_words(recognizer: Recognizer, words: Boolean)
|
||||
|
||||
external fun vosk_recognizer_set_partial_words(recognizer: Recognizer, partial_words: Boolean)
|
||||
|
||||
external fun vosk_recognizer_set_nlsml(recognizer: Recognizer, nlsml: Boolean)
|
||||
|
||||
|
||||
external fun vosk_recognizer_accept_waveform(
|
||||
recognizer: Recognizer,
|
||||
data: ByteArray?,
|
||||
len: Int
|
||||
): Int
|
||||
|
||||
external fun vosk_recognizer_accept_waveform_s(
|
||||
recognizer: Recognizer,
|
||||
data: ShortArray?,
|
||||
len: Int
|
||||
): Int
|
||||
|
||||
external fun vosk_recognizer_accept_waveform_f(
|
||||
recognizer: Recognizer,
|
||||
data: FloatArray?,
|
||||
len: Int
|
||||
): Int
|
||||
|
||||
|
||||
external fun vosk_recognizer_result(recognizer: Recognizer): String
|
||||
|
||||
external fun vosk_recognizer_final_result(recognizer: Recognizer): String
|
||||
|
||||
external fun vosk_recognizer_partial_result(recognizer: Recognizer): String
|
||||
|
||||
|
||||
external fun vosk_recognizer_reset(recognizer: Recognizer)
|
||||
|
||||
external fun vosk_recognizer_free(recognizer: Recognizer)
|
||||
|
||||
external fun vosk_set_log_level(level: Int)
|
||||
|
||||
external fun vosk_gpu_init()
|
||||
|
||||
external fun vosk_gpu_thread_init()
|
||||
|
||||
|
||||
external fun vosk_batch_model_new(path: String): Pointer?
|
||||
|
||||
external fun vosk_batch_model_free(model: BatchModel)
|
||||
|
||||
external fun vosk_batch_model_wait(model: BatchModel)
|
||||
|
||||
external fun vosk_batch_recognizer_new(batchModel: BatchModel, sampleRate: Float): Pointer
|
||||
|
||||
external fun vosk_batch_recognizer_free(recognizer: BatchRecognizer)
|
||||
|
||||
external fun vosk_batch_recognizer_accept_waveform(
|
||||
recognizer: BatchRecognizer,
|
||||
data: ByteArray?,
|
||||
length: Int
|
||||
)
|
||||
|
||||
external fun vosk_batch_recognizer_set_nlsml(
|
||||
recognizer: BatchRecognizer,
|
||||
nlsml: Boolean
|
||||
)
|
||||
|
||||
external fun vosk_batch_recognizer_finish_stream(
|
||||
recognizer: BatchRecognizer
|
||||
)
|
||||
|
||||
external fun vosk_batch_recognizer_front_result(
|
||||
recognizer: BatchRecognizer
|
||||
): String
|
||||
|
||||
external fun vosk_batch_recognizer_pop(recognizer: BatchRecognizer)
|
||||
|
||||
external fun vosk_batch_recognizer_get_pending_chunks(recognizer: BatchRecognizer): Int
|
||||
|
||||
external fun vosk_text_processor_new(tagger: Char, verbalizer: Char): Pointer
|
||||
|
||||
external fun vosk_text_processor_free(processor: TextProcessor)
|
||||
|
||||
external fun vosk_text_processor_itn(processor: TextProcessor, input: Char): Char
|
||||
|
||||
external fun vosk_recognizer_set_endpointer_mode(recognizer: Recognizer, ordinal: Int)
|
||||
|
||||
external fun vosk_recognizer_set_endpointer_delays(
|
||||
recognizer: Recognizer,
|
||||
tStartMax: Float,
|
||||
tEnd: Float,
|
||||
tMax: Float
|
||||
)
|
||||
}
|
||||
@@ -1,95 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import com.sun.jna.PointerType
|
||||
import org.vosk.exception.IOException
|
||||
import org.vosk.exception.ModelException
|
||||
import java.io.File
|
||||
import java.nio.file.Path
|
||||
import kotlin.io.path.absolutePathString
|
||||
|
||||
|
||||
/**
|
||||
* Model stores all the data required for recognition
|
||||
*
|
||||
* It contains static data and can be shared across processing
|
||||
* threads.
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
* @constructor Loads model data from the file and returns the model object
|
||||
* @throws IOException if the path provided is invalid
|
||||
*/
|
||||
actual class Model : Freeable, PointerType, AutoCloseable {
|
||||
|
||||
/**
|
||||
* Empty constructor for JNA
|
||||
*/
|
||||
constructor()
|
||||
|
||||
/**
|
||||
* Loads model data from the file and returns the model object
|
||||
*/
|
||||
@Throws(IOException::class)
|
||||
actual constructor(path: String) : super(
|
||||
LibVosk.vosk_model_new(path) ?: throw ModelException(path)
|
||||
)
|
||||
|
||||
/**
|
||||
* Constructor using a Path, will retrieve absolutePath
|
||||
*
|
||||
* @param path to batch model
|
||||
*/
|
||||
@Throws(IOException::class)
|
||||
constructor(path: Path) : this(path.absolutePathString())
|
||||
|
||||
/**
|
||||
* Constructor using a File, will retrieve absolutePath
|
||||
*
|
||||
* @param file to batch model
|
||||
*/
|
||||
@Throws(IOException::class)
|
||||
constructor(file: File) : this(file.absolutePath)
|
||||
|
||||
/**
|
||||
* Check if a word can be recognized by the model
|
||||
* @param word: the word
|
||||
* @returns the word symbol if @param word exists inside the model
|
||||
* or -1 otherwise.
|
||||
* Reminding that word symbol 0 is for <epsilon>
|
||||
*/
|
||||
actual fun findWord(word: String): Int =
|
||||
LibVosk.vosk_model_find_word(this, word)
|
||||
|
||||
/**
|
||||
* Releases the model memory
|
||||
*
|
||||
* The model object is reference-counted so if some recognizer
|
||||
* depends on this model, model might still stay alive. When
|
||||
* last recognizer is released, model will be released too.
|
||||
*/
|
||||
actual override fun free() {
|
||||
LibVosk.vosk_model_free(this)
|
||||
}
|
||||
|
||||
/**
|
||||
* @see free
|
||||
*/
|
||||
override fun close() {
|
||||
free()
|
||||
}
|
||||
}
|
||||
@@ -1,353 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import com.sun.jna.PointerType
|
||||
import org.vosk.exception.RecognizerException
|
||||
|
||||
/**
|
||||
* Recognizer object is the main object which processes data.
|
||||
*
|
||||
* Each recognizer usually runs in own thread and takes audio as input.
|
||||
* Once audio is processed recognizer returns JSON object as a string
|
||||
* which represent decoded information - words, confidences, times, n-best lists,
|
||||
* speaker information and so on
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
*/
|
||||
actual class Recognizer : Freeable, PointerType, AutoCloseable {
|
||||
|
||||
/**
|
||||
* Empty constructor for JNA
|
||||
*/
|
||||
constructor()
|
||||
|
||||
/**
|
||||
* Creates the recognizer object
|
||||
*
|
||||
* The recognizers process the speech and return text using shared model data
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The sample rate of the audio you going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @throws RecognizerException if a problem occurred
|
||||
*/
|
||||
@Throws(RecognizerException::class)
|
||||
actual constructor(model: Model, sampleRate: Float) :
|
||||
super(LibVosk.vosk_recognizer_new(model, sampleRate) ?: throw RecognizerException())
|
||||
|
||||
/**
|
||||
* Creates the recognizer object with speaker recognition
|
||||
*
|
||||
* With the speaker recognition mode the recognizer not just recognize
|
||||
* text but also return speaker vectors one can use for speaker identification
|
||||
*
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The sample rate of the audio you going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @param speakerModel speaker model for speaker identification
|
||||
* @throws RecognizerException if a problem occurred
|
||||
*/
|
||||
@Throws(RecognizerException::class)
|
||||
actual constructor(
|
||||
model: Model,
|
||||
sampleRate: Float,
|
||||
speakerModel: SpeakerModel
|
||||
) : super(
|
||||
LibVosk.vosk_recognizer_new_spk(model, sampleRate, speakerModel)
|
||||
?: throw RecognizerException()
|
||||
)
|
||||
|
||||
/**
|
||||
* Creates the recognizer object with the phrase list
|
||||
*
|
||||
* Sometimes when you want to improve recognition accuracy and when you don't need
|
||||
* to recognize large vocabulary you can specify a list of phrases to recognize. This
|
||||
* will improve recognizer speed and accuracy but might return [unk] if user said
|
||||
* something different.
|
||||
*
|
||||
* Only recognizers with lookahead models support this type of quick configuration.
|
||||
* Precompiled HCLG graph models are not supported.
|
||||
*
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The valuesample rate of the audio you going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @param grammar The string with the list of phrases to recognize as JSON array of strings,
|
||||
* for example "["one two three four five", "[unk]"]".
|
||||
*
|
||||
* @throws RecognizerException if a problem occurred
|
||||
*/
|
||||
@Throws(RecognizerException::class)
|
||||
actual constructor(
|
||||
model: Model,
|
||||
sampleRate: Float,
|
||||
grammar: String
|
||||
) : super(
|
||||
LibVosk.vosk_recognizer_new_grm(model, sampleRate, grammar) ?: throw RecognizerException()
|
||||
)
|
||||
|
||||
/**
|
||||
* JVM analog of Kotlin extension
|
||||
* @see [org.vosk.json.Recognizer]
|
||||
*/
|
||||
@Throws(RecognizerException::class)
|
||||
constructor(
|
||||
model: Model,
|
||||
sampleRate: Float,
|
||||
grammar: List<String>
|
||||
) : super(
|
||||
// We need the full qualifer to avoid any build issues
|
||||
@Suppress("RemoveRedundantQualifierName")
|
||||
org.vosk.json.Recognizer(model, sampleRate, grammar).pointer
|
||||
)
|
||||
|
||||
/**
|
||||
* Adds speaker model to already initialized recognizer
|
||||
*
|
||||
* Can add speaker recognition model to already created recognizer. Helps to initialize
|
||||
* speaker recognition for grammar-based recognizer.
|
||||
*
|
||||
* @param speakerModel Speaker recognition model
|
||||
*/
|
||||
actual fun setSpeakerModel(speakerModel: SpeakerModel) {
|
||||
LibVosk.vosk_recognizer_set_spk_model(this, speakerModel)
|
||||
}
|
||||
|
||||
/**
|
||||
* Reconfigures recognizer to use grammar
|
||||
*
|
||||
* @param grammar Set of phrases in JSON array of strings or "[]" to use default model graph.
|
||||
* See also vosk_recognizer_new_grm
|
||||
*/
|
||||
actual fun setGrammar(grammar: String) {
|
||||
LibVosk.vosk_recognizer_set_grm(this, grammar)
|
||||
}
|
||||
|
||||
/**
|
||||
* Configures recognizer to output n-best results
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "alternatives": [
|
||||
* { "text": "one two three four five", "confidence": 0.97 },
|
||||
* { "text": "one two three for five", "confidence": 0.03 },
|
||||
* ]
|
||||
* }
|
||||
* </pre>
|
||||
*
|
||||
* @param maxAlternatives - maximum alternatives to return from recognition results
|
||||
*/
|
||||
actual fun setMaxAlternatives(maxAlternatives: Int) {
|
||||
LibVosk.vosk_recognizer_set_max_alternatives(this, maxAlternatives)
|
||||
}
|
||||
|
||||
/**
|
||||
* Enables words with times in the output
|
||||
*
|
||||
* <pre>
|
||||
* "result" : [{
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.110000,
|
||||
* "start" : 0.870000,
|
||||
* "word" : "what"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.530000,
|
||||
* "start" : 1.110000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.950000,
|
||||
* "start" : 1.530000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 2.340000,
|
||||
* "start" : 1.950000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 2.610000,
|
||||
* "start" : 2.340000,
|
||||
* "word" : "one"
|
||||
* }],
|
||||
* </pre>
|
||||
*
|
||||
* @param words - boolean value
|
||||
*/
|
||||
actual fun setOutputWordTimes(words: Boolean) {
|
||||
LibVosk.vosk_recognizer_set_words(this, words)
|
||||
}
|
||||
|
||||
/**
|
||||
* Like [setOutputWordTimes] return words and confidences in partial results
|
||||
*
|
||||
* @param partialWords - boolean value
|
||||
*/
|
||||
actual fun setPartialWords(partialWords: Boolean) {
|
||||
LibVosk.vosk_recognizer_set_partial_words(this, partialWords)
|
||||
}
|
||||
|
||||
/**
|
||||
* Set NLSML output
|
||||
* @param nlsml - boolean value
|
||||
*/
|
||||
actual fun setNLSML(nlsml: Boolean) {
|
||||
LibVosk.vosk_recognizer_set_nlsml(this, nlsml)
|
||||
}
|
||||
|
||||
/**
|
||||
* Accept voice data
|
||||
*
|
||||
* accept and process new chunk of voice data
|
||||
*
|
||||
* @param data Audio data in PCM 16-bit mono format.
|
||||
* @param length Length of the audio data.
|
||||
* @returns
|
||||
* 1 - If silence is occurred and you can retrieve a new utterance with result method
|
||||
* 0 - If decoding continues
|
||||
* -1 - If exception occurred
|
||||
*/
|
||||
@Throws(AcceptWaveformException::class)
|
||||
actual fun acceptWaveform(data: ByteArray): Boolean {
|
||||
val result = LibVosk.vosk_recognizer_accept_waveform(this, data, data.size)
|
||||
if (result == -1) throw AcceptWaveformException(data)
|
||||
return result == 1
|
||||
}
|
||||
|
||||
/**
|
||||
* Same as [acceptWaveform] but the version with the short data for language bindings where you have
|
||||
* audio as array of shorts
|
||||
*/
|
||||
@Throws(AcceptWaveformException::class)
|
||||
actual fun acceptWaveform(data: ShortArray): Boolean {
|
||||
val result = LibVosk.vosk_recognizer_accept_waveform_s(this, data, data.size)
|
||||
if (result == -1) throw AcceptWaveformException(data)
|
||||
return result == 1
|
||||
}
|
||||
|
||||
/**
|
||||
* Same as [acceptWaveform] but the version with the float data for language bindings where you have
|
||||
* audio as array of floats
|
||||
*/
|
||||
@Throws(AcceptWaveformException::class)
|
||||
actual fun acceptWaveform(data: FloatArray): Boolean {
|
||||
val result = LibVosk.vosk_recognizer_accept_waveform_f(this, data, data.size)
|
||||
if (result == -1) throw AcceptWaveformException(data)
|
||||
return result == 1
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Returns speech recognition result
|
||||
*
|
||||
* @returns the result in JSON format which contains decoded line, decoded
|
||||
* words, times in seconds and confidences. You can parse this result
|
||||
* with any json parser
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "text" : "what zero zero zero one"
|
||||
* }
|
||||
* </pre>
|
||||
*
|
||||
* If alternatives enabled it returns result with alternatives, see also [setMaxAlternatives].
|
||||
*
|
||||
* If word times enabled returns word time, see also [setOutputWordTimes].
|
||||
*/
|
||||
actual val result: String
|
||||
get() = LibVosk.vosk_recognizer_result(this)
|
||||
|
||||
/**
|
||||
* Returns partial speech recognition
|
||||
*
|
||||
* @returns partial speech recognition text which is not yet finalized.
|
||||
* result may change as recognizer process more data.
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "partial" : "cyril one eight zero"
|
||||
* }
|
||||
* </pre>
|
||||
*/
|
||||
actual val finalResult: String
|
||||
get() = LibVosk.vosk_recognizer_final_result(this)
|
||||
|
||||
/**
|
||||
* Returns speech recognition result. Same as result, but doesn't wait for silence
|
||||
* You usually call it in the end of the stream to get final bits of audio. It
|
||||
* flushes the feature pipeline, so all remaining audio chunks got processed.
|
||||
*
|
||||
* @returns speech result in JSON format.
|
||||
*/
|
||||
actual val partialResult: String
|
||||
get() = LibVosk.vosk_recognizer_partial_result(this)
|
||||
|
||||
/**
|
||||
* Resets the recognizer
|
||||
*
|
||||
* Resets current results so the recognition can continue from scratch
|
||||
*/
|
||||
actual fun reset() {
|
||||
LibVosk.vosk_recognizer_reset(this)
|
||||
}
|
||||
|
||||
/**
|
||||
* Releases recognizer object
|
||||
*
|
||||
* Underlying model is also unreferenced and if needed released
|
||||
*/
|
||||
actual override fun free() {
|
||||
LibVosk.vosk_recognizer_free(this)
|
||||
}
|
||||
|
||||
/**
|
||||
* @see free
|
||||
*/
|
||||
override fun close() {
|
||||
free()
|
||||
}
|
||||
|
||||
/**
|
||||
* Set endpointer scaling factor
|
||||
*
|
||||
* @param mode Endpointer mode
|
||||
**/
|
||||
actual fun setEndPointerMode(mode: EndPointerMode) {
|
||||
LibVosk.vosk_recognizer_set_endpointer_mode(this, mode.ordinal)
|
||||
}
|
||||
|
||||
/**
|
||||
* Set endpointer delays
|
||||
*
|
||||
* @param tStartMax timeout for stopping recognition in case of initial silence (usually around 5.0)
|
||||
* @param tEnd timeout for stopping recognition in milliseconds after we recognized something (usually around 0.5 - 1.0)
|
||||
* @param tMax timeout for forcing utterance end in milliseconds (usually around 20-30)
|
||||
**/
|
||||
actual fun setEndPointerDelays(
|
||||
tStartMax: Float,
|
||||
tEnd: Float,
|
||||
tMax: Float
|
||||
) {
|
||||
LibVosk.vosk_recognizer_set_endpointer_delays(this, tStartMax, tEnd, tMax)
|
||||
}
|
||||
}
|
||||
@@ -1,84 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import com.sun.jna.PointerType
|
||||
import org.vosk.exception.ModelException
|
||||
import java.io.File
|
||||
import java.nio.file.Path
|
||||
import kotlin.io.path.absolutePathString
|
||||
|
||||
/**
|
||||
* Speaker model is the same as model but contains the data
|
||||
* for speaker identification.
|
||||
*
|
||||
* @since 26 / 12 / 2022
|
||||
* @constructor Loads speaker model data from the file and returns the model object
|
||||
* @throws ModelException if the path provided is invalid
|
||||
*/
|
||||
actual class SpeakerModel : Freeable, PointerType, AutoCloseable {
|
||||
|
||||
/**
|
||||
* Empty constructor for JNA
|
||||
*/
|
||||
constructor()
|
||||
|
||||
/**
|
||||
* Loads speaker model data from the file and returns the model object
|
||||
*
|
||||
* @param path the path of the model on the filesystem
|
||||
*/
|
||||
@Throws(ModelException::class)
|
||||
actual constructor(path: String) : super(
|
||||
LibVosk.vosk_spk_model_new(path) ?: throw ModelException(path)
|
||||
)
|
||||
|
||||
/**
|
||||
* Constructor using a Path, will retrieve absolutePath
|
||||
*
|
||||
* @param path to batch model
|
||||
*/
|
||||
@Throws(ModelException::class)
|
||||
constructor(path: Path) : this(path.absolutePathString())
|
||||
|
||||
/**
|
||||
* Constructor using a File, will retrieve absolutePath
|
||||
*
|
||||
* @param file to batch model
|
||||
*/
|
||||
@Throws(ModelException::class)
|
||||
constructor(file: File) : this(file.absolutePath)
|
||||
|
||||
/**
|
||||
* Releases the model memory
|
||||
*
|
||||
* The model object is reference-counted so if some recognizer
|
||||
* depends on this model, model might still stay alive. When
|
||||
* last recognizer is released, model will be released too.
|
||||
*/
|
||||
actual override fun free() {
|
||||
LibVosk.vosk_spk_model_free(this)
|
||||
}
|
||||
|
||||
/**
|
||||
* @see free
|
||||
*/
|
||||
override fun close() {
|
||||
free()
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,51 +0,0 @@
|
||||
/*
|
||||
* Copyright 2024 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import com.sun.jna.PointerType
|
||||
|
||||
/**
|
||||
* Inverse text normalization
|
||||
*
|
||||
* @since 2024/06/19
|
||||
*/
|
||||
actual class TextProcessor :
|
||||
Freeable, PointerType, AutoCloseable {
|
||||
|
||||
/**
|
||||
* Create text processor
|
||||
*/
|
||||
actual constructor(tagger: Char, verbalizer: Char) :
|
||||
super(LibVosk.vosk_text_processor_new(tagger, verbalizer))
|
||||
|
||||
|
||||
/** Release text processor */
|
||||
actual override fun free() {
|
||||
LibVosk.vosk_text_processor_free(this)
|
||||
}
|
||||
|
||||
/** Convert string */
|
||||
actual fun itn(input: Char): Char =
|
||||
LibVosk.vosk_text_processor_itn(this, input)
|
||||
|
||||
/**
|
||||
* @see free
|
||||
*/
|
||||
override fun close() {
|
||||
free()
|
||||
}
|
||||
}
|
||||
@@ -1,55 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
|
||||
/**
|
||||
* 26 / 12 / 2022
|
||||
*
|
||||
* Control overarching features of libvosk.
|
||||
*/
|
||||
actual object Vosk {
|
||||
/**
|
||||
* Set log level for Kaldi messages
|
||||
*
|
||||
* @param logLevel the level
|
||||
*/
|
||||
@JvmStatic
|
||||
actual fun setLogLevel(logLevel: LogLevel) {
|
||||
LibVosk.vosk_set_log_level(logLevel.value)
|
||||
}
|
||||
|
||||
/**
|
||||
* Init, automatically select a CUDA device and allow multithreading.
|
||||
* Must be called once from the main thread.
|
||||
* Has no effect if HAVE_CUDA flag is not set.
|
||||
*/
|
||||
@JvmStatic
|
||||
actual fun gpuInit() {
|
||||
LibVosk.vosk_gpu_init()
|
||||
}
|
||||
|
||||
/**
|
||||
* Init CUDA device in a multi-threaded environment.
|
||||
* Must be called for each thread.
|
||||
* Has no effect if HAVE_CUDA flag is not set.
|
||||
*/
|
||||
@JvmStatic
|
||||
actual fun gpuThreadInit() {
|
||||
LibVosk.vosk_gpu_thread_init()
|
||||
}
|
||||
}
|
||||
@@ -1,55 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.flow.Flow
|
||||
import kotlinx.coroutines.flow.flowOn
|
||||
import kotlinx.coroutines.flow.map
|
||||
import org.vosk.json.WaveformResult
|
||||
import org.vosk.json.finalResultAsJson
|
||||
import org.vosk.json.partialResultAsJson
|
||||
import org.vosk.json.resultAsJson
|
||||
|
||||
/**
|
||||
* Feed an [Flow] of [ByteArray] into a [Recognizer].
|
||||
*
|
||||
* The returned flow will emit an [WaveformResult] for each result parsed.
|
||||
*
|
||||
* This expects a null terminator to signify the end of a stream.
|
||||
*
|
||||
* This will not close the [recognizer] at the end of reading.
|
||||
*
|
||||
* Any exceptions will be fed into the flow,
|
||||
* and should be collected via [kotlinx.coroutines.flow.catch].
|
||||
*
|
||||
* Flows on [Dispatchers.IO] to prevent blocking the main thread.
|
||||
*
|
||||
* @see [Recognizer.acceptWaveform]
|
||||
*/
|
||||
fun Flow<ByteArray?>.feed(recognizer: Recognizer): Flow<WaveformResult> =
|
||||
map {
|
||||
if (it != null) {
|
||||
if (recognizer.acceptWaveform(it)) {
|
||||
WaveformResult.Result(recognizer.resultAsJson())
|
||||
} else {
|
||||
WaveformResult.PartialResult(recognizer.partialResultAsJson())
|
||||
}
|
||||
} else {
|
||||
WaveformResult.FinalResult(recognizer.finalResultAsJson())
|
||||
}
|
||||
}.flowOn(Dispatchers.IO)
|
||||
@@ -1,22 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk.exception
|
||||
|
||||
/**
|
||||
* Analog of [java.io.IOException]
|
||||
*/
|
||||
actual typealias IOException = java.io.IOException
|
||||
@@ -1,212 +0,0 @@
|
||||
/*
|
||||
* Copyright 2023 Alpha Cephei Inc.
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
import kotlinx.coroutines.*
|
||||
import kotlinx.coroutines.flow.*
|
||||
import kotlinx.coroutines.test.runTest
|
||||
import org.vosk.*
|
||||
import java.io.BufferedInputStream
|
||||
import java.io.FileInputStream
|
||||
import java.io.IOException
|
||||
import java.nio.ByteBuffer
|
||||
import java.nio.ByteOrder
|
||||
import javax.sound.sampled.AudioSystem
|
||||
import javax.sound.sampled.UnsupportedAudioFileException
|
||||
import kotlin.test.Test
|
||||
|
||||
class DecoderTest {
|
||||
val modelPath = System.getenv("MODEL")
|
||||
val testFile = System.getenv("AUDIO")
|
||||
|
||||
init {
|
||||
System.load(System.getenv("LIBRARY"))
|
||||
Vosk.setLogLevel(LogLevel.DEBUG)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun grammarList() {
|
||||
Model(modelPath).use { model ->
|
||||
Recognizer(model, 16000f, listOf("one")).apply {
|
||||
setMaxAlternatives(10)
|
||||
setOutputWordTimes(true)
|
||||
setPartialWords(true)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@Throws(IOException::class, UnsupportedAudioFileException::class)
|
||||
fun decoderTest() {
|
||||
Model(modelPath).use { model ->
|
||||
AudioSystem.getAudioInputStream(BufferedInputStream(FileInputStream(testFile)))
|
||||
.use { ais ->
|
||||
Recognizer(model, 16000f).apply {
|
||||
setMaxAlternatives(10)
|
||||
setOutputWordTimes(true)
|
||||
setPartialWords(true)
|
||||
}.use { recognizer ->
|
||||
val b = ByteArray(4096)
|
||||
while (ais.read(b) >= 0) {
|
||||
if (recognizer.acceptWaveform(b)) {
|
||||
println(recognizer.result)
|
||||
} else {
|
||||
println(recognizer.partialResult)
|
||||
}
|
||||
}
|
||||
println(recognizer.finalResult)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@OptIn(ExperimentalCoroutinesApi::class)
|
||||
@Test
|
||||
@Throws(IOException::class, UnsupportedAudioFileException::class)
|
||||
fun decoderTestFlow() = runTest {
|
||||
Model(modelPath).use { model ->
|
||||
Recognizer(model, 16000f).apply {
|
||||
setMaxAlternatives(10)
|
||||
setOutputWordTimes(true)
|
||||
setPartialWords(true)
|
||||
}.use { recognizer ->
|
||||
AudioSystem.getAudioInputStream(BufferedInputStream(FileInputStream(testFile)))
|
||||
.use { ais ->
|
||||
flow {
|
||||
val b = ByteArray(4096)
|
||||
while (ais.read(b) >= 0) {
|
||||
emit(b)
|
||||
}
|
||||
emit(null)
|
||||
}.flowOn(Dispatchers.IO)
|
||||
.feed(recognizer)
|
||||
.collect {
|
||||
println(it)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* This test aims to simulate the situation for Dicio,
|
||||
* In which we can receive the audio input stream before the recognizer is setup.
|
||||
*
|
||||
* It is recommended to use a large model on desktop to properly see how long it takes to load up.
|
||||
*/
|
||||
@Test
|
||||
@Throws(IOException::class, UnsupportedAudioFileException::class)
|
||||
fun decoderTestFlowBuffered() {
|
||||
// Scope to buffer into
|
||||
val scope = CoroutineScope(Dispatchers.IO)
|
||||
|
||||
AudioSystem.getAudioInputStream(BufferedInputStream(FileInputStream(testFile))).use { ais ->
|
||||
val byteFlow = flow {
|
||||
val b = ByteArray(4096)
|
||||
while (ais.read(b) >= 0) {
|
||||
emit(b)
|
||||
}
|
||||
emit(null)
|
||||
}.flowOn(Dispatchers.IO)
|
||||
.shareIn(scope, SharingStarted.Eagerly, 100)
|
||||
|
||||
// Tell us the current buffer size
|
||||
println("Buffered size:" + byteFlow.replayCache.size)
|
||||
|
||||
var startTime = System.currentTimeMillis()
|
||||
|
||||
Model(modelPath).use { model ->
|
||||
var resultTime = System.currentTimeMillis() - startTime
|
||||
println("Model initialized in: $resultTime")
|
||||
startTime = System.currentTimeMillis()
|
||||
|
||||
Recognizer(model, 16000f).apply {
|
||||
setMaxAlternatives(10)
|
||||
setOutputWordTimes(true)
|
||||
setPartialWords(true)
|
||||
}.use { recognizer ->
|
||||
resultTime = System.currentTimeMillis() - startTime
|
||||
println("Recognizer initialized in: $resultTime")
|
||||
println("Buffered size:" + byteFlow.replayCache.size)
|
||||
|
||||
runBlocking {
|
||||
byteFlow
|
||||
.feed(recognizer)
|
||||
.take(byteFlow.replayCache.size)
|
||||
.collect {
|
||||
println(it)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@Throws(IOException::class, UnsupportedAudioFileException::class)
|
||||
fun decoderTestShort() {
|
||||
Model(modelPath).use { model ->
|
||||
AudioSystem.getAudioInputStream(BufferedInputStream(FileInputStream(testFile)))
|
||||
.use { ais ->
|
||||
Recognizer(model, 16000f).use { recognizer ->
|
||||
val b = ByteArray(4096)
|
||||
val s = ShortArray(2048)
|
||||
while (ais.read(b) >= 0) {
|
||||
ByteBuffer.wrap(b).order(ByteOrder.LITTLE_ENDIAN).asShortBuffer().get(s)
|
||||
if (recognizer.acceptWaveform(s)) {
|
||||
println(recognizer.result)
|
||||
} else {
|
||||
println(recognizer.partialResult)
|
||||
}
|
||||
}
|
||||
println(recognizer.finalResult)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@Throws(IOException::class, UnsupportedAudioFileException::class)
|
||||
fun decoderTestGrammar() {
|
||||
Model(modelPath).use { model ->
|
||||
AudioSystem.getAudioInputStream(BufferedInputStream(FileInputStream(testFile)))
|
||||
.use { ais ->
|
||||
Recognizer(
|
||||
model, 16000f, "[\"one two three four five six seven eight nine zero oh\"]"
|
||||
).use { recognizer ->
|
||||
val b = ByteArray(4096)
|
||||
while (ais.read(b) >= 0) {
|
||||
if (recognizer.acceptWaveform(b)) {
|
||||
println(recognizer.result)
|
||||
} else {
|
||||
println(recognizer.partialResult)
|
||||
}
|
||||
}
|
||||
println(recognizer.finalResult)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun decoderTestException() {
|
||||
try {
|
||||
val model = Model("model_missing")
|
||||
assert(false)
|
||||
} catch (e: IOException) {
|
||||
assert(true)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,25 +0,0 @@
|
||||
headers = libvosk/vosk_api.h
|
||||
package = libvosk
|
||||
|
||||
noStringConversion = \
|
||||
vosk_batch_recognizer_accept_waveform \
|
||||
vosk_recognizer_accept_waveform
|
||||
|
||||
compilerOpts.linux = \
|
||||
-I/usr/include/libvosk/ \
|
||||
-I/usr/local/include/libvosk/ \
|
||||
-I/usr/include/ \
|
||||
-I/usr/local/include/
|
||||
|
||||
compilerOpts.linux_x64 = \
|
||||
-I/usr/lib64/libvosk/ \
|
||||
-I/usr/local/lib64/libvosk/
|
||||
|
||||
linkerOpts.linux = \
|
||||
-L/usr/lib/ \
|
||||
-L/usr/local/lib/ \
|
||||
-llibvosk
|
||||
|
||||
linkerOpts.linux_x64 = \
|
||||
-L/usr/lib64/ \
|
||||
-L/usr/local/lib64/
|
||||
@@ -1,50 +0,0 @@
|
||||
/*
|
||||
* Copyright 2020 Alpha Cephei Inc. & Doomsdayrs
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import cnames.structs.VoskBatchModel
|
||||
import kotlinx.cinterop.CPointer
|
||||
import libvosk.vosk_batch_model_free
|
||||
import libvosk.vosk_batch_model_new
|
||||
import libvosk.vosk_batch_model_wait
|
||||
|
||||
/**
|
||||
* Batch model object
|
||||
*
|
||||
* 26 / 12 / 2022
|
||||
*/
|
||||
actual class BatchModel(val pointer: CPointer<VoskBatchModel>) : Freeable {
|
||||
/**
|
||||
* Creates the batch recognizer object
|
||||
*/
|
||||
@Throws(IOException::class)
|
||||
actual constructor(path: String) : this(vosk_batch_model_new(path) ?: throw ioException(path))
|
||||
|
||||
/**
|
||||
* Releases batch model object
|
||||
*/
|
||||
actual override fun free() {
|
||||
vosk_batch_model_free(pointer)
|
||||
}
|
||||
|
||||
/**
|
||||
* Wait for the processing
|
||||
*/
|
||||
actual fun await() {
|
||||
vosk_batch_model_wait(pointer)
|
||||
}
|
||||
}
|
||||
@@ -1,86 +0,0 @@
|
||||
/*
|
||||
* Copyright 2020 Alpha Cephei Inc. & Doomsdayrs
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import cnames.structs.VoskBatchRecognizer
|
||||
import kotlinx.cinterop.CPointer
|
||||
import kotlinx.cinterop.toCValues
|
||||
import kotlinx.cinterop.toKString
|
||||
import libvosk.*
|
||||
|
||||
/**
|
||||
* Batch recognizer object
|
||||
*
|
||||
* 26 / 12 / 2022
|
||||
*/
|
||||
actual class BatchRecognizer(val pointer: CPointer<VoskBatchRecognizer>) : Freeable {
|
||||
/**
|
||||
* Creates batch recognizer object
|
||||
*/
|
||||
actual constructor(
|
||||
model: BatchModel,
|
||||
sampleRate: Float
|
||||
) : this(vosk_batch_recognizer_new(model.pointer, sampleRate)!!)
|
||||
|
||||
/**
|
||||
* Releases batch recognizer object
|
||||
*/
|
||||
actual override fun free() {
|
||||
vosk_batch_recognizer_free(pointer)
|
||||
}
|
||||
|
||||
/**
|
||||
* Accept batch voice data
|
||||
*/
|
||||
actual fun acceptWaveform(data: ByteArray) {
|
||||
vosk_batch_recognizer_accept_waveform(pointer, data.toCValues(), data.size)
|
||||
}
|
||||
|
||||
/**
|
||||
* Set NLSML output
|
||||
* @param nlsml - boolean value
|
||||
*/
|
||||
actual fun setNLSML(nlsml: Boolean) {
|
||||
vosk_batch_recognizer_set_nlsml(pointer, nlsml.toInt())
|
||||
}
|
||||
|
||||
/**
|
||||
* Closes the stream
|
||||
*/
|
||||
actual fun finishStream() {
|
||||
vosk_batch_recognizer_finish_stream(pointer)
|
||||
}
|
||||
|
||||
/**
|
||||
* Return results
|
||||
*/
|
||||
actual val frontResult: String
|
||||
get() = vosk_batch_recognizer_front_result(pointer)!!.toKString()
|
||||
|
||||
/**
|
||||
* Release and free first retrieved result
|
||||
*/
|
||||
actual fun pop() {
|
||||
vosk_batch_recognizer_pop(pointer)
|
||||
}
|
||||
|
||||
/**
|
||||
* Get amount of pending chunks for more intelligent waiting
|
||||
*/
|
||||
actual val pendingChunks: Int
|
||||
get() = vosk_batch_recognizer_get_pending_chunks(pointer)
|
||||
}
|
||||
@@ -1,23 +0,0 @@
|
||||
/*
|
||||
* Copyright 2020 Alpha Cephei Inc. & Doomsdayrs
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
/**
|
||||
* Use a [Freeable], then immediately free it and return any values.
|
||||
*/
|
||||
fun <T : Freeable, R> T.use(block: (T) -> R): R =
|
||||
block(this).also { free() }
|
||||
@@ -1,3 +0,0 @@
|
||||
package org.vosk
|
||||
|
||||
actual class IOException actual constructor(message: String?) : Exception()
|
||||
@@ -1,63 +0,0 @@
|
||||
/*
|
||||
* Copyright 2020 Alpha Cephei Inc. & Doomsdayrs
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import cnames.structs.VoskModel
|
||||
import kotlinx.cinterop.CPointer
|
||||
import libvosk.vosk_model_find_word
|
||||
import libvosk.vosk_model_free
|
||||
import libvosk.vosk_model_new
|
||||
|
||||
/**
|
||||
* Model stores all the data required for recognition
|
||||
*
|
||||
* It contains static data and can be shared across processing
|
||||
* threads.
|
||||
*
|
||||
* 26 / 12 / 2022
|
||||
*/
|
||||
actual class Model(val pointer: CPointer<VoskModel>) : Freeable {
|
||||
/**
|
||||
* Loads model data from the file and returns the model object
|
||||
*
|
||||
* @param path: the path of the model on the filesystem
|
||||
* @returns model object or NULL if problem occured
|
||||
*/
|
||||
@Throws(IOException::class)
|
||||
actual constructor(path: String) : this(vosk_model_new(path) ?: throw ioException(path))
|
||||
|
||||
/**
|
||||
* Check if a word can be recognized by the model
|
||||
* @param word: the word
|
||||
* @returns the word symbol if @param word exists inside the model
|
||||
* or -1 otherwise.
|
||||
* Reminding that word symbol 0 is for <epsilon>
|
||||
*/
|
||||
actual fun findWord(word: String): Int =
|
||||
vosk_model_find_word(pointer, word)
|
||||
|
||||
/**
|
||||
* Releases the model memory
|
||||
*
|
||||
* The model object is reference-counted so if some recognizer
|
||||
* depends on this model, model might still stay alive. When
|
||||
* last recognizer is released, model will be released too.
|
||||
*/
|
||||
actual override fun free() {
|
||||
vosk_model_free(pointer)
|
||||
}
|
||||
}
|
||||
@@ -1,310 +0,0 @@
|
||||
/*
|
||||
* Copyright 2020 Alpha Cephei Inc. & Doomsdayrs
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import kotlinx.cinterop.CPointer
|
||||
import kotlinx.cinterop.toCValues
|
||||
import kotlinx.cinterop.toKString
|
||||
import libvosk.*
|
||||
|
||||
/**
|
||||
* Recognizer object is the main object which processes data.
|
||||
*
|
||||
* Each recognizer usually runs in own thread and takes audio as input.
|
||||
* Once audio is processed recognizer returns JSON object as a string
|
||||
* which represent decoded information - words, confidences, times, n-best lists,
|
||||
* speaker information and so on
|
||||
*
|
||||
* 26 / 12 / 2022
|
||||
*/
|
||||
actual class Recognizer(val pointer: CPointer<VoskRecognizer>) : Freeable {
|
||||
/**
|
||||
* Creates the recognizer object
|
||||
*
|
||||
* The recognizers process the speech and return text using shared model data
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The sample rate of the audio you going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @returns recognizer object or NULL if problem occured
|
||||
*/
|
||||
actual constructor(
|
||||
model: Model,
|
||||
sampleRate: Float
|
||||
) : this(
|
||||
vosk_recognizer_new(
|
||||
model.pointer,
|
||||
sampleRate
|
||||
)!!
|
||||
)
|
||||
|
||||
/**
|
||||
* Creates the recognizer object with speaker recognition
|
||||
*
|
||||
* With the speaker recognition mode the recognizer not just recognize
|
||||
* text but also return speaker vectors one can use for speaker identification
|
||||
*
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The sample rate of the audio you going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @param speakerModel speaker model for speaker identification
|
||||
* @returns recognizer object or NULL if problem occured
|
||||
*/
|
||||
actual constructor(
|
||||
model: Model,
|
||||
sampleRate: Float,
|
||||
speakerModel: SpeakerModel
|
||||
) : this(
|
||||
vosk_recognizer_new_spk(
|
||||
model.pointer,
|
||||
sampleRate,
|
||||
speakerModel.pointer
|
||||
)!!
|
||||
)
|
||||
|
||||
/**
|
||||
* Creates the recognizer object with the phrase list
|
||||
*
|
||||
* Sometimes when you want to improve recognition accuracy and when you don't need
|
||||
* to recognize large vocabulary you can specify a list of phrases to recognize. This
|
||||
* will improve recognizer speed and accuracy but might return [unk] if user said
|
||||
* something different.
|
||||
*
|
||||
* Only recognizers with lookahead models support this type of quick configuration.
|
||||
* Precompiled HCLG graph models are not supported.
|
||||
*
|
||||
* @param model VoskModel containing static data for recognizer. Model can be
|
||||
* shared across recognizers, even running in different threads.
|
||||
* @param sampleRate The sample rate of the audio you going to feed into the recognizer.
|
||||
* Make sure this rate matches the audio content, it is a common
|
||||
* issue causing accuracy problems.
|
||||
* @param grammar The string with the list of phrases to recognize as JSON array of strings,
|
||||
* for example "["one two three four five", "[unk]"]".
|
||||
*
|
||||
* @returns recognizer object or NULL if problem occured
|
||||
*/
|
||||
actual constructor(
|
||||
model: Model,
|
||||
sampleRate: Float,
|
||||
grammar: String
|
||||
) : this(
|
||||
vosk_recognizer_new_grm(
|
||||
model.pointer,
|
||||
sampleRate,
|
||||
grammar
|
||||
)!!
|
||||
)
|
||||
|
||||
/**
|
||||
* Adds speaker model to already initialized recognizer
|
||||
*
|
||||
* Can add speaker recognition model to already created recognizer. Helps to initialize
|
||||
* speaker recognition for grammar-based recognizer.
|
||||
*
|
||||
* @param speakerModel Speaker recognition model
|
||||
*/
|
||||
actual fun setSpeakerModel(speakerModel: SpeakerModel) {
|
||||
vosk_recognizer_set_spk_model(pointer, speakerModel.pointer)
|
||||
}
|
||||
|
||||
/**
|
||||
* Reconfigures recognizer to use grammar
|
||||
*
|
||||
* @param grammar Set of phrases in JSON array of strings or "[]" to use default model graph.
|
||||
* See also vosk_recognizer_new_grm
|
||||
*/
|
||||
actual fun setGrammar(grammar: String) {
|
||||
vosk_recognizer_set_grm(pointer, grammar)
|
||||
}
|
||||
|
||||
/**
|
||||
* Configures recognizer to output n-best results
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "alternatives": [
|
||||
* { "text": "one two three four five", "confidence": 0.97 },
|
||||
* { "text": "one two three for five", "confidence": 0.03 },
|
||||
* ]
|
||||
* }
|
||||
* </pre>
|
||||
*
|
||||
* @param maxAlternatives - maximum alternatives to return from recognition results
|
||||
*/
|
||||
actual fun setMaxAlternatives(maxAlternatives: Int) {
|
||||
vosk_recognizer_set_max_alternatives(pointer, maxAlternatives)
|
||||
}
|
||||
|
||||
/**
|
||||
* Enables words with times in the output
|
||||
*
|
||||
* <pre>
|
||||
* "result" : [{
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.110000,
|
||||
* "start" : 0.870000,
|
||||
* "word" : "what"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.530000,
|
||||
* "start" : 1.110000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.950000,
|
||||
* "start" : 1.530000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 2.340000,
|
||||
* "start" : 1.950000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 2.610000,
|
||||
* "start" : 2.340000,
|
||||
* "word" : "one"
|
||||
* }],
|
||||
* </pre>
|
||||
*
|
||||
* @param words - boolean value
|
||||
*/
|
||||
actual fun setWords(words: Boolean) {
|
||||
vosk_recognizer_set_words(pointer, words.toInt())
|
||||
}
|
||||
|
||||
/**
|
||||
* Like above return words and confidences in partial results
|
||||
*
|
||||
* @param partialWords - boolean value
|
||||
*/
|
||||
actual fun setPartialWords(partialWords: Boolean) {
|
||||
vosk_recognizer_set_partial_words(pointer, partialWords.toInt())
|
||||
}
|
||||
|
||||
/**
|
||||
* Set NLSML output
|
||||
* @param nlsml - boolean value
|
||||
*/
|
||||
actual fun setNLSML(nlsml: Boolean) {
|
||||
vosk_recognizer_set_nlsml(pointer, nlsml.toInt())
|
||||
}
|
||||
|
||||
/**
|
||||
* Accept voice data
|
||||
*
|
||||
* accept and process new chunk of voice data
|
||||
*
|
||||
* @param data - audio data in PCM 16-bit mono format
|
||||
* @param length - length of the audio data
|
||||
* @returns 1 if silence is occured and you can retrieve a new utterance with result method
|
||||
* 0 if decoding continues
|
||||
* -1 if exception occured
|
||||
*/
|
||||
@Throws(AcceptWaveformException::class)
|
||||
actual fun acceptWaveform(data: ByteArray): Boolean {
|
||||
val result = vosk_recognizer_accept_waveform(pointer, data.toCValues(), data.size)
|
||||
if (result == -1) throw AcceptWaveformException(data)
|
||||
return result == 1
|
||||
}
|
||||
|
||||
/**
|
||||
* Same as above but the version with the short data for language bindings where you have
|
||||
* audio as array of shorts
|
||||
*/
|
||||
@Throws(AcceptWaveformException::class)
|
||||
actual fun acceptWaveform(data: ShortArray): Boolean {
|
||||
val result = vosk_recognizer_accept_waveform_s(pointer, data.toCValues(), data.size)
|
||||
if (result == -1) throw AcceptWaveformException(data)
|
||||
return result == 1
|
||||
}
|
||||
|
||||
/**
|
||||
* Same as above but the version with the float data for language bindings where you have
|
||||
* audio as array of floats
|
||||
*/
|
||||
@Throws(AcceptWaveformException::class)
|
||||
actual fun acceptWaveform(data: FloatArray): Boolean {
|
||||
val result = vosk_recognizer_accept_waveform_f(pointer, data.toCValues(), data.size)
|
||||
if (result == -1) throw AcceptWaveformException(data)
|
||||
return result == 1
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns speech recognition result
|
||||
*
|
||||
* @returns the result in JSON format which contains decoded line, decoded
|
||||
* words, times in seconds and confidences. You can parse this result
|
||||
* with any json parser
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "text" : "what zero zero zero one"
|
||||
* }
|
||||
* </pre>
|
||||
*
|
||||
* If alternatives enabled it returns result with alternatives, see also vosk_recognizer_set_max_alternatives().
|
||||
*
|
||||
* If word times enabled returns word time, see also vosk_recognizer_set_word_times().
|
||||
*/
|
||||
actual val result: String
|
||||
get() = vosk_recognizer_result(pointer)!!.toKString()
|
||||
|
||||
/**
|
||||
* Returns partial speech recognition
|
||||
*
|
||||
* @returns partial speech recognition text which is not yet finalized.
|
||||
* result may change as recognizer process more data.
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "partial" : "cyril one eight zero"
|
||||
* }
|
||||
* </pre>
|
||||
*/
|
||||
actual val finalResult: String
|
||||
get() = vosk_recognizer_result(pointer)!!.toKString()
|
||||
|
||||
/**
|
||||
* Returns speech recognition result. Same as result, but doesn't wait for silence
|
||||
* You usually call it in the end of the stream to get final bits of audio. It
|
||||
* flushes the feature pipeline, so all remaining audio chunks got processed.
|
||||
*
|
||||
* @returns speech result in JSON format.
|
||||
*/
|
||||
actual val partialResult: String
|
||||
get() = vosk_recognizer_partial_result(pointer)!!.toKString()
|
||||
|
||||
/**
|
||||
* Resets the recognizer
|
||||
*
|
||||
* Resets current results so the recognition can continue from scratch */
|
||||
actual fun reset() {
|
||||
vosk_recognizer_reset(pointer)
|
||||
}
|
||||
|
||||
/**
|
||||
* Releases recognizer object
|
||||
*
|
||||
* Underlying model is also unreferenced and if needed released */
|
||||
actual override fun free() {
|
||||
vosk_recognizer_free(pointer)
|
||||
}
|
||||
}
|
||||
@@ -1,50 +0,0 @@
|
||||
/*
|
||||
* Copyright 2020 Alpha Cephei Inc. & Doomsdayrs
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import cnames.structs.VoskSpkModel
|
||||
import kotlinx.cinterop.CPointer
|
||||
import libvosk.vosk_spk_model_free
|
||||
import libvosk.vosk_spk_model_new
|
||||
|
||||
/**
|
||||
* Speaker model is the same as model but contains the data
|
||||
* for speaker identification.
|
||||
*
|
||||
* 26 / 12 / 2022
|
||||
*/
|
||||
actual class SpeakerModel(val pointer: CPointer<VoskSpkModel>) : Freeable {
|
||||
/**
|
||||
* Loads speaker model data from the file and returns the model object
|
||||
*
|
||||
* @param path: the path of the model on the filesystem
|
||||
* @returns model object or NULL if problem occurred
|
||||
*/
|
||||
@Throws(IOException::class)
|
||||
actual constructor(path: String) : this(vosk_spk_model_new(path) ?: throw ioException(path))
|
||||
|
||||
/**
|
||||
* Releases the model memory
|
||||
*
|
||||
* The model object is reference-counted so if some recognizer
|
||||
* depends on this model, model might still stay alive. When
|
||||
* last recognizer is released, model will be released too.
|
||||
*/
|
||||
actual override fun free() {
|
||||
vosk_spk_model_free(pointer)
|
||||
}
|
||||
}
|
||||
@@ -1,26 +0,0 @@
|
||||
/*
|
||||
* Copyright 2020 Alpha Cephei Inc. & Doomsdayrs
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
/*
|
||||
* 26 / 12 / 2022
|
||||
*/
|
||||
|
||||
/**
|
||||
* Converts a boolean to an int
|
||||
*/
|
||||
internal inline fun Boolean.toInt() = if (this) 1 else 0
|
||||
@@ -1,52 +0,0 @@
|
||||
/*
|
||||
* Copyright 2020 Alpha Cephei Inc. & Doomsdayrs
|
||||
*
|
||||
* Licensed under the Apache License, Version 2.0 (the "License");
|
||||
* you may not use this file except in compliance with the License.
|
||||
* You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.vosk
|
||||
|
||||
import libvosk.vosk_set_log_level
|
||||
import libvosk.vosk_gpu_init
|
||||
import libvosk.vosk_gpu_thread_init
|
||||
|
||||
/**
|
||||
* 26 / 12 / 2022
|
||||
*/
|
||||
actual object Vosk {
|
||||
/** Set log level for Kaldi messages
|
||||
*
|
||||
* @param logLevel the level
|
||||
*/
|
||||
actual fun setLogLevel(logLevel: LogLevel) {
|
||||
vosk_set_log_level(logLevel.value)
|
||||
}
|
||||
|
||||
/**
|
||||
* Init, automatically select a CUDA device and allow multithreading.
|
||||
* Must be called once from the main thread.
|
||||
* Has no effect if HAVE_CUDA flag is not set.
|
||||
*/
|
||||
actual fun gpuInit() {
|
||||
vosk_gpu_init()
|
||||
}
|
||||
|
||||
/**
|
||||
* Init CUDA device in a multi-threaded environment.
|
||||
* Must be called for each thread.
|
||||
* Has no effect if HAVE_CUDA flag is not set.
|
||||
*/
|
||||
actual fun gpuThreadInit() {
|
||||
vosk_gpu_thread_init()
|
||||
}
|
||||
}
|
||||
+7
-19
@@ -69,18 +69,16 @@ const vosk_recognizer_ptr = ref.refType(vosk_recognizer);
|
||||
|
||||
let soname;
|
||||
if (os.platform() == 'win32') {
|
||||
// Update path to load dependent dlls
|
||||
let currentPath = process.env.Path;
|
||||
let dllDirectory = path.resolve(path.join(__dirname, 'lib', 'win-x86_64'));
|
||||
process.env.Path = dllDirectory + path.delimiter + currentPath;
|
||||
// Update path to load dependent dlls
|
||||
let currentPath = process.env.Path;
|
||||
let dllDirectory = path.resolve(path.join(__dirname, "lib", "win-x86_64"));
|
||||
process.env.Path = dllDirectory + path.delimiter + currentPath;
|
||||
|
||||
soname = path.join(__dirname, 'lib', 'win-x86_64', 'libvosk.dll');
|
||||
soname = path.join(__dirname, "lib", "win-x86_64", "libvosk.dll")
|
||||
} else if (os.platform() == 'darwin') {
|
||||
soname = path.join(__dirname, 'lib', 'osx-universal', 'libvosk.dylib');
|
||||
} else if (os.platform() == 'linux' && os.arch() == 'arm64') {
|
||||
soname = path.join(__dirname, 'lib', 'linux-arm64', 'libvosk.so');
|
||||
soname = path.join(__dirname, "lib", "osx-universal", "libvosk.dylib")
|
||||
} else {
|
||||
soname = path.join(__dirname, 'lib', 'linux-x86_64', 'libvosk.so');
|
||||
soname = path.join(__dirname, "lib", "linux-x86_64", "libvosk.so")
|
||||
}
|
||||
|
||||
const libvosk = ffi.Library(soname, {
|
||||
@@ -130,9 +128,6 @@ class Model {
|
||||
* @type {unknown}
|
||||
*/
|
||||
this.handle = libvosk.vosk_model_new(modelPath);
|
||||
if (!this.handle) {
|
||||
throw new Error('Failed to create a model.');
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -166,9 +161,6 @@ class SpeakerModel {
|
||||
* @type {unknown}
|
||||
*/
|
||||
this.handle = libvosk.vosk_spk_model_new(modelPath);
|
||||
if (!this.handle) {
|
||||
throw new Error('Failed to create a speaker model.');
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -243,10 +235,6 @@ class Recognizer {
|
||||
: hasOwnProperty(param, 'grammar')
|
||||
? libvosk.vosk_recognizer_new_grm(model.handle, sampleRate, JSON.stringify(param.grammar))
|
||||
: libvosk.vosk_recognizer_new(model.handle, sampleRate);
|
||||
|
||||
if (!this.handle) {
|
||||
throw new Error('Failed to create a recognizer.');
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "vosk",
|
||||
"version": "0.3.75",
|
||||
"version": "0.3.45",
|
||||
"description": "Node binding for continuous offline voice recoginition with Vosk library.",
|
||||
"repository": {
|
||||
"type": "git",
|
||||
|
||||
@@ -1,55 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
import wave
|
||||
import sys
|
||||
|
||||
from vosk import Model, KaldiRecognizer, SetLogLevel, EndpointerMode
|
||||
|
||||
# You can set log level to -1 to disable debug messages
|
||||
SetLogLevel(0)
|
||||
|
||||
wf = wave.open(sys.argv[1], "rb")
|
||||
if wf.getnchannels() != 1 or wf.getsampwidth() != 2 or wf.getcomptype() != "NONE":
|
||||
print("Audio file must be WAV format mono PCM.")
|
||||
sys.exit(1)
|
||||
|
||||
model = Model(lang="en-us")
|
||||
|
||||
# You can also init model by name or with a folder path
|
||||
# model = Model(model_name="vosk-model-en-us-0.21")
|
||||
# model = Model("models/en")
|
||||
|
||||
rec = KaldiRecognizer(model, wf.getframerate())
|
||||
rec.SetWords(True)
|
||||
rec.SetPartialWords(True)
|
||||
rec.SetEndpointerMode(EndpointerMode.VERY_LONG)
|
||||
|
||||
while True:
|
||||
data = wf.readframes(4000)
|
||||
if len(data) == 0:
|
||||
break
|
||||
if rec.AcceptWaveform(data):
|
||||
print(rec.Result())
|
||||
else:
|
||||
print(rec.PartialResult())
|
||||
|
||||
print(rec.FinalResult())
|
||||
|
||||
|
||||
wf = wave.open(sys.argv[1], "rb")
|
||||
if wf.getnchannels() != 1 or wf.getsampwidth() != 2 or wf.getcomptype() != "NONE":
|
||||
print("Audio file must be WAV format mono PCM.")
|
||||
sys.exit(1)
|
||||
|
||||
rec.SetEndpointerDelays(0.5, 0.3, 10.0)
|
||||
|
||||
while True:
|
||||
data = wf.readframes(4000)
|
||||
if len(data) == 0:
|
||||
break
|
||||
if rec.AcceptWaveform(data):
|
||||
print(rec.Result())
|
||||
else:
|
||||
print(rec.PartialResult())
|
||||
|
||||
print(rec.FinalResult())
|
||||
@@ -1,39 +1,40 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
import json
|
||||
import gradio as gr
|
||||
|
||||
from vosk import KaldiRecognizer, Model
|
||||
|
||||
model = Model(lang="en-us")
|
||||
|
||||
def transcribe(stream, new_chunk):
|
||||
|
||||
sample_rate, audio_data = new_chunk
|
||||
audio_data = audio_data.tobytes()
|
||||
|
||||
if stream is None:
|
||||
rec = KaldiRecognizer(model, sample_rate)
|
||||
result = []
|
||||
else:
|
||||
rec, result = stream
|
||||
|
||||
if rec.AcceptWaveform(audio_data):
|
||||
text_result = json.loads(rec.Result())["text"]
|
||||
if text_result != "":
|
||||
result.append(text_result)
|
||||
partial_result = ""
|
||||
else:
|
||||
partial_result = json.loads(rec.PartialResult())["partial"] + " "
|
||||
|
||||
return (rec, result), "\n".join(result) + "\n" + partial_result
|
||||
|
||||
gr.Interface(
|
||||
fn=transcribe,
|
||||
inputs=[
|
||||
"state", gr.Audio(sources=["microphone"], type="numpy", streaming=True),
|
||||
],
|
||||
outputs=[
|
||||
"state", "text",
|
||||
],
|
||||
live=True).launch(share=True)
|
||||
#!/usr/bin/env python3
|
||||
|
||||
import json
|
||||
import gradio as gr
|
||||
|
||||
from vosk import KaldiRecognizer, Model
|
||||
|
||||
model = Model(lang="en-us")
|
||||
|
||||
def transcribe(data, state):
|
||||
sample_rate, audio_data = data
|
||||
audio_data = (audio_data >> 16).astype("int16").tobytes()
|
||||
|
||||
if state is None:
|
||||
rec = KaldiRecognizer(model, sample_rate)
|
||||
result = []
|
||||
else:
|
||||
rec, result = state
|
||||
|
||||
if rec.AcceptWaveform(audio_data):
|
||||
text_result = json.loads(rec.Result())["text"]
|
||||
if text_result != "":
|
||||
result.append(text_result)
|
||||
partial_result = ""
|
||||
else:
|
||||
partial_result = json.loads(rec.PartialResult())["partial"] + " "
|
||||
|
||||
return "\n".join(result) + "\n" + partial_result, (rec, result)
|
||||
|
||||
gr.Interface(
|
||||
fn=transcribe,
|
||||
inputs=[
|
||||
gr.Audio(source="microphone", type="numpy", streaming=True),
|
||||
"state"
|
||||
],
|
||||
outputs=[
|
||||
"textbox",
|
||||
"state"
|
||||
],
|
||||
live=True).launch(share=True)
|
||||
|
||||
@@ -1,11 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
import wave
|
||||
import sys
|
||||
|
||||
from vosk import Processor
|
||||
|
||||
proc = Processor("ru_itn_tagger.fst", "ru_itn_verbalizer.fst")
|
||||
print (proc.process("у нас десять яблок"))
|
||||
print (proc.process("у нас десять яблок и десять миллилитров воды точка"))
|
||||
print (proc.process("мы пришли в восемь часов пять минут"))
|
||||
+1
-1
@@ -45,7 +45,7 @@ with open("README.md", "rb") as fh:
|
||||
|
||||
setuptools.setup(
|
||||
name="vosk",
|
||||
version="0.3.75",
|
||||
version="0.3.45",
|
||||
author="Alpha Cephei Inc",
|
||||
author_email="contact@alphacephei.com",
|
||||
description="Offline open source speech recognition API based on Kaldi and Vosk",
|
||||
|
||||
@@ -28,7 +28,7 @@ def recognize(line):
|
||||
|
||||
def main():
|
||||
p = Pool(8)
|
||||
texts = p.map(recognize, open(sys.argv[1], encoding="utf-8").readlines())
|
||||
texts = p.map(recognize, open(sys.argv[1], encoding="uft-8").readlines())
|
||||
print ("\n".join(texts))
|
||||
|
||||
main()
|
||||
|
||||
+1
-29
@@ -3,7 +3,6 @@ import sys
|
||||
import srt
|
||||
import datetime
|
||||
import json
|
||||
import enum
|
||||
|
||||
import requests
|
||||
from urllib.request import urlretrieve
|
||||
@@ -58,8 +57,7 @@ class Model:
|
||||
raise Exception("Failed to create a model")
|
||||
|
||||
def __del__(self):
|
||||
if _c is not None:
|
||||
_c.vosk_model_free(self._handle)
|
||||
_c.vosk_model_free(self._handle)
|
||||
|
||||
def vosk_model_find_word(self, word):
|
||||
return _c.vosk_model_find_word(self._handle, word.encode("utf-8"))
|
||||
@@ -142,12 +140,6 @@ class SpkModel:
|
||||
def __del__(self):
|
||||
_c.vosk_spk_model_free(self._handle)
|
||||
|
||||
class EndpointerMode(enum.Enum):
|
||||
DEFAULT = 0
|
||||
SHORT = 1
|
||||
LONG = 2
|
||||
VERY_LONG = 3
|
||||
|
||||
class KaldiRecognizer:
|
||||
|
||||
def __init__(self, *args):
|
||||
@@ -180,12 +172,6 @@ class KaldiRecognizer:
|
||||
def SetNLSML(self, enable_nlsml):
|
||||
_c.vosk_recognizer_set_nlsml(self._handle, 1 if enable_nlsml else 0)
|
||||
|
||||
def SetEndpointerMode(self, mode):
|
||||
_c.vosk_recognizer_set_endpointer_mode(self._handle, mode.value)
|
||||
|
||||
def SetEndpointerDelays(self, t_start_max, t_end, t_max):
|
||||
_c.vosk_recognizer_set_endpointer_delays(self._handle, t_start_max, t_end, t_max)
|
||||
|
||||
def SetSpkModel(self, spk_model):
|
||||
_c.vosk_recognizer_set_spk_model(self._handle, spk_model._handle)
|
||||
|
||||
@@ -287,17 +273,3 @@ class BatchRecognizer:
|
||||
|
||||
def GetPendingChunks(self):
|
||||
return _c.vosk_batch_recognizer_get_pending_chunks(self._handle)
|
||||
|
||||
class Processor:
|
||||
|
||||
def __init__(self, *args):
|
||||
self._handle = _c.vosk_text_processor_new(args[0].encode('utf-8'), args[1].encode('utf-8'))
|
||||
|
||||
if self._handle == _ffi.NULL:
|
||||
raise Exception("Failed to create processor")
|
||||
|
||||
def __del__(self):
|
||||
_c.vosk_text_processor_free(self._handle)
|
||||
|
||||
def process(self, text):
|
||||
return _ffi.string(_c.vosk_text_processor_itn(self._handle, text.encode('utf-8'))).decode('utf-8')
|
||||
|
||||
@@ -15,8 +15,8 @@ parser.add_argument(
|
||||
"--model", "-m", type=str,
|
||||
help="model path")
|
||||
parser.add_argument(
|
||||
"--server", "-s", type=str,
|
||||
help="use server for recognition. For example ws://localhost:2700")
|
||||
"--server", "-s", const="ws://localhost:2700", action="store_const",
|
||||
help="use server for recognition")
|
||||
parser.add_argument(
|
||||
"--list-models", default=False, action="store_true",
|
||||
help="list available models")
|
||||
|
||||
@@ -99,7 +99,7 @@ class Transcriber:
|
||||
monologues = {"schemaVersion":"2.0", "monologues":[], "text":[]}
|
||||
for part in result:
|
||||
if part["text"] != "":
|
||||
monologues["text"] += [part["text"]]
|
||||
monologue["text"] += part["text"]
|
||||
for _, res in enumerate(result):
|
||||
if not "result" in res:
|
||||
continue
|
||||
@@ -133,12 +133,6 @@ class Transcriber:
|
||||
start_time = timer()
|
||||
proc = await self.resample_ffmpeg_async(input_file)
|
||||
result, tot_samples = await self.recognize_stream_server(proc)
|
||||
await proc.wait()
|
||||
|
||||
# Bad input, continue
|
||||
if tot_samples == 0:
|
||||
self.queue.task_done()
|
||||
continue
|
||||
|
||||
processed_result = self.format_result(result)
|
||||
if output_file != "":
|
||||
@@ -148,6 +142,8 @@ class Transcriber:
|
||||
else:
|
||||
print(processed_result)
|
||||
|
||||
await proc.wait()
|
||||
|
||||
elapsed = timer() - start_time
|
||||
logging.info("Execution time: {:.3f} sec; "\
|
||||
"xRT {:.3f}".format(elapsed, float(elapsed) * (2 * SAMPLE_RATE) / tot_samples))
|
||||
@@ -169,10 +165,8 @@ class Transcriber:
|
||||
rec = KaldiRecognizer(self.model, SAMPLE_RATE)
|
||||
rec.SetWords(True)
|
||||
result, tot_samples = self.recognize_stream(rec, stream)
|
||||
if tot_samples == 0:
|
||||
return
|
||||
|
||||
processed_result = self.format_result(result)
|
||||
|
||||
if inputdata[1] != "":
|
||||
logging.info("File {} processing complete".format(inputdata[1]))
|
||||
with open(inputdata[1], "w", encoding="utf-8") as fh:
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user