Compare commits
171 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| d917af21ab | |||
| 948c8c7cfd | |||
| dec8e1acf3 | |||
| 3c020526ce | |||
| b0da19b07f | |||
| 70bcd9b018 | |||
| d2604b609d | |||
| e960e0cacf | |||
| 77b114d434 | |||
| 2860092cef | |||
| c6119c4835 | |||
| 415d1927e0 | |||
| 3d1e21d242 | |||
| 7a75e1c3f4 | |||
| c3430e448a | |||
| ae49ea60d2 | |||
| 02dc0ce0c8 | |||
| 15697a18e8 | |||
| 481881e59d | |||
| 11a25b26a7 | |||
| bf87358f6c | |||
| 04a8242230 | |||
| ceb96c301c | |||
| 307df8fdc0 | |||
| c44875a2dd | |||
| 31f11990ca | |||
| b79b85856d | |||
| 6e861c6a19 | |||
| cf62296a51 | |||
| 387b132814 | |||
| de83de8624 | |||
| fe91b5a717 | |||
| a052506a5d | |||
| 6f2d6d0d69 | |||
| a8ae6025bd | |||
| 1b8332a609 | |||
| fe16ec7e57 | |||
| c5ce5e46bd | |||
| 99d66670bc | |||
| 072b42cac6 | |||
| acadb5b4c2 | |||
| 2e1de6c3af | |||
| 67de30908b | |||
| 9d398eff0e | |||
| 02b7312a00 | |||
| 08c35e84f3 | |||
| b92a5c1fc7 | |||
| 8997a587c5 | |||
| dc3d03d742 | |||
| 84df40715c | |||
| b639fb501a | |||
| 746ff47757 | |||
| 2d62db8118 | |||
| 7a2adcd9ba | |||
| 9ccf3ef0e8 | |||
| 0edab6d558 | |||
| 4b344f0cd8 | |||
| 17ddc4d5ba | |||
| 3e33860c47 | |||
| 0269a10833 | |||
| 7af3e9a334 | |||
| d666876be1 | |||
| 564fab7ec1 | |||
| 155c6c2a2a | |||
| d43cbe9344 | |||
| 57cc474c9f | |||
| 586603f8e1 | |||
| d57887d22a | |||
| 6183bcfc5f | |||
| 65f6113b4d | |||
| 8d88b89db1 | |||
| 62885e8963 | |||
| 9fc094a5da | |||
| f97383c17f | |||
| 4b892ec5e7 | |||
| 41035485db | |||
| 9696f4c917 | |||
| 55abf5f5ac | |||
| a1b2e41710 | |||
| dff4ab26e4 | |||
| 83b6e1cdf7 | |||
| 38dbaa15ea | |||
| 0e531b6061 | |||
| de94ef5537 | |||
| 6ef9d13877 | |||
| 1c7b94757d | |||
| 83486e0bef | |||
| 9787e8a53f | |||
| f59d6685ad | |||
| 8a986ef384 | |||
| 78f9f55e14 | |||
| c2e006f664 | |||
| 4b8e9737a5 | |||
| 722b09eaa4 | |||
| f5b4f5a1f2 | |||
| 4407d8da55 | |||
| 73b73527cd | |||
| 4df0e3a741 | |||
| 1a771e0172 | |||
| 4333c3c242 | |||
| b831c9ad57 | |||
| e5c08f7710 | |||
| 51c2968595 | |||
| 5993376322 | |||
| c4281622b9 | |||
| 9d26014de2 | |||
| f6c115d215 | |||
| 6bd102d778 | |||
| d3d6af5712 | |||
| 876093446f | |||
| 336f219f09 | |||
| 584251cbdc | |||
| a34995a788 | |||
| 998e5da227 | |||
| e04c15e367 | |||
| b81f69d407 | |||
| 0ac2064281 | |||
| 1d00bd244e | |||
| 8f5efc58c9 | |||
| a0c5ae1b5e | |||
| 99f48f9de1 | |||
| 1948b23f32 | |||
| c9eb572fc5 | |||
| 7d9895ff81 | |||
| d507210ef8 | |||
| db0a3d23d5 | |||
| f4f920f3cd | |||
| 75993ea276 | |||
| ee9bacb092 | |||
| b1e775c67b | |||
| d631e567aa | |||
| 9edf45be42 | |||
| 8b4b3c646a | |||
| 31bb0557d9 | |||
| 4593183cf9 | |||
| fdc45f1187 | |||
| b3d3c6d12c | |||
| f5f0794def | |||
| 55664fcca0 | |||
| afbf330f16 | |||
| 25aadf61bc | |||
| be47056467 | |||
| 8f623e0aea | |||
| 78025435e4 | |||
| 910455802e | |||
| b12914955c | |||
| dfe11eaf83 | |||
| 37fbe1a52b | |||
| af11bb2361 | |||
| 8da8697c1e | |||
| 944dc87531 | |||
| 5c4dd4644e | |||
| 9bbd172cfd | |||
| dbf9de77c3 | |||
| 8b790cd162 | |||
| 80219066e9 | |||
| 26fa5f098f | |||
| d75bb36131 | |||
| 30c5e8ca79 | |||
| c00e36fab6 | |||
| 80e60b9118 | |||
| e3a95a44bc | |||
| 86caf526f5 | |||
| e09f32b4b4 | |||
| 3803ab345d | |||
| 889b43136f | |||
| b517cf46af | |||
| ffd810fe00 | |||
| a1a0ed70a1 | |||
| 1554d9ede7 | |||
| bca0b86e37 |
+35
-5
@@ -5,9 +5,12 @@
|
||||
# Java class files
|
||||
*.class
|
||||
|
||||
# Object files
|
||||
*.o
|
||||
|
||||
# Gradle files
|
||||
.gradle/
|
||||
build/
|
||||
android/build/
|
||||
gradlew
|
||||
gradlew.bat
|
||||
gradle
|
||||
@@ -24,14 +27,41 @@ wheelhouse
|
||||
__pycache__
|
||||
*.egg-info
|
||||
python/dist
|
||||
python/build
|
||||
python/vosk/*.so
|
||||
python/test/db
|
||||
python/test/hyp
|
||||
python/test/model
|
||||
python/test/ref
|
||||
python/test/result.txt
|
||||
python/test/wav.scp
|
||||
|
||||
# Java
|
||||
*.so
|
||||
java/org
|
||||
java/model-en
|
||||
java/*.cc
|
||||
java/model-spk/
|
||||
java/model/
|
||||
|
||||
# CSharp
|
||||
csharp/gen
|
||||
csharp/*.exe
|
||||
csharp/*.c
|
||||
*.dll
|
||||
*.so
|
||||
*.nupkg
|
||||
csharp/demo/model
|
||||
csharp/demo/test.wav
|
||||
csharp/demo/bin
|
||||
csharp/demo/obj
|
||||
|
||||
# Node
|
||||
nodejs/demo/model
|
||||
nodejs/demo/model-spk
|
||||
nodejs/demo/test.wav
|
||||
nodejs/node_modules
|
||||
nodejs/package-lock.json
|
||||
|
||||
# C
|
||||
c/test_vosk
|
||||
c/test_vosk_speaker
|
||||
c/oprofile_data
|
||||
c/model
|
||||
c/test.wav
|
||||
|
||||
+1
-2
@@ -7,11 +7,10 @@ matrix:
|
||||
services:
|
||||
- docker
|
||||
env: DOCKER_IMAGE=alphacep/kaldi-manylinux:latest
|
||||
PLAT=manylinux2010_x86_64
|
||||
|
||||
install:
|
||||
- docker pull $DOCKER_IMAGE
|
||||
|
||||
script:
|
||||
- docker run --rm -e PLAT=$PLAT -v `pwd`:/io $DOCKER_IMAGE $PRE_CMD /io/travis/build-wheels.sh
|
||||
- docker run --rm -v `pwd`:/io $DOCKER_IMAGE $PRE_CMD /io/travis/build-wheels.sh
|
||||
- ls wheelhouse/
|
||||
|
||||
@@ -1,134 +1,25 @@
|
||||
[](https://travis-ci.com/alphacep/vosk-api)
|
||||
# About
|
||||
|
||||
Language bindings for Vosk and Kaldi to access speech recognition from various languages and on various platforms
|
||||
Vosk is an offline open source speech recognition toolkit. It enables
|
||||
speech recognition models for 17 languages and dialects - English, Indian
|
||||
English, German, French, Spanish, Portuguese, Chinese, Russian, Turkish,
|
||||
Vietnamese, Italian, Dutch, Catalan, Arabic, Greek, Farsi, Filipino.
|
||||
|
||||
* Python on Linux, Windows and RPi
|
||||
* Node
|
||||
* Android
|
||||
* iOS
|
||||
Vosk models are small (50 Mb) but provide continuous large vocabulary
|
||||
transcription, zero-latency response with streaming API, reconfigurable
|
||||
vocabulary and speaker identification.
|
||||
|
||||
## Android build
|
||||
Speech recognition bindings implemented for various programming languages
|
||||
like Python, Java, Node.JS, C#, C++ and others.
|
||||
|
||||
```
|
||||
cd android
|
||||
gradle build
|
||||
```
|
||||
Vosk supplies speech recognition for chatbots, smart home appliances,
|
||||
virtual assistants. It can also create subtitles for movies,
|
||||
transcription for lectures and interviews.
|
||||
|
||||
Please note that medium blog post about 64-bit is not relevant anymore, the script builds x86, arm64 and armv7 libraries automatically without any modifications.
|
||||
Vosk scales from small devices like Raspberry Pi or Android smartphone to
|
||||
big clusters.
|
||||
|
||||
For example of Android application using Vosk-API check https://github.com/alphacep/kaldi-android-demo project
|
||||
# Documentation
|
||||
|
||||
## iOS build
|
||||
|
||||
Available on request. Drop as a mail at [contact@alphacephei.com](mailto:contact@alphacephei.com).
|
||||
|
||||
## Python installation from Pypi
|
||||
|
||||
The easiest way to install vosk api is with pip. You do not have to compile anything. We currently support only Linux on x86_64 and Raspberry Pi. Other systems (windows, mac) will come soon.
|
||||
|
||||
Make sure you have newer pip and python:
|
||||
|
||||
* Python version >= 3.4
|
||||
* pip version >= 19.0
|
||||
|
||||
Uprade python and pip if needed. Then install vosk on Linux with a simple command
|
||||
|
||||
```
|
||||
pip3 install vosk
|
||||
```
|
||||
|
||||
## Websocket Server and GRPC server
|
||||
|
||||
We also provide a websocket server and grpc server which can be used in telephony and other applications. With bigger models adapted for 8khz audio it provides more accuracy.
|
||||
|
||||
The server is installed with docker and can run with a single command:
|
||||
|
||||
```
|
||||
docker run -d -p 2700:2700 alphacep/kaldi-en:latest
|
||||
```
|
||||
|
||||
For details see https://github.com/alphacep/vosk-server
|
||||
|
||||
|
||||
## Compilation from source
|
||||
|
||||
If you still want to build from scratch, you can compile Kaldi and Vosk yourself. The compilation is straightforward but might be a little confusing for newbie. In case you want to follow this, please watch the errors.
|
||||
|
||||
#### Kaldi compilation for local python, node and java modules
|
||||
|
||||
```
|
||||
git clone -b lookahead --single-branch https://github.com/alphacep/kaldi
|
||||
cd kaldi/tools
|
||||
make
|
||||
```
|
||||
|
||||
install all dependencies and repeat `make` if needed
|
||||
|
||||
```
|
||||
extras/install_openblas.sh
|
||||
cd ../src
|
||||
./configure --mathlib=OPENBLAS --shared --use-cuda=no
|
||||
make -j 10
|
||||
```
|
||||
|
||||
#### Python module build
|
||||
|
||||
Then build the python module
|
||||
|
||||
```
|
||||
export KALDI_ROOT=<KALDI_ROOT>
|
||||
cd python
|
||||
python3 setup.py install
|
||||
```
|
||||
|
||||
#### Java example API build
|
||||
|
||||
Or Java
|
||||
|
||||
```
|
||||
cd java && KALDI_ROOT=<KALDI_ROOT> make
|
||||
wget https://github.com/alphacep/kaldi-android-demo/releases/download/2020-01/alphacep-model-android-en-us-0.3.tar.gz
|
||||
tar xf alphacep-model-android-en-us-0.3.tar.gz
|
||||
mv alphacep-model-android-en-us-0.3 model
|
||||
make run
|
||||
```
|
||||
|
||||
#### C# build
|
||||
|
||||
Or C#
|
||||
|
||||
```
|
||||
cd csharp && KALDI_ROOT=<KALDI_ROOT> make
|
||||
wget https://github.com/alphacep/kaldi-android-demo/releases/download/2020-01/alphacep-model-android-en-us-0.3.tar.gz
|
||||
tar xf alphacep-model-android-en-us-0.3.tar.gz
|
||||
mv alphacep-model-android-en-us-0.3 model
|
||||
mono test.exe
|
||||
```
|
||||
|
||||
## Running the example code with python
|
||||
|
||||
Run like this:
|
||||
|
||||
```
|
||||
cd vosk-api/python/example
|
||||
wget https://github.com/alphacep/kaldi-android-demo/releases/download/2020-01/alphacep-model-android-en-us-0.3.tar.gz
|
||||
tar xf alphacep-model-android-en-us-0.3.tar.gz
|
||||
mv alphacep-model-android-en-us-0.3 model-en
|
||||
python3 ./test_simple.py test.wav
|
||||
```
|
||||
|
||||
To run with your audio file make sure it has proper format - PCM 16khz 16bit mono, otherwise decoding will not work.
|
||||
|
||||
You can find other examples of using a microphone, decoding with a fixed small vocabulary or speaker identification setup in [python/example subfolder](https://github.com/alphacep/vosk-api/tree/master/python/example)
|
||||
|
||||
## Models for different languages
|
||||
|
||||
For information about models see [the documentation on available models](https://github.com/alphacep/vosk-api/blob/master/doc/models.md).
|
||||
|
||||
## Contact Us
|
||||
|
||||
If you have any questions, feel free to
|
||||
|
||||
* Post an issue here on github
|
||||
* Send us an e-mail at [contact@alphacephei.com](mailto:contact@alphacephei.com)
|
||||
* Join our group dedicated to speech recognition on Telegram [@speech_recognition](https://t.me/speech_recognition)
|
||||
For installation instructions, examples and documentation visit [Vosk
|
||||
Website](https://alphacephei.com/vosk).
|
||||
|
||||
+10
-4
@@ -8,6 +8,9 @@ set(KALDI_SUFFIX "arm_32")
|
||||
elseif ("x${ANDROID_ABI}" STREQUAL "xarm64-v8a")
|
||||
set(OPENBLAS_ARCH "armv8")
|
||||
set(KALDI_SUFFIX "arm_64")
|
||||
elseif ("x${ANDROID_ABI}" STREQUAL "xx86")
|
||||
set(OPENBLAS_ARCH "atom")
|
||||
set(KALDI_SUFFIX "x86")
|
||||
else ("x${ANDROID_ABI}" STREQUAL "xarmeabi-v7a")
|
||||
set(OPENBLAS_ARCH "atom")
|
||||
set(KALDI_SUFFIX "x86_64")
|
||||
@@ -19,6 +22,8 @@ set(LIB_ROOT "${PROJECT_SOURCE_DIR}/build/kaldi_${KALDI_SUFFIX}/local")
|
||||
set(API_SOURCES
|
||||
"${PROJECT_SOURCE_DIR}/../src/kaldi_recognizer.cc"
|
||||
"${PROJECT_SOURCE_DIR}/../src/kaldi_recognizer.h"
|
||||
"${PROJECT_SOURCE_DIR}/../src/language_model.cc"
|
||||
"${PROJECT_SOURCE_DIR}/../src/language_model.h"
|
||||
"${PROJECT_SOURCE_DIR}/../src/model.cc"
|
||||
"${PROJECT_SOURCE_DIR}/../src/model.h"
|
||||
"${PROJECT_SOURCE_DIR}/../src/spk_model.cc"
|
||||
@@ -27,16 +32,16 @@ set(API_SOURCES
|
||||
"${PROJECT_SOURCE_DIR}/../src/vosk_api.h"
|
||||
)
|
||||
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -O3 -DFST_NO_DYNAMIC_LINKING")
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++17 -O3 -DFST_NO_DYNAMIC_LINKING")
|
||||
|
||||
add_library( kaldi_jni SHARED
|
||||
add_library( vosk_jni SHARED
|
||||
build/generated-src/cpp/vosk_wrap.cc
|
||||
${API_SOURCES}
|
||||
)
|
||||
|
||||
include_directories("${PROJECT_SOURCE_DIR}/../src" "build/kaldi_${KALDI_SUFFIX}/kaldi/src" "build/kaldi_${KALDI_SUFFIX}/local/include")
|
||||
|
||||
target_link_libraries( kaldi_jni
|
||||
target_link_libraries( vosk_jni
|
||||
${KALDI_ROOT}/src/online2/kaldi-online2.a
|
||||
${KALDI_ROOT}/src/decoder/kaldi-decoder.a
|
||||
${KALDI_ROOT}/src/ivector/kaldi-ivector.a
|
||||
@@ -45,6 +50,7 @@ target_link_libraries( kaldi_jni
|
||||
${KALDI_ROOT}/src/tree/kaldi-tree.a
|
||||
${KALDI_ROOT}/src/feat/kaldi-feat.a
|
||||
${KALDI_ROOT}/src/lat/kaldi-lat.a
|
||||
${KALDI_ROOT}/src/lm/kaldi-lm.a
|
||||
${KALDI_ROOT}/src/hmm/kaldi-hmm.a
|
||||
${KALDI_ROOT}/src/transform/kaldi-transform.a
|
||||
${KALDI_ROOT}/src/cudamatrix/kaldi-cudamatrix.a
|
||||
@@ -54,7 +60,7 @@ target_link_libraries( kaldi_jni
|
||||
${KALDI_ROOT}/src/base/kaldi-base.a
|
||||
${LIB_ROOT}/lib/libfst.a
|
||||
${LIB_ROOT}/lib/libfstngram.a
|
||||
${LIB_ROOT}/lib/libopenblas_${OPENBLAS_ARCH}-r0.3.7.a
|
||||
${LIB_ROOT}/lib/libopenblas.a
|
||||
${LIB_ROOT}/lib/libclapack.a
|
||||
${LIB_ROOT}/lib/liblapack.a
|
||||
${LIB_ROOT}/lib/libblas.a
|
||||
|
||||
+2
-21
@@ -1,22 +1,3 @@
|
||||
This is still work in progress, more to come
|
||||
Vosk library for Android
|
||||
|
||||
## TODO
|
||||
|
||||
* Optimize graph construction, current one is below accuracy
|
||||
|
||||
* Load model from the AAR (mmap them in tflite style)
|
||||
|
||||
* Add decoding speed measurement
|
||||
|
||||
* Add wakeup word
|
||||
|
||||
* Add speakerid
|
||||
|
||||
* Integrate proper hardware optimized neural network library. Candidates are:
|
||||
|
||||
* https://github.com/XiaoMi/mace
|
||||
* https://github.com/Tencent/ncnn
|
||||
* https://developer.android.com/ndk/guides/neuralnetworks/ (since API level 27)
|
||||
* https://github.com/google/XNNPACK
|
||||
|
||||
* Quantization for the models
|
||||
See for details https://alphacephei.com/vosk/android
|
||||
|
||||
+32
-21
@@ -31,36 +31,38 @@ fi
|
||||
|
||||
set -x
|
||||
|
||||
OS_NAME=`echo $(uname -s) | tr '[:upper:]' '[:lower:]'`
|
||||
ANDROID_NDK_HOME=$ANDROID_SDK_HOME/ndk-bundle
|
||||
ANDROID_TOOLCHAIN_PATH=$ANDROID_NDK_HOME/toolchains/llvm/prebuilt/linux-x86_64
|
||||
ANDROID_TOOLCHAIN_PATH=$ANDROID_NDK_HOME/toolchains/llvm/prebuilt/${OS_NAME}-x86_64
|
||||
WORKDIR_X86=`pwd`/build/kaldi_x86
|
||||
WORKDIR_X86_64=`pwd`/build/kaldi_x86_64
|
||||
WORKDIR_ARM32=`pwd`/build/kaldi_arm_32
|
||||
WORKDIR_ARM64=`pwd`/build/kaldi_arm_64
|
||||
PATH=$PATH:$ANDROID_NDK_HOME/toolchains/llvm/prebuilt/linux-x86_64/bin
|
||||
OPENFST_VERSION=1.6.7
|
||||
PATH=$PATH:$ANDROID_NDK_HOME/toolchains/llvm/prebuilt/${OS_NAME}-x86_64/bin
|
||||
OPENFST_VERSION=1.8.0
|
||||
|
||||
mkdir -p $WORKDIR_ARM64/local/lib $WORKDIR_ARM32/local/lib $WORKDIR_X86_64/local/lib
|
||||
mkdir -p $WORKDIR_ARM64/local/lib $WORKDIR_ARM32/local/lib $WORKDIR_X86_64/local/lib $WORKDIR_X86/local/lib
|
||||
|
||||
# Build standalone CLAPACK since gfortran is missing
|
||||
cd build
|
||||
git clone https://github.com/simonlynen/android_libs
|
||||
cd android_libs/lapack
|
||||
sed -i 's/APP_STL := gnustl_static/APP_STL := c++_static/g' jni/Application.mk && \
|
||||
sed -i 's/android-10/android-21/g' project.properties && \
|
||||
sed -i 's/APP_ABI := armeabi armeabi-v7a/APP_ABI := armeabi-v7a arm64-v8a x86_64/g' jni/Application.mk && \
|
||||
sed -i 's/LOCAL_MODULE:= testlapack/#LOCAL_MODULE:= testlapack/g' jni/Android.mk && \
|
||||
sed -i 's/LOCAL_SRC_FILES:= testclapack.cpp/#LOCAL_SRC_FILES:= testclapack.cpp/g' jni/Android.mk && \
|
||||
sed -i 's/LOCAL_STATIC_LIBRARIES := lapack/#LOCAL_STATIC_LIBRARIES := lapack/g' jni/Android.mk && \
|
||||
sed -i 's/include $(BUILD_SHARED_LIBRARY)/#include $(BUILD_SHARED_LIBRARY)/g' jni/Android.mk && \
|
||||
sed -i.bak -e 's/APP_STL := gnustl_static/APP_STL := c++_static/g' jni/Application.mk && \
|
||||
sed -i.bak -e 's/android-10/android-21/g' project.properties && \
|
||||
sed -i.bak -e 's/APP_ABI := armeabi armeabi-v7a/APP_ABI := armeabi-v7a arm64-v8a x86_64 x86/g' jni/Application.mk && \
|
||||
sed -i.bak -e 's/LOCAL_MODULE:= testlapack/#LOCAL_MODULE:= testlapack/g' jni/Android.mk && \
|
||||
sed -i.bak -e 's/LOCAL_SRC_FILES:= testclapack.cpp/#LOCAL_SRC_FILES:= testclapack.cpp/g' jni/Android.mk && \
|
||||
sed -i.bak -e 's/LOCAL_STATIC_LIBRARIES := lapack/#LOCAL_STATIC_LIBRARIES := lapack/g' jni/Android.mk && \
|
||||
sed -i.bak -e 's/include $(BUILD_SHARED_LIBRARY)/#include $(BUILD_SHARED_LIBRARY)/g' jni/Android.mk && \
|
||||
${ANDROID_NDK_HOME}/ndk-build && \
|
||||
cp obj/local/armeabi-v7a/*.a ${WORKDIR_ARM32}/local/lib && \
|
||||
cp obj/local/arm64-v8a/*.a ${WORKDIR_ARM64}/local/lib
|
||||
cp obj/local/x86_64/*.a ${WORKDIR_X86_64}/local/lib
|
||||
cp obj/local/x86/*.a ${WORKDIR_X86}/local/lib
|
||||
|
||||
# Architecture-specific part
|
||||
|
||||
|
||||
for arch in arm32 arm64 x86_64; do
|
||||
for arch in arm32 arm64 x86_64 x86; do
|
||||
#for arch in x86_64; do
|
||||
|
||||
case $arch in
|
||||
@@ -91,22 +93,28 @@ case $arch in
|
||||
CXX=x86_64-linux-android21-clang++
|
||||
ARCHFLAGS=""
|
||||
;;
|
||||
x86)
|
||||
BLAS_ARCH=ATOM
|
||||
WORKDIR=$WORKDIR_X86
|
||||
HOST=i686-linux-android
|
||||
AR=i686-linux-android-ar
|
||||
CC=i686-linux-android21-clang
|
||||
CXX=i686-linux-android21-clang++
|
||||
ARCHFLAGS=""
|
||||
;;
|
||||
esac
|
||||
|
||||
# openblas first
|
||||
cd $WORKDIR
|
||||
git clone -b v0.3.7 --single-branch https://github.com/xianyi/OpenBLAS
|
||||
git clone -b v0.3.13 --single-branch https://github.com/xianyi/OpenBLAS
|
||||
make -C OpenBLAS TARGET=$BLAS_ARCH ONLY_CBLAS=1 AR=$AR CC=$CC HOSTCC=gcc ARM_SOFTFP_ABI=1 USE_THREAD=0 NUM_THREADS=1 -j4
|
||||
make -C OpenBLAS install PREFIX=$WORKDIR/local
|
||||
|
||||
# tools directory --> we'll only compile OpenFST
|
||||
cd $WORKDIR
|
||||
wget -c -T 10 -t 1 http://www.openfst.org/twiki/pub/FST/FstDownload/openfst-${OPENFST_VERSION}.tar.gz || \
|
||||
wget -c -T 10 -t 3 http://www.openslr.org/resources/2/openfst-${OPENFST_VERSION}.tar.gz
|
||||
|
||||
tar -zxvf openfst-${OPENFST_VERSION}.tar.gz
|
||||
cd openfst-${OPENFST_VERSION}
|
||||
|
||||
git clone https://github.com/alphacep/openfst
|
||||
cd openfst
|
||||
autoreconf -i
|
||||
CXX=$CXX CXXFLAGS="$ARCHFLAGS -O3 -DFST_NO_DYNAMIC_LINKING" ./configure --prefix=${WORKDIR}/local \
|
||||
--enable-shared --enable-static --with-pic --disable-bin \
|
||||
--enable-lookahead-fsts --enable-ngram-fsts --host=$HOST --build=x86-linux-gnu
|
||||
@@ -117,9 +125,12 @@ make install
|
||||
cd $WORKDIR
|
||||
git clone -b android-mix --single-branch https://github.com/alphacep/kaldi
|
||||
cd $WORKDIR/kaldi/src
|
||||
if [ "`uname`" == "Darwin" ]; then
|
||||
sed -i.bak -e 's/libfst.dylib/libfst.a/' configure
|
||||
fi
|
||||
|
||||
CXX=$CXX CXXFLAGS="$ARCHFLAGS -O3 -DFST_NO_DYNAMIC_LINKING" ./configure --use-cuda=no \
|
||||
--mathlib=OPENBLAS --shared \
|
||||
--mathlib=OPENBLAS_CLAPACK --shared \
|
||||
--android-incdir=${ANDROID_TOOLCHAIN_PATH}/sysroot/usr/include \
|
||||
--host=$HOST --openblas-root=${WORKDIR}/local \
|
||||
--fst-root=${WORKDIR}/local --fst-version=${OPENFST_VERSION}
|
||||
|
||||
+46
-4
@@ -5,9 +5,17 @@ buildscript {
|
||||
}
|
||||
dependencies {
|
||||
classpath 'com.android.tools.build:gradle:3.5.3'
|
||||
classpath 'com.jfrog.bintray.gradle:gradle-bintray-plugin:1.8.5'
|
||||
}
|
||||
}
|
||||
|
||||
plugins {
|
||||
id "com.jfrog.bintray" version "1.8.5"
|
||||
}
|
||||
|
||||
def archiveName = "vosk-android"
|
||||
def libVersion = "0.3.17"
|
||||
|
||||
allprojects {
|
||||
repositories {
|
||||
google()
|
||||
@@ -16,22 +24,23 @@ allprojects {
|
||||
}
|
||||
|
||||
apply plugin: 'com.android.library'
|
||||
apply plugin: 'maven-publish'
|
||||
|
||||
android {
|
||||
compileSdkVersion 29
|
||||
defaultConfig {
|
||||
minSdkVersion 21
|
||||
targetSdkVersion 29
|
||||
versionCode 5
|
||||
versionName "5.2"
|
||||
setProperty("archivesBaseName", "kaldi-android-$versionName")
|
||||
versionCode 6
|
||||
versionName = libVersion
|
||||
archivesBaseName = archiveName
|
||||
externalNativeBuild {
|
||||
cmake {
|
||||
arguments "-DCMAKE_VERBOSE_MAKEFILE=ON", "-DANDROID_ARM_NEON=TRUE", "-DCMAKE_CXX_FLAGS_RELEASE=-O3"
|
||||
}
|
||||
}
|
||||
ndk {
|
||||
abiFilters 'armeabi-v7a', 'arm64-v8a', 'x86_64'
|
||||
abiFilters 'armeabi-v7a', 'arm64-v8a', 'x86_64', 'x86'
|
||||
}
|
||||
}
|
||||
sourceSets {
|
||||
@@ -46,6 +55,39 @@ android {
|
||||
}
|
||||
}
|
||||
|
||||
Properties properties = new Properties()
|
||||
properties.load(project.rootProject.file('local.properties').newDataInputStream())
|
||||
|
||||
bintray {
|
||||
user = properties.getProperty("bintray.user")
|
||||
key = properties.getProperty("bintray.apikey")
|
||||
pkg {
|
||||
repo = 'vosk'
|
||||
name = 'vosk-android'
|
||||
userOrg = "alphacep"
|
||||
licenses = ['Apache2.0']
|
||||
websiteUrl = 'https://github.com/alphacep/vosk-api'
|
||||
issueTrackerUrl = 'https://github.com/alphacep/vosk-api/issues'
|
||||
vcsUrl = 'https://github.com/alphacep/vosk-api'
|
||||
version {
|
||||
name = libVersion
|
||||
vcsTag = libVersion
|
||||
}
|
||||
}
|
||||
publications = ['aar']
|
||||
}
|
||||
|
||||
publishing {
|
||||
publications {
|
||||
aar(MavenPublication) {
|
||||
groupId 'com.alphacep'
|
||||
artifactId archiveName
|
||||
version libVersion
|
||||
artifact("$buildDir/outputs/aar/$archiveName-release.aar")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
task swig {
|
||||
doLast {
|
||||
mkdir 'build/generated-src/java'
|
||||
|
||||
@@ -0,0 +1,71 @@
|
||||
buildscript {
|
||||
repositories {
|
||||
google()
|
||||
jcenter()
|
||||
}
|
||||
dependencies {
|
||||
classpath 'com.android.tools.build:gradle:3.5.3'
|
||||
classpath 'com.jfrog.bintray.gradle:gradle-bintray-plugin:1.8.5'
|
||||
}
|
||||
}
|
||||
|
||||
plugins {
|
||||
id "com.jfrog.bintray"
|
||||
}
|
||||
|
||||
def archiveName = "vosk-model-en"
|
||||
def libVersion = "0.3.17"
|
||||
|
||||
allprojects {
|
||||
repositories {
|
||||
google()
|
||||
jcenter()
|
||||
}
|
||||
}
|
||||
|
||||
apply plugin: 'com.android.library'
|
||||
apply plugin: 'maven-publish'
|
||||
|
||||
android {
|
||||
compileSdkVersion 29
|
||||
defaultConfig {
|
||||
minSdkVersion 21
|
||||
targetSdkVersion 29
|
||||
versionCode 6
|
||||
versionName = libVersion
|
||||
archivesBaseName = archiveName
|
||||
}
|
||||
}
|
||||
|
||||
Properties properties = new Properties()
|
||||
properties.load(project.rootProject.file('local.properties').newDataInputStream())
|
||||
|
||||
bintray {
|
||||
user = properties.getProperty("bintray.user")
|
||||
key = properties.getProperty("bintray.apikey")
|
||||
pkg {
|
||||
repo = 'vosk'
|
||||
name = 'vosk-model-en'
|
||||
userOrg = "alphacep"
|
||||
licenses = ['Apache2.0']
|
||||
websiteUrl = 'https://github.com/alphacep/vosk-api'
|
||||
issueTrackerUrl = 'https://github.com/alphacep/vosk-api/issues'
|
||||
vcsUrl = 'https://github.com/alphacep/vosk-api'
|
||||
version {
|
||||
name = libVersion
|
||||
vcsTag = libVersion
|
||||
}
|
||||
}
|
||||
publications = ['aar']
|
||||
}
|
||||
|
||||
publishing {
|
||||
publications {
|
||||
aar(MavenPublication) {
|
||||
groupId 'com.alphacep'
|
||||
artifactId archiveName
|
||||
version libVersion
|
||||
artifact("$buildDir/outputs/aar/$archiveName-release.aar")
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,3 @@
|
||||
<manifest xmlns:android="http://schemas.android.com/apk/res/android"
|
||||
package="org.kaldi.model.en">
|
||||
</manifest>
|
||||
@@ -0,0 +1,7 @@
|
||||
US English model for mobile Vosk applications
|
||||
|
||||
Copyright 2020 Alpha Cephei Inc
|
||||
|
||||
Accuracy: 10.38 (tedlium test) 9.85 (librispeech test-clean)
|
||||
Speed: 0.11xRT (desktop)
|
||||
Latency: 0.15s (right context)
|
||||
@@ -0,0 +1 @@
|
||||
0.3.16
|
||||
@@ -0,0 +1 @@
|
||||
include 'model-en'
|
||||
@@ -241,10 +241,6 @@ public class Assets {
|
||||
if (!items.get(path).equals(externalItems.get(path))
|
||||
|| !(new File(externalDir, path).exists()))
|
||||
newItems.add(path);
|
||||
else
|
||||
Log.i(TAG,
|
||||
String.format("Skipping asset %s: checksums are equal", path));
|
||||
|
||||
}
|
||||
|
||||
unusedItems.addAll(externalItems.keySet());
|
||||
@@ -252,13 +248,11 @@ public class Assets {
|
||||
|
||||
for (String path : newItems) {
|
||||
File file = copy(path);
|
||||
Log.i(TAG, String.format("Copying asset %s to %s", path, file));
|
||||
}
|
||||
|
||||
for (String path : unusedItems) {
|
||||
File file = new File(externalDir, path);
|
||||
file.delete();
|
||||
Log.i(TAG, String.format("Removing asset %s", file));
|
||||
}
|
||||
|
||||
updateItemList(items);
|
||||
|
||||
+17
-23
@@ -29,19 +29,18 @@ import android.os.Looper;
|
||||
import android.util.Log;
|
||||
|
||||
/**
|
||||
* Main class to access recognizer functions. After configuration this class
|
||||
* starts a listener thread which records the data and recognizes it using
|
||||
* VOSK engine. Recognition events are passed to a client using
|
||||
* Service that records audio in a thread, passes it to a recognizer and emits
|
||||
* recognition results. Recognition events are passed to a client using
|
||||
* {@link RecognitionListener}
|
||||
*
|
||||
*
|
||||
*/
|
||||
public class SpeechRecognizer {
|
||||
public class SpeechService {
|
||||
|
||||
protected static final String TAG = SpeechRecognizer.class.getSimpleName();
|
||||
protected static final String TAG = SpeechService.class.getSimpleName();
|
||||
|
||||
private final KaldiRecognizer recognizer;
|
||||
|
||||
private final int sampleRate;
|
||||
private final int sampleRate;
|
||||
private final static float BUFFER_SIZE_SECONDS = 0.4f;
|
||||
private int bufferSize;
|
||||
private final AudioRecord recorder;
|
||||
@@ -53,17 +52,18 @@ public class SpeechRecognizer {
|
||||
private final Collection<RecognitionListener> listeners = new HashSet<RecognitionListener>();
|
||||
|
||||
/**
|
||||
* Creates speech recognizer. Recognizer holds the AudioRecord object, so you
|
||||
* Creates speech service. Service holds the AudioRecord object, so you
|
||||
* need to call {@link release} in order to properly finalize it.
|
||||
*
|
||||
* @throws IOException thrown if audio recorder can not be created for some reason.
|
||||
*/
|
||||
public SpeechRecognizer(Model model) throws IOException {
|
||||
recognizer = new KaldiRecognizer(model, 16000.0f);
|
||||
sampleRate = 16000;
|
||||
bufferSize = Math.round(sampleRate * BUFFER_SIZE_SECONDS);
|
||||
public SpeechService(KaldiRecognizer recognizer, float sampleRate) throws IOException {
|
||||
this.recognizer = recognizer;
|
||||
this.sampleRate = (int)sampleRate;
|
||||
|
||||
bufferSize = Math.round(this.sampleRate * BUFFER_SIZE_SECONDS);
|
||||
recorder = new AudioRecord(
|
||||
AudioSource.VOICE_RECOGNITION, sampleRate,
|
||||
AudioSource.VOICE_RECOGNITION, this.sampleRate,
|
||||
AudioFormat.CHANNEL_IN_MONO,
|
||||
AudioFormat.ENCODING_PCM_16BIT, bufferSize * 2);
|
||||
|
||||
@@ -148,8 +148,7 @@ public class SpeechRecognizer {
|
||||
public boolean stop() {
|
||||
boolean result = stopRecognizerThread();
|
||||
if (result) {
|
||||
Log.i(TAG, "Stop recognition");
|
||||
mainHandler.post(new ResultEvent(recognizer.FinalResult(), true));
|
||||
mainHandler.post(new ResultEvent(recognizer.Result(), true));
|
||||
}
|
||||
return result;
|
||||
}
|
||||
@@ -162,10 +161,7 @@ public class SpeechRecognizer {
|
||||
*/
|
||||
public boolean cancel() {
|
||||
boolean result = stopRecognizerThread();
|
||||
if (result) {
|
||||
Log.i(TAG, "Cancel recognition");
|
||||
}
|
||||
|
||||
recognizer.Result(); // Reset recognizer state
|
||||
return result;
|
||||
}
|
||||
|
||||
@@ -175,9 +171,9 @@ public class SpeechRecognizer {
|
||||
public void shutdown() {
|
||||
recorder.release();
|
||||
}
|
||||
|
||||
|
||||
private final class RecognizerThread extends Thread {
|
||||
|
||||
|
||||
private int remainingSamples;
|
||||
private int timeoutSamples;
|
||||
private final static int NO_TIMEOUT = -1;
|
||||
@@ -206,8 +202,6 @@ public class SpeechRecognizer {
|
||||
return;
|
||||
}
|
||||
|
||||
Log.d(TAG, "Starting decoding");
|
||||
|
||||
short[] buffer = new short[bufferSize];
|
||||
|
||||
while (!interrupted()
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
CFLAGS=-I../src
|
||||
LDFLAGS=-L../src -lvosk -ldl -lpthread -Wl,-rpath=../src
|
||||
|
||||
all: test_vosk test_vosk_speaker
|
||||
|
||||
test_vosk: test_vosk.o
|
||||
g++ $^ -o $@ $(LDFLAGS)
|
||||
|
||||
test_vosk_speaker: test_vosk_speaker.o
|
||||
g++ $^ -o $@ $(LDFLAGS)
|
||||
|
||||
%.o: %.c
|
||||
g++ $(CFLAGS) -c -o $@ $<
|
||||
|
||||
clean:
|
||||
rm -f *.o *.a test_vosk test_vosk_speaker
|
||||
@@ -0,0 +1,29 @@
|
||||
#include <vosk_api.h>
|
||||
#include <stdio.h>
|
||||
|
||||
int main() {
|
||||
FILE *wavin;
|
||||
char buf[3200];
|
||||
int nread, final;
|
||||
|
||||
VoskModel *model = vosk_model_new("model");
|
||||
VoskRecognizer *recognizer = vosk_recognizer_new(model, 16000.0);
|
||||
|
||||
wavin = fopen("test.wav", "rb");
|
||||
fseek(wavin, 44, SEEK_SET);
|
||||
while (!feof(wavin)) {
|
||||
nread = fread(buf, 1, sizeof(buf), wavin);
|
||||
final = vosk_recognizer_accept_waveform(recognizer, buf, nread);
|
||||
if (final) {
|
||||
printf("%s\n", vosk_recognizer_result(recognizer));
|
||||
} else {
|
||||
printf("%s\n", vosk_recognizer_partial_result(recognizer));
|
||||
}
|
||||
}
|
||||
printf("%s\n", vosk_recognizer_final_result(recognizer));
|
||||
|
||||
vosk_recognizer_free(recognizer);
|
||||
vosk_model_free(model);
|
||||
fclose(wavin);
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
#include <vosk_api.h>
|
||||
#include <stdio.h>
|
||||
|
||||
int main() {
|
||||
FILE *wavin;
|
||||
char buf[3200];
|
||||
int nread, final;
|
||||
|
||||
VoskModel *model = vosk_model_new("model");
|
||||
VoskSpkModel *spk_model = vosk_spk_model_new("spk-model");
|
||||
VoskRecognizer *recognizer = vosk_recognizer_new_spk(model, spk_model, 16000.0);
|
||||
|
||||
wavin = fopen("test.wav", "rb");
|
||||
fseek(wavin, 44, SEEK_SET);
|
||||
while (!feof(wavin)) {
|
||||
nread = fread(buf, 1, sizeof(buf), wavin);
|
||||
final = vosk_recognizer_accept_waveform(recognizer, buf, nread);
|
||||
if (final) {
|
||||
printf("%s\n", vosk_recognizer_result(recognizer));
|
||||
} else {
|
||||
printf("%s\n", vosk_recognizer_partial_result(recognizer));
|
||||
}
|
||||
}
|
||||
printf("%s\n", vosk_recognizer_final_result(recognizer));
|
||||
|
||||
vosk_recognizer_free(recognizer);
|
||||
vosk_spk_model_free(spk_model);
|
||||
vosk_model_free(model);
|
||||
return 0;
|
||||
}
|
||||
@@ -1,53 +0,0 @@
|
||||
KALDI_ROOT ?= $(HOME)/kaldi
|
||||
CFLAGS := -std=c++11 -g -O2 -DPIC -fPIC -Wno-unused-function
|
||||
CPPFLAGS := -I$(KALDI_ROOT)/src -I$(KALDI_ROOT)/tools/openfst/include -I../src -DFST_NO_DYNAMIC_LINKING
|
||||
|
||||
KALDI_LIBS = \
|
||||
${KALDI_ROOT}/src/online2/kaldi-online2.a \
|
||||
${KALDI_ROOT}/src/decoder/kaldi-decoder.a \
|
||||
${KALDI_ROOT}/src/ivector/kaldi-ivector.a \
|
||||
${KALDI_ROOT}/src/gmm/kaldi-gmm.a \
|
||||
${KALDI_ROOT}/src/nnet3/kaldi-nnet3.a \
|
||||
${KALDI_ROOT}/src/tree/kaldi-tree.a \
|
||||
${KALDI_ROOT}/src/feat/kaldi-feat.a \
|
||||
${KALDI_ROOT}/src/lat/kaldi-lat.a \
|
||||
${KALDI_ROOT}/src/hmm/kaldi-hmm.a \
|
||||
${KALDI_ROOT}/src/transform/kaldi-transform.a \
|
||||
${KALDI_ROOT}/src/cudamatrix/kaldi-cudamatrix.a \
|
||||
${KALDI_ROOT}/src/matrix/kaldi-matrix.a \
|
||||
${KALDI_ROOT}/src/fstext/kaldi-fstext.a \
|
||||
${KALDI_ROOT}/src/util/kaldi-util.a \
|
||||
${KALDI_ROOT}/src/base/kaldi-base.a \
|
||||
${KALDI_ROOT}/tools/openfst/lib/libfst.a \
|
||||
${KALDI_ROOT}/tools/openfst/lib/libfstngram.a \
|
||||
${KALDI_ROOT}/tools/OpenBLAS/libopenblas.a \
|
||||
-lgfortran -lstdc++
|
||||
|
||||
all: test.exe
|
||||
|
||||
test.exe: libkaldiwrap.so test.cs
|
||||
mcs test.cs gen/*.cs
|
||||
|
||||
VOSK_SOURCES = \
|
||||
vosk_wrap.c \
|
||||
../src/kaldi_recognizer.cc \
|
||||
../src/kaldi_recognizer.h \
|
||||
../src/model.cc \
|
||||
../src/model.h \
|
||||
../src/spk_model.cc \
|
||||
../src/spk_model.h \
|
||||
../src/vosk_api.cc \
|
||||
../src/vosk_api.h
|
||||
|
||||
libkaldiwrap.so: $(VOSK_SOURCES)
|
||||
$(CXX) -fpermissive $(CFLAGS) $(CPPFLAGS) -shared -o $@ $(VOSK_SOURCES) $(KALDI_LIBS)
|
||||
|
||||
vosk_wrap.c: ../src/vosk.i
|
||||
swig -csharp -DSWIG_CSHARP_NO_EXCEPTION_HELPER -dllimport "libkaldiwrap.so" \
|
||||
-namespace "Kaldi" -outdir gen -o vosk_wrap.c ../src/vosk.i
|
||||
|
||||
run:
|
||||
mono test.exe
|
||||
|
||||
clean:
|
||||
$(RM) *.so vosk_wrap.c *.o gen/*.cs test.exe
|
||||
@@ -0,0 +1,12 @@
|
||||
This is a nuget-based wrapper for libvosk library
|
||||
|
||||
See demo folder for example how to use the library. You can simply run
|
||||
"dotnet run" to run the demo. Make sure you unpacked the model and the
|
||||
test file.
|
||||
|
||||
See the nuget folder for the sources of the wrapper. Run build.sh to
|
||||
build nuget package.
|
||||
|
||||
Note we only support win64 and linux64 for now. No support for win32
|
||||
since it is a little bit painful to load the libraries depending on
|
||||
architecture. In theory we can add OSX some time or even Android.
|
||||
@@ -1,14 +1,15 @@
|
||||
using System;
|
||||
using System.IO;
|
||||
using Kaldi;
|
||||
using Vosk;
|
||||
|
||||
public class Test
|
||||
{
|
||||
public static void Main()
|
||||
{
|
||||
|
||||
Vosk.Vosk.SetLogLevel(0);
|
||||
Model model = new Model("model");
|
||||
KaldiRecognizer rec = new KaldiRecognizer(model, 16000.0f);
|
||||
VoskRecognizer rec = new VoskRecognizer(model, 16000.0f);
|
||||
|
||||
using(Stream source = File.OpenRead("test.wav")) {
|
||||
byte[] buffer = new byte[4096];
|
||||
@@ -23,7 +24,7 @@ public class Test
|
||||
}
|
||||
Console.WriteLine(rec.FinalResult());
|
||||
|
||||
rec = new KaldiRecognizer(model, 16000.0f);
|
||||
rec = new VoskRecognizer(model, 16000.0f);
|
||||
|
||||
using(Stream source = File.OpenRead("test.wav")) {
|
||||
byte[] buffer = new byte[4096];
|
||||
@@ -0,0 +1,13 @@
|
||||
<Project Sdk="Microsoft.NET.Sdk">
|
||||
|
||||
<PropertyGroup>
|
||||
<OutputType>Exe</OutputType>
|
||||
<TargetFramework>net5.0</TargetFramework>
|
||||
<RootNamespace>VoskDemo</RootNamespace>
|
||||
</PropertyGroup>
|
||||
|
||||
<ItemGroup>
|
||||
<PackageReference Include="Vosk" Version="0.3.19" />
|
||||
</ItemGroup>
|
||||
|
||||
</Project>
|
||||
@@ -0,0 +1,30 @@
|
||||
<?xml version="1.0"?>
|
||||
<package>
|
||||
<metadata>
|
||||
<id>Vosk</id>
|
||||
<version>0.3.19</version>
|
||||
<authors>Alpha Cephei Inc</authors>
|
||||
<owners>Alpha Cephei Inc</owners>
|
||||
<license type="expression">Apache-2.0</license>
|
||||
<projectUrl>https://alphacephei.com/vosk/</projectUrl>
|
||||
<requireLicenseAcceptance>false</requireLicenseAcceptance>
|
||||
<description>Vosk is an offline open source speech recognition toolkit. It enables speech recognition models for 16 languages and dialects - English, Indian English, German, French, Spanish, Portuguese, Chinese, Russian, Turkish, Vietnamese, Italian, Dutch, Catalan, Arabic, Greek, Farsi.
|
||||
|
||||
Vosk models are small (50 Mb) but provide continuous large vocabulary transcription, zero-latency response with streaming API, reconfigurable vocabulary and speaker identification.
|
||||
|
||||
Speech recognition bindings implemented for various programming languages like Python, Java, Node.JS, C#, C++ and others.
|
||||
|
||||
Vosk supplies speech recognition for chatbots, smart home appliances, virtual assistants. It can also create subtitles for movies, transcription for lectures and interviews.
|
||||
|
||||
Vosk scales from small devices like Raspberry Pi or Android smartphone to big clusters.</description>
|
||||
<releaseNotes>See for details https://github.com/alphacep/vosk-api/releases</releaseNotes>
|
||||
<copyright>Copyright 2020 Alpha Cephei Inc</copyright>
|
||||
<tags>speech recognition voice stt asr speech-to-text ai offline privacy</tags>
|
||||
<dependencies>
|
||||
<group targetFramework=".NETStandard2.0"/>
|
||||
</dependencies>
|
||||
</metadata>
|
||||
<files>
|
||||
<file src="**" exclude="src/*.cs;build.sh;**/.keep-me;*.nupkg" />
|
||||
</files>
|
||||
</package>
|
||||
Executable
+2
@@ -0,0 +1,2 @@
|
||||
mcs -out:lib/netstandard2.0/Vosk.dll -target:library src/*.cs
|
||||
nuget pack
|
||||
@@ -0,0 +1,10 @@
|
||||
<Project xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<ItemGroup>
|
||||
<NativeLibs Include="$(MSBuildThisFileDirectory)\lib\linux-x64\*.so" Condition="'$([MSBuild]::IsOsPlatform(Linux))'" />
|
||||
<NativeLibs Include="$(MSBuildThisFileDirectory)\lib\win-x64\*.dll" Condition="'$([MSBuild]::IsOsPlatform(Windows))'" />
|
||||
<None Include="@(NativeLibs)">
|
||||
<Link>%(FileName)%(Extension)</Link>
|
||||
<CopyToOutputDirectory>PreserveNewest</CopyToOutputDirectory>
|
||||
</None>
|
||||
</ItemGroup>
|
||||
</Project>
|
||||
@@ -0,0 +1,41 @@
|
||||
namespace Vosk {
|
||||
|
||||
public class Model : global::System.IDisposable {
|
||||
private global::System.Runtime.InteropServices.HandleRef handle;
|
||||
|
||||
internal Model(global::System.IntPtr cPtr) {
|
||||
handle = new global::System.Runtime.InteropServices.HandleRef(this, cPtr);
|
||||
}
|
||||
|
||||
internal static global::System.Runtime.InteropServices.HandleRef getCPtr(Model obj) {
|
||||
return (obj == null) ? new global::System.Runtime.InteropServices.HandleRef(null, global::System.IntPtr.Zero) : obj.handle;
|
||||
}
|
||||
|
||||
~Model() {
|
||||
Dispose(false);
|
||||
}
|
||||
|
||||
public void Dispose() {
|
||||
Dispose(true);
|
||||
global::System.GC.SuppressFinalize(this);
|
||||
}
|
||||
|
||||
protected virtual void Dispose(bool disposing) {
|
||||
lock(this) {
|
||||
if (handle.Handle != global::System.IntPtr.Zero) {
|
||||
VoskPINVOKE.delete_Model(handle);
|
||||
handle = new global::System.Runtime.InteropServices.HandleRef(null, global::System.IntPtr.Zero);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
public Model(string model_path) : this(VoskPINVOKE.new_Model(model_path)) {
|
||||
}
|
||||
|
||||
public int vosk_model_find_word(string word) {
|
||||
return VoskPINVOKE.Model_vosk_model_find_word(handle, word);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
namespace Vosk {
|
||||
|
||||
public class SpkModel : global::System.IDisposable {
|
||||
private global::System.Runtime.InteropServices.HandleRef handle;
|
||||
|
||||
internal SpkModel(global::System.IntPtr cPtr) {
|
||||
handle = new global::System.Runtime.InteropServices.HandleRef(this, cPtr);
|
||||
}
|
||||
|
||||
internal static global::System.Runtime.InteropServices.HandleRef getCPtr(SpkModel obj) {
|
||||
return (obj == null) ? new global::System.Runtime.InteropServices.HandleRef(null, global::System.IntPtr.Zero) : obj.handle;
|
||||
}
|
||||
|
||||
~SpkModel() {
|
||||
Dispose(false);
|
||||
}
|
||||
|
||||
public void Dispose() {
|
||||
Dispose(true);
|
||||
global::System.GC.SuppressFinalize(this);
|
||||
}
|
||||
|
||||
protected virtual void Dispose(bool disposing) {
|
||||
lock(this) {
|
||||
if (handle.Handle != global::System.IntPtr.Zero) {
|
||||
VoskPINVOKE.delete_SpkModel(handle);
|
||||
handle = new global::System.Runtime.InteropServices.HandleRef(null, global::System.IntPtr.Zero);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
public SpkModel(string model_path) : this(VoskPINVOKE.new_SpkModel(model_path)) {
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,10 @@
|
||||
namespace Vosk {
|
||||
|
||||
public class Vosk {
|
||||
public static void SetLogLevel(int level) {
|
||||
VoskPINVOKE.SetLogLevel(level);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,57 @@
|
||||
namespace Vosk {
|
||||
|
||||
class VoskPINVOKE {
|
||||
|
||||
static VoskPINVOKE() {
|
||||
}
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_model_new")]
|
||||
public static extern global::System.IntPtr new_Model(string jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_model_free")]
|
||||
public static extern void delete_Model(global::System.Runtime.InteropServices.HandleRef jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_model_find_word")]
|
||||
public static extern int Model_vosk_model_find_word(global::System.Runtime.InteropServices.HandleRef jarg1, string jarg2);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_spk_model_new")]
|
||||
public static extern global::System.IntPtr new_SpkModel(string jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_spk_model_free")]
|
||||
public static extern void delete_SpkModel(global::System.Runtime.InteropServices.HandleRef jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_new")]
|
||||
public static extern global::System.IntPtr new_VoskRecognizer(global::System.Runtime.InteropServices.HandleRef jarg1, float jarg2);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_new_spk")]
|
||||
public static extern global::System.IntPtr new_VoskRecognizerSpk(global::System.Runtime.InteropServices.HandleRef jarg1, global::System.Runtime.InteropServices.HandleRef jarg2, float jarg3);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_new_grm")]
|
||||
public static extern global::System.IntPtr new_VoskRecognizerGrm(global::System.Runtime.InteropServices.HandleRef jarg1, float jarg2, string jarg3);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_free")]
|
||||
public static extern void delete_VoskRecognizer(global::System.Runtime.InteropServices.HandleRef jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_accept_waveform")]
|
||||
public static extern bool VoskRecognizer_AcceptWaveform(global::System.Runtime.InteropServices.HandleRef jarg1, [global::System.Runtime.InteropServices.In, global::System.Runtime.InteropServices.MarshalAs(global::System.Runtime.InteropServices.UnmanagedType.LPArray)]byte[] jarg2, int jarg3);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_accept_waveform_s")]
|
||||
public static extern bool VoskRecognizer_AcceptWaveformShort(global::System.Runtime.InteropServices.HandleRef jarg1, [global::System.Runtime.InteropServices.In, global::System.Runtime.InteropServices.MarshalAs(global::System.Runtime.InteropServices.UnmanagedType.LPArray)]short[] jarg2, int jarg3);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_accept_waveform_f")]
|
||||
public static extern bool VoskRecognizer_AcceptWaveformFloat(global::System.Runtime.InteropServices.HandleRef jarg1, [global::System.Runtime.InteropServices.In, global::System.Runtime.InteropServices.MarshalAs(global::System.Runtime.InteropServices.UnmanagedType.LPArray)]float[] jarg2, int jarg3);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_result")]
|
||||
public static extern global::System.IntPtr VoskRecognizer_Result(global::System.Runtime.InteropServices.HandleRef jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_partial_result")]
|
||||
public static extern global::System.IntPtr VoskRecognizer_PartialResult(global::System.Runtime.InteropServices.HandleRef jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_final_result")]
|
||||
public static extern global::System.IntPtr VoskRecognizer_FinalResult(global::System.Runtime.InteropServices.HandleRef jarg1);
|
||||
|
||||
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_set_log_level")]
|
||||
public static extern void SetLogLevel(int jarg1);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
namespace Vosk {
|
||||
|
||||
public class VoskRecognizer : global::System.IDisposable {
|
||||
private global::System.Runtime.InteropServices.HandleRef handle;
|
||||
|
||||
internal VoskRecognizer(global::System.IntPtr cPtr) {
|
||||
handle = new global::System.Runtime.InteropServices.HandleRef(this, cPtr);
|
||||
}
|
||||
|
||||
internal static global::System.Runtime.InteropServices.HandleRef getCPtr(VoskRecognizer obj) {
|
||||
return (obj == null) ? new global::System.Runtime.InteropServices.HandleRef(null, global::System.IntPtr.Zero) : obj.handle;
|
||||
}
|
||||
|
||||
~VoskRecognizer() {
|
||||
Dispose(false);
|
||||
}
|
||||
|
||||
public void Dispose() {
|
||||
Dispose(true);
|
||||
global::System.GC.SuppressFinalize(this);
|
||||
}
|
||||
|
||||
protected virtual void Dispose(bool disposing) {
|
||||
lock(this) {
|
||||
if (handle.Handle != global::System.IntPtr.Zero) {
|
||||
VoskPINVOKE.delete_VoskRecognizer(handle);
|
||||
handle = new global::System.Runtime.InteropServices.HandleRef(null, global::System.IntPtr.Zero);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
public VoskRecognizer(Model model, float sample_rate) : this(VoskPINVOKE.new_VoskRecognizer(Model.getCPtr(model), sample_rate)) {
|
||||
}
|
||||
|
||||
public VoskRecognizer(Model model, SpkModel spk_model, float sample_rate) : this(VoskPINVOKE.new_VoskRecognizerSpk(Model.getCPtr(model), SpkModel.getCPtr(spk_model), sample_rate)) {
|
||||
}
|
||||
|
||||
public VoskRecognizer(Model model, float sample_rate, string grammar) : this(VoskPINVOKE.new_VoskRecognizerGrm(Model.getCPtr(model), sample_rate, grammar)) {
|
||||
}
|
||||
|
||||
public bool AcceptWaveform(byte[] data, int len) {
|
||||
return VoskPINVOKE.VoskRecognizer_AcceptWaveform(handle, data, len);
|
||||
}
|
||||
|
||||
public bool AcceptWaveform(short[] sdata, int len) {
|
||||
return VoskPINVOKE.VoskRecognizer_AcceptWaveformShort(handle, sdata, len);
|
||||
}
|
||||
|
||||
public bool AcceptWaveform(float[] fdata, int len) {
|
||||
return VoskPINVOKE.VoskRecognizer_AcceptWaveformFloat(handle, fdata, len);
|
||||
}
|
||||
|
||||
public string Result() {
|
||||
return global::System.Runtime.InteropServices.Marshal.PtrToStringUTF8(VoskPINVOKE.VoskRecognizer_Result(handle));
|
||||
}
|
||||
|
||||
public string PartialResult() {
|
||||
return global::System.Runtime.InteropServices.Marshal.PtrToStringUTF8(VoskPINVOKE.VoskRecognizer_PartialResult(handle));
|
||||
}
|
||||
|
||||
public string FinalResult() {
|
||||
return global::System.Runtime.InteropServices.Marshal.PtrToStringUTF8(VoskPINVOKE.VoskRecognizer_FinalResult(handle));
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
+1
-21
@@ -1,21 +1 @@
|
||||
## Accuracy issues
|
||||
|
||||
Accuracy of modern systems is still unstable, that means sometimes you can have a very good accuracy and sometimes it could be bad.
|
||||
It is hard to make a system that will work good. And there could be many reasons for that:
|
||||
|
||||
* Audio has very bad quality
|
||||
* Vocabulary of the system doesn't match (yes, we still use fixed vocabulary)
|
||||
* Audio conditions like accent were not really the ones that were used in training
|
||||
* Some unpredictable audio issues like frame drop or frame coding bugs
|
||||
* Software bugs
|
||||
|
||||
It is hard to guess what is going on under the hood without getting your hands dirty. For that reason in case of any accuracy
|
||||
issues you must provide the following for analysis:
|
||||
|
||||
* Who are you, where are you from and why are you doing that. We don't like dealing with anonymous
|
||||
* The complete and exact description of the system you want to build - what is it going to do, what do you want to build
|
||||
* The precise description of hardware you are trying to run the system on
|
||||
* The detailed list of software versions you are using
|
||||
* Audio samples to demonstrate the problem together with the reference transcription for those samples
|
||||
|
||||
Remember, the more information you provide the faster you get a solution.
|
||||
See https://alphacephei.com/vosk/accuracy
|
||||
|
||||
+1
-42
@@ -1,42 +1 @@
|
||||
## Updating the language model
|
||||
|
||||
The Kaldi model used in Vosk is compiled from 3 data sources:
|
||||
|
||||
* dictionary
|
||||
* acoustic model
|
||||
* language model
|
||||
|
||||
You can rebuild all three with different level of effort, but sometimes you just
|
||||
need to adjust the probability of the words to improve the recognition. For
|
||||
that it is enough to recompile the language model from the text. To do that
|
||||
|
||||
1) Take a text that reflects the speech you want to recognize
|
||||
2) Remove punctuation, convert everything to the lowercase, you can do it with a python script
|
||||
3) Build openfst and opengrm inside kaldi
|
||||
|
||||
```
|
||||
export KALDI_ROOT=`pwd`/kaldi
|
||||
git clone https://github.com/kaldi-asr/kaldi
|
||||
cd kaldi/tools
|
||||
make
|
||||
# install all required dependencies and repeat `make` if needed
|
||||
extras/install_opengrm.sh
|
||||
```
|
||||
|
||||
4) Now lets build a grammar
|
||||
|
||||
```
|
||||
export PATH=$KALDI_ROOT/tools/openfst/bin:$PATH
|
||||
export LD_LIBRARY_PATH=$KALDI_ROOT/tools/openfst/lib/fst
|
||||
cd model
|
||||
fstsymbols --save_osymbols=words.txt Gr.fst > /dev/null
|
||||
farcompilestrings --fst_type=compact --symbols=words.txt --keep_symbols text.txt | \
|
||||
ngramcount | ngrammake | \
|
||||
fstconvert --fst_type=ngram > Gr.fst
|
||||
```
|
||||
|
||||
Use created Gr.fst instead of standard one in your model.
|
||||
|
||||
For more details see OpenGRM documentation http://www.opengrm.org/twiki/bin/view/GRM/NGramLibrary
|
||||
|
||||
You can not introduce new words this way, that is something we will cover later.
|
||||
See https://alphacephei.com/vosk/adaptation
|
||||
|
||||
+1
-67
@@ -1,67 +1 @@
|
||||
# Models
|
||||
|
||||
This is the list of models compatible with Vosk-API.
|
||||
|
||||
To add a new model here create an issue on Github.
|
||||
|
||||
### English
|
||||
|
||||
| Model | Size | Accuracy | Notes |
|
||||
|-----------------------------------------------------------------------------------------------------------|-------|------------|----------------------------------------------------------------------------------------------|
|
||||
| [kaldi-en-us-aspire-0.1](http://alphacephei.com/kaldi/kaldi-en-us-aspire-0.1.tar.gz) | 363M | TBD | Trained on Fisher + more or less recent LM. Pretty outdated but still ok even even for calls |
|
||||
| [alphacep-model-android-en-us-0.3](http://alphacephei.com/kaldi/alphacep-model-android-en-us-0.3.tar.gz) | 36M | TBD | Lightweight wideband model for Android and RPi |
|
||||
|
||||
### Chinese
|
||||
|
||||
| Model | Size | Accuracy | Notes |
|
||||
|-----------------------------------------------------------------------------------------------------------|-------|------------|----------------------------------------------------------------------------------------------|
|
||||
| [kaldi-cn-0.1.tar.gz](http://alphacephei.com/kaldi/kaldi-cn-0.1.tar.gz) | 195M | TBD | Big narrowband Chinese model for server processing |
|
||||
| [alphacep-model-android-cn-0.3](http://alphacephei.com/kaldi/alphacep-model-android-cn-0.3.tar.gz) | 32M | TBD | Lightweight wideband model for Android and RPi |
|
||||
|
||||
### Russian
|
||||
|
||||
| Model | Size | Accuracy | Notes |
|
||||
|-----------------------------------------------------------------------------------------------------------|-------|------------|----------------------------------------------------------------------------------------------|
|
||||
| [kaldi-ru-0.9.tar.gz](http://alphacephei.com/kaldi/kaldi-ru-0.9.tar.gz) | 2.5G | TBD | Big narrowband Russian model for server processing |
|
||||
| [alphacep-model-android-ru-0.3](http://alphacephei.com/kaldi/alphacep-model-android-ru-0.3.tar.gz) | 39M | TBD | Lightweight wideband model for Android and RPi |
|
||||
|
||||
### French
|
||||
|
||||
| Model | Size | Accuracy | Notes |
|
||||
|-------------------------------------------------------------------------------------------------------------------------|-------|------------|----------------------------------------------------------------------------------------------|
|
||||
| [alphacep-model-android-fr-pguyot-0.3](http://alphacephei.com/kaldi/alphacep-model-android-fr-pguyot-0.3.tar.gz) | 39M | TBD | Lightweight wideband model for Android and RPi trained by [Paul Guyot](https://github.com/pguyot/zamia-speech/releases) |
|
||||
|
||||
### German
|
||||
|
||||
| Model | Size | Accuracy | Notes |
|
||||
|-----------------------------------------------------------------------------------------------------------|-------|------------|------------------------------------------------------------------------------------------------------|
|
||||
| [tuda-de](http://ltdata1.informatik.uni-hamburg.de/kaldi_tuda_de/de_400k_nnet3chain_tdnn1f_2048_sp_bi.tar.bz2) | 566M | TBD | Wideband server model from [tuda-de](https://github.com/uhh-lt/kaldi-tuda-de) |
|
||||
| [alphacep-model-android-de-zamia-0.3](http://alphacephei.com/kaldi/alphacep-model-android-de-zamia-0.3.tar.gz) | 49M | TBD | Lightweight wideband model for Android and RPi |
|
||||
|
||||
### Spanish
|
||||
|
||||
| Model | Size | Accuracy | Notes |
|
||||
|-----------------------------------------------------------------------------------------------------------|-------|------------|----------------------------------------------------------------------------------------------|
|
||||
| [alphacep-model-android-es-0.3](http://alphacephei.com/kaldi/alphacep-model-android-es-0.3.tar.gz) | 33M | TBD | Lightweight wideband model for Android and RPi |
|
||||
|
||||
### Portuguese
|
||||
|
||||
| Model | Size | Accuracy | Notes |
|
||||
|-----------------------------------------------------------------------------------------------------------|-------|------------|----------------------------------------------------------------------------------------------|
|
||||
| [alphacep-model-android-pt-0.3](http://alphacephei.com/kaldi/alphacep-model-android-pt-0.3.tar.gz) | 31M | TBD | Lightweight wideband model for Android and RPi |
|
||||
|
||||
### Dutch
|
||||
|
||||
https://github.com/opensource-spraakherkenning-nl/Kaldi_NL
|
||||
|
||||
### Greek
|
||||
|
||||
| Model | Size | Accuracy | Notes |
|
||||
|-----------------------------------------------------------------------------------------------------------|-------|------------|----------------------------------------------------------------------------------------------|
|
||||
| [kaldi-el-gr-0.6.tar.gz](http://alphacephei.com/kaldi/kaldi-el-gr-0.6.tar.gz) | 1.1G | TBD | Big narrowband Greek model for server processing, not extremely accurate though |
|
||||
|
||||
### Vietnamese
|
||||
|
||||
| Model | Size | Accuracy | Notes |
|
||||
|-----------------------------------------------------------------------------------------------------------|-------|------------|----------------------------------------------------------------------------------------------|
|
||||
| [alphacep-model-android-vn-0.3](http://alphacephei.com/kaldi/alphacep-model-android-vn-0.3.tar.gz) | 32M | TBD | Lightweight wideband model for Android and RPi |
|
||||
See https://alphacephei.com/vosk/models
|
||||
|
||||
@@ -16,56 +16,12 @@
|
||||
9237523C240C642000DD6076 /* libkaldiwrap.a in Frameworks */ = {isa = PBXBuildFile; fileRef = 9237523A240C642000DD6076 /* libkaldiwrap.a */; };
|
||||
92375244240C6DAF00DD6076 /* Accelerate.framework in Frameworks */ = {isa = PBXBuildFile; fileRef = 92375243240C6DAF00DD6076 /* Accelerate.framework */; };
|
||||
92375246240C6DC900DD6076 /* libstdc++.tbd in Frameworks */ = {isa = PBXBuildFile; fileRef = 92375245240C6DC900DD6076 /* libstdc++.tbd */; };
|
||||
92375266240C6EFE00DD6076 /* disambig_tid.int in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375248240C6E3D00DD6076 /* disambig_tid.int */; };
|
||||
92375267240C6EFE00DD6076 /* final.mdl in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375249240C6E3D00DD6076 /* final.mdl */; };
|
||||
92375268240C6EFE00DD6076 /* Gr.fst in CopyFiles */ = {isa = PBXBuildFile; fileRef = 9237524A240C6E3D00DD6076 /* Gr.fst */; };
|
||||
92375269240C6EFE00DD6076 /* HCLr.fst in CopyFiles */ = {isa = PBXBuildFile; fileRef = 9237524B240C6E3D00DD6076 /* HCLr.fst */; };
|
||||
9237526A240C6EFE00DD6076 /* mfcc.conf in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375253240C6E3D00DD6076 /* mfcc.conf */; };
|
||||
9237526B240C6EFE00DD6076 /* word_boundary.int in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375254240C6E3D00DD6076 /* word_boundary.int */; };
|
||||
9237526C240C6EFE00DD6076 /* words.txt in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375255240C6E3D00DD6076 /* words.txt */; };
|
||||
9237526E240C6F1500DD6076 /* final.dubm in CopyFiles */ = {isa = PBXBuildFile; fileRef = 9237524D240C6E3D00DD6076 /* final.dubm */; };
|
||||
9237526F240C6F1500DD6076 /* final.ie in CopyFiles */ = {isa = PBXBuildFile; fileRef = 9237524E240C6E3D00DD6076 /* final.ie */; };
|
||||
92375270240C6F1500DD6076 /* final.mat in CopyFiles */ = {isa = PBXBuildFile; fileRef = 9237524F240C6E3D00DD6076 /* final.mat */; };
|
||||
92375271240C6F1500DD6076 /* global_cmvn.stats in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375250240C6E3D00DD6076 /* global_cmvn.stats */; };
|
||||
92375272240C6F1500DD6076 /* online_cmvn.conf in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375251240C6E3D00DD6076 /* online_cmvn.conf */; };
|
||||
92375273240C6F1500DD6076 /* splice.conf in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375252240C6E3D00DD6076 /* splice.conf */; };
|
||||
92375274240C6F1E00DD6076 /* 10001-90210-01803.wav in Resources */ = {isa = PBXBuildFile; fileRef = 92375256240C6E3D00DD6076 /* 10001-90210-01803.wav */; };
|
||||
92BACED125BE125A00B5CC93 /* vosk-model-small-en-us-0.15 in Resources */ = {isa = PBXBuildFile; fileRef = 928CC50C25BE124400490481 /* vosk-model-small-en-us-0.15 */; };
|
||||
92D6B8D325BDFEAC007FF08D /* VoskModel.swift in Sources */ = {isa = PBXBuildFile; fileRef = 92D6B8D225BDFEAC007FF08D /* VoskModel.swift */; };
|
||||
92D86BD6253F823F0040D53F /* vosk-model-spk-0.4 in Resources */ = {isa = PBXBuildFile; fileRef = 92D86BD4253F823F0040D53F /* vosk-model-spk-0.4 */; };
|
||||
/* End PBXBuildFile section */
|
||||
|
||||
/* Begin PBXCopyFilesBuildPhase section */
|
||||
92375265240C6ECF00DD6076 /* CopyFiles */ = {
|
||||
isa = PBXCopyFilesBuildPhase;
|
||||
buildActionMask = 2147483647;
|
||||
dstPath = "model-en";
|
||||
dstSubfolderSpec = 7;
|
||||
files = (
|
||||
92375266240C6EFE00DD6076 /* disambig_tid.int in CopyFiles */,
|
||||
92375267240C6EFE00DD6076 /* final.mdl in CopyFiles */,
|
||||
92375268240C6EFE00DD6076 /* Gr.fst in CopyFiles */,
|
||||
92375269240C6EFE00DD6076 /* HCLr.fst in CopyFiles */,
|
||||
9237526A240C6EFE00DD6076 /* mfcc.conf in CopyFiles */,
|
||||
9237526B240C6EFE00DD6076 /* word_boundary.int in CopyFiles */,
|
||||
9237526C240C6EFE00DD6076 /* words.txt in CopyFiles */,
|
||||
);
|
||||
runOnlyForDeploymentPostprocessing = 0;
|
||||
};
|
||||
9237526D240C6F0400DD6076 /* CopyFiles */ = {
|
||||
isa = PBXCopyFilesBuildPhase;
|
||||
buildActionMask = 2147483647;
|
||||
dstPath = "model-en/ivector";
|
||||
dstSubfolderSpec = 7;
|
||||
files = (
|
||||
9237526E240C6F1500DD6076 /* final.dubm in CopyFiles */,
|
||||
9237526F240C6F1500DD6076 /* final.ie in CopyFiles */,
|
||||
92375270240C6F1500DD6076 /* final.mat in CopyFiles */,
|
||||
92375271240C6F1500DD6076 /* global_cmvn.stats in CopyFiles */,
|
||||
92375272240C6F1500DD6076 /* online_cmvn.conf in CopyFiles */,
|
||||
92375273240C6F1500DD6076 /* splice.conf in CopyFiles */,
|
||||
);
|
||||
runOnlyForDeploymentPostprocessing = 0;
|
||||
};
|
||||
/* End PBXCopyFilesBuildPhase section */
|
||||
|
||||
/* Begin PBXFileReference section */
|
||||
9237521E240C550B00DD6076 /* VoskApiTest.app */ = {isa = PBXFileReference; explicitFileType = wrapper.application; includeInIndex = 0; path = VoskApiTest.app; sourceTree = BUILT_PRODUCTS_DIR; };
|
||||
92375221240C550B00DD6076 /* AppDelegate.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = AppDelegate.swift; sourceTree = "<group>"; };
|
||||
@@ -78,22 +34,12 @@
|
||||
9237523A240C642000DD6076 /* libkaldiwrap.a */ = {isa = PBXFileReference; lastKnownFileType = archive.ar; path = libkaldiwrap.a; sourceTree = "<group>"; };
|
||||
92375243240C6DAF00DD6076 /* Accelerate.framework */ = {isa = PBXFileReference; lastKnownFileType = wrapper.framework; name = Accelerate.framework; path = System/Library/Frameworks/Accelerate.framework; sourceTree = SDKROOT; };
|
||||
92375245240C6DC900DD6076 /* libstdc++.tbd */ = {isa = PBXFileReference; lastKnownFileType = "sourcecode.text-based-dylib-definition"; name = "libstdc++.tbd"; path = "usr/lib/libstdc++.tbd"; sourceTree = SDKROOT; };
|
||||
92375248240C6E3D00DD6076 /* disambig_tid.int */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = text; path = disambig_tid.int; sourceTree = "<group>"; };
|
||||
92375249240C6E3D00DD6076 /* final.mdl */ = {isa = PBXFileReference; lastKnownFileType = file; path = final.mdl; sourceTree = "<group>"; };
|
||||
9237524A240C6E3D00DD6076 /* Gr.fst */ = {isa = PBXFileReference; lastKnownFileType = file; path = Gr.fst; sourceTree = "<group>"; };
|
||||
9237524B240C6E3D00DD6076 /* HCLr.fst */ = {isa = PBXFileReference; lastKnownFileType = file; path = HCLr.fst; sourceTree = "<group>"; };
|
||||
9237524D240C6E3D00DD6076 /* final.dubm */ = {isa = PBXFileReference; lastKnownFileType = file; path = final.dubm; sourceTree = "<group>"; };
|
||||
9237524E240C6E3D00DD6076 /* final.ie */ = {isa = PBXFileReference; lastKnownFileType = file; path = final.ie; sourceTree = "<group>"; };
|
||||
9237524F240C6E3D00DD6076 /* final.mat */ = {isa = PBXFileReference; lastKnownFileType = file; path = final.mat; sourceTree = "<group>"; };
|
||||
92375250240C6E3D00DD6076 /* global_cmvn.stats */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = text; path = global_cmvn.stats; sourceTree = "<group>"; };
|
||||
92375251240C6E3D00DD6076 /* online_cmvn.conf */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = text; path = online_cmvn.conf; sourceTree = "<group>"; };
|
||||
92375252240C6E3D00DD6076 /* splice.conf */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = text; path = splice.conf; sourceTree = "<group>"; };
|
||||
92375253240C6E3D00DD6076 /* mfcc.conf */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = text; path = mfcc.conf; sourceTree = "<group>"; };
|
||||
92375254240C6E3D00DD6076 /* word_boundary.int */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = text; path = word_boundary.int; sourceTree = "<group>"; };
|
||||
92375255240C6E3D00DD6076 /* words.txt */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = text; path = words.txt; sourceTree = "<group>"; };
|
||||
92375256240C6E3D00DD6076 /* 10001-90210-01803.wav */ = {isa = PBXFileReference; lastKnownFileType = audio.wav; path = "10001-90210-01803.wav"; sourceTree = "<group>"; };
|
||||
928CC50C25BE124400490481 /* vosk-model-small-en-us-0.15 */ = {isa = PBXFileReference; lastKnownFileType = folder; name = "vosk-model-small-en-us-0.15"; path = "/Users/shmyrev/Documents/IOS/VoskApiTest/VoskApiTest/Vosk/vosk-model-small-en-us-0.15"; sourceTree = "<absolute>"; };
|
||||
92AA22AD244CDD1200DA464B /* vosk_api.h */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = sourcecode.c.h; path = vosk_api.h; sourceTree = "<group>"; };
|
||||
92AA22AE244CDD5200DA464B /* bridging.h */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = sourcecode.c.h; path = bridging.h; sourceTree = "<group>"; };
|
||||
92D6B8D225BDFEAC007FF08D /* VoskModel.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = VoskModel.swift; sourceTree = "<group>"; };
|
||||
92D86BD4253F823F0040D53F /* vosk-model-spk-0.4 */ = {isa = PBXFileReference; lastKnownFileType = folder; path = "vosk-model-spk-0.4"; sourceTree = "<group>"; };
|
||||
/* End PBXFileReference section */
|
||||
|
||||
/* Begin PBXFrameworksBuildPhase section */
|
||||
@@ -139,6 +85,7 @@
|
||||
9237522A240C550B00DD6076 /* LaunchScreen.storyboard */,
|
||||
9237522D240C550B00DD6076 /* Info.plist */,
|
||||
92375233240C558900DD6076 /* Vosk.swift */,
|
||||
92D6B8D225BDFEAC007FF08D /* VoskModel.swift */,
|
||||
);
|
||||
path = VoskApiTest;
|
||||
sourceTree = "<group>";
|
||||
@@ -146,7 +93,8 @@
|
||||
92375239240C642000DD6076 /* Vosk */ = {
|
||||
isa = PBXGroup;
|
||||
children = (
|
||||
92375247240C6E3D00DD6076 /* model-android */,
|
||||
928CC50C25BE124400490481 /* vosk-model-small-en-us-0.15 */,
|
||||
92D86BD4253F823F0040D53F /* vosk-model-spk-0.4 */,
|
||||
92375256240C6E3D00DD6076 /* 10001-90210-01803.wav */,
|
||||
92AA22AD244CDD1200DA464B /* vosk_api.h */,
|
||||
9237523A240C642000DD6076 /* libkaldiwrap.a */,
|
||||
@@ -164,34 +112,6 @@
|
||||
name = Frameworks;
|
||||
sourceTree = "<group>";
|
||||
};
|
||||
92375247240C6E3D00DD6076 /* model-android */ = {
|
||||
isa = PBXGroup;
|
||||
children = (
|
||||
92375248240C6E3D00DD6076 /* disambig_tid.int */,
|
||||
92375249240C6E3D00DD6076 /* final.mdl */,
|
||||
9237524A240C6E3D00DD6076 /* Gr.fst */,
|
||||
9237524B240C6E3D00DD6076 /* HCLr.fst */,
|
||||
9237524C240C6E3D00DD6076 /* ivector */,
|
||||
92375253240C6E3D00DD6076 /* mfcc.conf */,
|
||||
92375254240C6E3D00DD6076 /* word_boundary.int */,
|
||||
92375255240C6E3D00DD6076 /* words.txt */,
|
||||
);
|
||||
path = "model-android";
|
||||
sourceTree = "<group>";
|
||||
};
|
||||
9237524C240C6E3D00DD6076 /* ivector */ = {
|
||||
isa = PBXGroup;
|
||||
children = (
|
||||
9237524D240C6E3D00DD6076 /* final.dubm */,
|
||||
9237524E240C6E3D00DD6076 /* final.ie */,
|
||||
9237524F240C6E3D00DD6076 /* final.mat */,
|
||||
92375250240C6E3D00DD6076 /* global_cmvn.stats */,
|
||||
92375251240C6E3D00DD6076 /* online_cmvn.conf */,
|
||||
92375252240C6E3D00DD6076 /* splice.conf */,
|
||||
);
|
||||
path = ivector;
|
||||
sourceTree = "<group>";
|
||||
};
|
||||
/* End PBXGroup section */
|
||||
|
||||
/* Begin PBXNativeTarget section */
|
||||
@@ -202,8 +122,6 @@
|
||||
9237521A240C550B00DD6076 /* Sources */,
|
||||
9237521B240C550B00DD6076 /* Frameworks */,
|
||||
9237521C240C550B00DD6076 /* Resources */,
|
||||
92375265240C6ECF00DD6076 /* CopyFiles */,
|
||||
9237526D240C6F0400DD6076 /* CopyFiles */,
|
||||
);
|
||||
buildRules = (
|
||||
);
|
||||
@@ -254,9 +172,11 @@
|
||||
isa = PBXResourcesBuildPhase;
|
||||
buildActionMask = 2147483647;
|
||||
files = (
|
||||
92BACED125BE125A00B5CC93 /* vosk-model-small-en-us-0.15 in Resources */,
|
||||
92375274240C6F1E00DD6076 /* 10001-90210-01803.wav in Resources */,
|
||||
9237522C240C550B00DD6076 /* LaunchScreen.storyboard in Resources */,
|
||||
92375229240C550B00DD6076 /* Assets.xcassets in Resources */,
|
||||
92D86BD6253F823F0040D53F /* vosk-model-spk-0.4 in Resources */,
|
||||
92375227240C550B00DD6076 /* Main.storyboard in Resources */,
|
||||
);
|
||||
runOnlyForDeploymentPostprocessing = 0;
|
||||
@@ -270,6 +190,7 @@
|
||||
files = (
|
||||
92375224240C550B00DD6076 /* ViewController.swift in Sources */,
|
||||
92375222240C550B00DD6076 /* AppDelegate.swift in Sources */,
|
||||
92D6B8D325BDFEAC007FF08D /* VoskModel.swift in Sources */,
|
||||
92375234240C558900DD6076 /* Vosk.swift in Sources */,
|
||||
);
|
||||
runOnlyForDeploymentPostprocessing = 0;
|
||||
@@ -412,7 +333,7 @@
|
||||
buildSettings = {
|
||||
ASSETCATALOG_COMPILER_APPICON_NAME = AppIcon;
|
||||
CLANG_ENABLE_MODULES = YES;
|
||||
ENABLE_BITCODE = NO;
|
||||
ENABLE_BITCODE = YES;
|
||||
INFOPLIST_FILE = VoskApiTest/Info.plist;
|
||||
LD_RUNPATH_SEARCH_PATHS = "$(inherited) @executable_path/Frameworks";
|
||||
LIBRARY_SEARCH_PATHS = (
|
||||
@@ -434,7 +355,7 @@
|
||||
buildSettings = {
|
||||
ASSETCATALOG_COMPILER_APPICON_NAME = AppIcon;
|
||||
CLANG_ENABLE_MODULES = YES;
|
||||
ENABLE_BITCODE = NO;
|
||||
ENABLE_BITCODE = YES;
|
||||
INFOPLIST_FILE = VoskApiTest/Info.plist;
|
||||
LD_RUNPATH_SEARCH_PATHS = "$(inherited) @executable_path/Frameworks";
|
||||
LIBRARY_SEARCH_PATHS = (
|
||||
|
||||
@@ -2,6 +2,9 @@
|
||||
<Workspace
|
||||
version = "1.0">
|
||||
<FileRef
|
||||
location = "self:VoskApiTest.xcodeproj">
|
||||
location = "group:/Users/shmyrev/Documents/IOS/VoskApiTest/VoskApiTest/Vosk/vosk-model-small-en-us-0.15">
|
||||
</FileRef>
|
||||
<FileRef
|
||||
location = "self:">
|
||||
</FileRef>
|
||||
</Workspace>
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<document type="com.apple.InterfaceBuilder3.CocoaTouch.Storyboard.XIB" version="3.0" toolsVersion="13771" targetRuntime="iOS.CocoaTouch" propertyAccessControl="none" useAutolayout="YES" useTraitCollections="YES" colorMatched="YES" initialViewController="BYZ-38-t0r">
|
||||
<device id="retina4_7" orientation="portrait">
|
||||
<document type="com.apple.InterfaceBuilder3.CocoaTouch.Storyboard.XIB" version="3.0" toolsVersion="13771" targetRuntime="iOS.CocoaTouch" propertyAccessControl="none" useAutolayout="YES" useTraitCollections="YES" colorMatched="YES" initialViewController="bdW-KL-Y8Z">
|
||||
<device id="retina5_5" orientation="portrait">
|
||||
<adaptation id="fullscreen"/>
|
||||
</device>
|
||||
<dependencies>
|
||||
@@ -10,23 +10,52 @@
|
||||
</dependencies>
|
||||
<scenes>
|
||||
<!--View Controller-->
|
||||
<scene sceneID="tne-QT-ifu">
|
||||
<scene sceneID="nEc-89-Iqu">
|
||||
<objects>
|
||||
<viewController id="BYZ-38-t0r" customClass="ViewController" customModule="VoskApiTest" customModuleProvider="target" sceneMemberID="viewController">
|
||||
<textView key="view" clipsSubviews="YES" multipleTouchEnabled="YES" contentMode="scaleToFill" editable="NO" textAlignment="natural" id="CtX-mx-X98">
|
||||
<rect key="frame" x="0.0" y="0.0" width="375" height="667"/>
|
||||
<autoresizingMask key="autoresizingMask" flexibleMaxX="YES" flexibleMaxY="YES"/>
|
||||
<viewController id="bdW-KL-Y8Z" customClass="ViewController" customModule="VoskApiTest" customModuleProvider="target" sceneMemberID="viewController">
|
||||
<layoutGuides>
|
||||
<viewControllerLayoutGuide type="top" id="Hyr-Dz-4mU"/>
|
||||
<viewControllerLayoutGuide type="bottom" id="w4A-5X-uBu"/>
|
||||
</layoutGuides>
|
||||
<view key="view" contentMode="scaleToFill" id="m5v-US-bvR">
|
||||
<rect key="frame" x="0.0" y="0.0" width="414" height="736"/>
|
||||
<autoresizingMask key="autoresizingMask" widthSizable="YES" heightSizable="YES"/>
|
||||
<subviews>
|
||||
<button opaque="NO" contentMode="scaleToFill" fixedFrame="YES" contentHorizontalAlignment="center" contentVerticalAlignment="center" buttonType="roundedRect" lineBreakMode="middleTruncation" translatesAutoresizingMaskIntoConstraints="NO" id="IaT-no-U3i" userLabel="Microphone">
|
||||
<rect key="frame" x="124" y="34" width="157" height="30"/>
|
||||
<autoresizingMask key="autoresizingMask" flexibleMaxX="YES" flexibleMaxY="YES"/>
|
||||
<state key="normal" title="Recognize Microphone"/>
|
||||
<connections>
|
||||
<action selector="runRecognizeMicrohpone:" destination="bdW-KL-Y8Z" eventType="touchUpInside" id="hGB-lz-N2B"/>
|
||||
</connections>
|
||||
</button>
|
||||
<button opaque="NO" contentMode="scaleToFill" fixedFrame="YES" contentHorizontalAlignment="center" contentVerticalAlignment="center" buttonType="roundedRect" lineBreakMode="middleTruncation" translatesAutoresizingMaskIntoConstraints="NO" id="GC5-nT-FQR" userLabel="File">
|
||||
<rect key="frame" x="90" y="84" width="221" height="41"/>
|
||||
<autoresizingMask key="autoresizingMask" flexibleMaxX="YES" flexibleMaxY="YES"/>
|
||||
<state key="normal" title="Recognize File"/>
|
||||
<connections>
|
||||
<action selector="runRecognizeFile:" destination="bdW-KL-Y8Z" eventType="touchUpInside" id="xp5-Yi-rnN"/>
|
||||
</connections>
|
||||
</button>
|
||||
<textView clipsSubviews="YES" multipleTouchEnabled="YES" contentMode="scaleToFill" fixedFrame="YES" text="Results here" textAlignment="natural" translatesAutoresizingMaskIntoConstraints="NO" id="w4X-cu-USq">
|
||||
<rect key="frame" x="11" y="112" width="383" height="569"/>
|
||||
<autoresizingMask key="autoresizingMask" flexibleMaxX="YES" flexibleMaxY="YES"/>
|
||||
<color key="backgroundColor" white="1" alpha="1" colorSpace="calibratedWhite"/>
|
||||
<fontDescription key="fontDescription" type="system" pointSize="14"/>
|
||||
<textInputTraits key="textInputTraits" autocapitalizationType="sentences"/>
|
||||
</textView>
|
||||
</subviews>
|
||||
<color key="backgroundColor" white="1" alpha="1" colorSpace="calibratedWhite"/>
|
||||
<fontDescription key="fontDescription" type="system" pointSize="14"/>
|
||||
<textInputTraits key="textInputTraits" autocapitalizationType="sentences"/>
|
||||
</textView>
|
||||
</view>
|
||||
<connections>
|
||||
<outlet property="mainText" destination="CtX-mx-X98" id="oJy-5J-NKp"/>
|
||||
<outlet property="mainText" destination="w4X-cu-USq" id="rZS-nz-Wql"/>
|
||||
<outlet property="recognizeFile" destination="GC5-nT-FQR" id="dRe-tc-IA0"/>
|
||||
<outlet property="recognizeMicrophone" destination="IaT-no-U3i" id="IuM-aa-pAP"/>
|
||||
</connections>
|
||||
</viewController>
|
||||
<placeholder placeholderIdentifier="IBFirstResponder" id="dkx-z0-nzr" sceneMemberID="firstResponder"/>
|
||||
<placeholder placeholderIdentifier="IBFirstResponder" id="nWA-4D-pA6" userLabel="First Responder" sceneMemberID="firstResponder"/>
|
||||
</objects>
|
||||
<point key="canvasLocation" x="32.799999999999997" y="32.833583208395808"/>
|
||||
<point key="canvasLocation" x="-17.39130434782609" y="-285.32608695652175"/>
|
||||
</scene>
|
||||
</scenes>
|
||||
</document>
|
||||
|
||||
@@ -3,30 +3,113 @@
|
||||
// VoskApiTest
|
||||
//
|
||||
// Created by Niсkolay Shmyrev on 01.03.20.
|
||||
// Copyright © 2020 Alpha Cephei. All rights reserved.
|
||||
// Copyright © 2020-2021 Alpha Cephei. All rights reserved.
|
||||
//
|
||||
|
||||
import UIKit
|
||||
import AVFoundation
|
||||
|
||||
enum WorkMode {
|
||||
case stopped
|
||||
case microphone
|
||||
case file
|
||||
}
|
||||
|
||||
class ViewController: UIViewController {
|
||||
|
||||
@IBOutlet var mainText: UITextView!
|
||||
|
||||
override func viewDidLoad() {
|
||||
super.viewDidLoad()
|
||||
|
||||
DispatchQueue.global(qos: .userInitiated).async {
|
||||
DispatchQueue.main.async {
|
||||
self.mainText.text = "Processing file..."
|
||||
var mode: WorkMode!
|
||||
|
||||
@IBOutlet weak var recognizeFile: UIButton!
|
||||
@IBOutlet weak var mainText: UITextView!
|
||||
@IBOutlet weak var recognizeMicrophone: UIButton!
|
||||
|
||||
var audioEngine : AVAudioEngine!
|
||||
var processingQueue: DispatchQueue!
|
||||
var model : VoskModel!
|
||||
|
||||
func setMode(mode: WorkMode) {
|
||||
switch mode {
|
||||
case .stopped:
|
||||
self.recognizeFile.isEnabled = true
|
||||
self.recognizeMicrophone.isEnabled = true
|
||||
self.recognizeMicrophone.setTitle("Recognize Microphone",for: .normal)
|
||||
case .microphone:
|
||||
self.recognizeFile.isEnabled = false
|
||||
self.recognizeMicrophone.isEnabled = true
|
||||
self.recognizeMicrophone.setTitle("Stop Microphone",for: .normal)
|
||||
self.mainText.text = ""
|
||||
case .file:
|
||||
self.recognizeFile.isEnabled = false
|
||||
self.recognizeMicrophone.isEnabled = false
|
||||
self.mainText.text = "Processing file..."
|
||||
}
|
||||
self.mode = mode
|
||||
}
|
||||
|
||||
func startAudioEngine() {
|
||||
do {
|
||||
|
||||
// Create a new audio engine.
|
||||
audioEngine = AVAudioEngine()
|
||||
|
||||
let inputNode = audioEngine.inputNode
|
||||
let formatInput = inputNode.inputFormat(forBus: 0)
|
||||
let formatPcm = AVAudioFormat.init(commonFormat: AVAudioCommonFormat.pcmFormatInt16, sampleRate: formatInput.sampleRate, channels: 1, interleaved: true)
|
||||
|
||||
let recognizer = Vosk(model: model, sampleRate: Float(formatInput.sampleRate))
|
||||
|
||||
inputNode.installTap(onBus: 0,
|
||||
bufferSize: UInt32(formatInput.sampleRate / 10),
|
||||
format: formatPcm) { buffer, time in
|
||||
self.processingQueue.async {
|
||||
let res = recognizer.recognizeData(buffer: buffer)
|
||||
DispatchQueue.main.async {
|
||||
self.mainText.text = res + "\n" + self.mainText.text
|
||||
}
|
||||
}
|
||||
}
|
||||
let vosk = Vosk()
|
||||
let res = vosk.recognizeFile()
|
||||
|
||||
// Start the stream of audio data.
|
||||
audioEngine.prepare()
|
||||
try audioEngine.start()
|
||||
} catch {
|
||||
print("Unable to start AVAudioEngine: \(error.localizedDescription)")
|
||||
}
|
||||
}
|
||||
|
||||
func stopAudioEngine() {
|
||||
audioEngine.stop()
|
||||
}
|
||||
|
||||
@IBAction func runRecognizeMicrohpone(_ sender: Any) {
|
||||
if (mode == .stopped) {
|
||||
setMode(mode: .microphone)
|
||||
startAudioEngine()
|
||||
} else {
|
||||
stopAudioEngine()
|
||||
setMode(mode: .stopped)
|
||||
}
|
||||
}
|
||||
|
||||
@IBAction func runRecognizeFile(_ sender: Any) {
|
||||
setMode(mode: .file)
|
||||
processingQueue.async {
|
||||
let recognizer = Vosk(model: self.model, sampleRate: 16000.0)
|
||||
let res = recognizer.recognizeFile()
|
||||
DispatchQueue.main.async {
|
||||
self.mainText.text = res
|
||||
self.setMode(mode: .stopped)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
override func viewDidLoad() {
|
||||
super.viewDidLoad()
|
||||
setMode(mode: .stopped)
|
||||
processingQueue = DispatchQueue(label: "recognizerQueue")
|
||||
model = VoskModel()
|
||||
}
|
||||
|
||||
override func didReceiveMemoryWarning() {
|
||||
super.didReceiveMemoryWarning()
|
||||
}
|
||||
|
||||
+33
-16
@@ -3,35 +3,52 @@
|
||||
// VoskApiTest
|
||||
//
|
||||
// Created by Niсkolay Shmyrev on 01.03.20.
|
||||
// Copyright © 2020 Alpha Cephei. All rights reserved.
|
||||
// Copyright © 2020-2021 Alpha Cephei. All rights reserved.
|
||||
//
|
||||
|
||||
import Foundation
|
||||
import AVFoundation
|
||||
|
||||
public final class Vosk {
|
||||
|
||||
var recognizer : OpaquePointer!
|
||||
|
||||
init(model: VoskModel, sampleRate: Float) {
|
||||
recognizer = vosk_recognizer_new_spk(model.model, model.spkModel, sampleRate)
|
||||
}
|
||||
|
||||
deinit {
|
||||
vosk_recognizer_free(recognizer);
|
||||
}
|
||||
|
||||
func recognizeFile() -> String {
|
||||
var sres = ""
|
||||
|
||||
if let resourcePath = Bundle.main.resourcePath {
|
||||
|
||||
let modelPath = resourcePath + "/model-en"
|
||||
|
||||
let model = vosk_model_new(modelPath);
|
||||
let recognizer = vosk_recognizer_new(model, 16000.0)
|
||||
|
||||
|
||||
let audioFile = URL(fileURLWithPath: resourcePath + "/10001-90210-01803.wav")
|
||||
|
||||
if let data = try? Data(contentsOf: audioFile) {
|
||||
let _ = data.withUnsafeBytes {
|
||||
vosk_recognizer_accept_waveform(recognizer, $0, Int32(data.count))
|
||||
}
|
||||
let res = vosk_recognizer_final_result(recognizer);
|
||||
sres = String(validatingUTF8: res!)!;
|
||||
print(sres);
|
||||
let _ = data.withUnsafeBytes {
|
||||
vosk_recognizer_accept_waveform(recognizer, $0, Int32(data.count))
|
||||
}
|
||||
let res = vosk_recognizer_final_result(recognizer);
|
||||
sres = String(validatingUTF8: res!)!;
|
||||
print(sres);
|
||||
}
|
||||
|
||||
vosk_recognizer_free(recognizer)
|
||||
vosk_model_free(model)
|
||||
}
|
||||
|
||||
return sres
|
||||
}
|
||||
|
||||
|
||||
func recognizeData(buffer : AVAudioPCMBuffer) -> String {
|
||||
let dataLen = Int(buffer.frameLength * 2)
|
||||
let channels = UnsafeBufferPointer(start: buffer.int16ChannelData, count: 1)
|
||||
let endOfSpeech = channels[0].withMemoryRebound(to: Int8.self, capacity: dataLen) {
|
||||
vosk_recognizer_accept_waveform(recognizer, $0, Int32(dataLen))
|
||||
}
|
||||
let res = endOfSpeech == 1 ?vosk_recognizer_result(recognizer) :vosk_recognizer_partial_result(recognizer)
|
||||
return String(validatingUTF8: res!)!;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,37 +12,200 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
/* This header contains the C API for Vosk speech recognition system */
|
||||
|
||||
#ifndef _VOSK_API_H_
|
||||
#define _VOSK_API_H_
|
||||
#ifndef VOSK_API_H
|
||||
#define VOSK_API_H
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/** Model stores all the data required for recognition
|
||||
* it contains static data and can be shared across processing
|
||||
* threads. */
|
||||
typedef struct VoskModel VoskModel;
|
||||
|
||||
|
||||
/** Speaker model is the same as model but contains the data
|
||||
* for speaker identification. */
|
||||
typedef struct VoskSpkModel VoskSpkModel;
|
||||
|
||||
|
||||
/** Recognizer object is the main object which processes data.
|
||||
* Each recognizer usually runs in own thread and takes audio as input.
|
||||
* Once audio is processed recognizer returns JSON object as a string
|
||||
* which represent decoded information - words, confidences, times, n-best lists,
|
||||
* speaker information and so on */
|
||||
typedef struct VoskRecognizer VoskRecognizer;
|
||||
|
||||
|
||||
/** Loads model data from the file and returns the model object
|
||||
*
|
||||
* @param model_path: the path of the model on the filesystem
|
||||
@ @returns model object */
|
||||
VoskModel *vosk_model_new(const char *model_path);
|
||||
|
||||
|
||||
/** Releases the model memory
|
||||
*
|
||||
* The model object is reference-counted so if some recognizer
|
||||
* depends on this model, model might still stay alive. When
|
||||
* last recognizer is released, model will be released too. */
|
||||
void vosk_model_free(VoskModel *model);
|
||||
|
||||
|
||||
/** Loads speaker model data from the file and returns the model object
|
||||
*
|
||||
* @param model_path: the path of the model on the filesystem
|
||||
* @returns model object */
|
||||
VoskSpkModel *vosk_spk_model_new(const char *model_path);
|
||||
|
||||
|
||||
/** Releases the model memory
|
||||
*
|
||||
* The model object is reference-counted so if some recognizer
|
||||
* depends on this model, model might still stay alive. When
|
||||
* last recognizer is released, model will be released too. */
|
||||
void vosk_spk_model_free(VoskSpkModel *model);
|
||||
|
||||
/** Creates the recognizer object
|
||||
*
|
||||
* The recognizers process the speech and return text using shared model data
|
||||
* @param sample_rate The sample rate of the audio you going to feed into the recognizer
|
||||
* @returns recognizer object */
|
||||
VoskRecognizer *vosk_recognizer_new(VoskModel *model, float sample_rate);
|
||||
|
||||
|
||||
/** Creates the recognizer object with speaker recognition
|
||||
*
|
||||
* With the speaker recognition mode the recognizer not just recognize
|
||||
* text but also return speaker vectors one can use for speaker identification
|
||||
*
|
||||
* @param spk_model speaker model for speaker identification
|
||||
* @param sample_rate The sample rate of the audio you going to feed into the recognizer
|
||||
* @returns recognizer object */
|
||||
VoskRecognizer *vosk_recognizer_new_spk(VoskModel *model, VoskSpkModel *spk_model, float sample_rate);
|
||||
|
||||
|
||||
/** Creates the recognizer object with the phrase list
|
||||
*
|
||||
* Sometimes when you want to improve recognition accuracy and when you don't need
|
||||
* to recognize large vocabulary you can specify a list of phrases to recognize. This
|
||||
* will improve recognizer speed and accuracy but might return [unk] if user said
|
||||
* something different.
|
||||
*
|
||||
* Only recognizers with lookahead models support this type of quick configuration.
|
||||
* Precompiled HCLG graph models are not supported.
|
||||
*
|
||||
* @param sample_rate The sample rate of the audio you going to feed into the recognizer
|
||||
* @param grammar The string with the list of phrases to recognize as JSON array of strings,
|
||||
* for example "["one two three four five", "[unk]"]".
|
||||
*
|
||||
* @returns recognizer object */
|
||||
VoskRecognizer *vosk_recognizer_new_grm(VoskModel *model, float sample_rate, const char *grammar);
|
||||
|
||||
|
||||
/** Accept voice data
|
||||
*
|
||||
* accept and process new chunk of voice data
|
||||
*
|
||||
* @param data - audio data in PCM 16-bit mono format
|
||||
* @param length - length of the audio data
|
||||
* @returns true if silence is occured and you can retrieve a new utterance with result method */
|
||||
int vosk_recognizer_accept_waveform(VoskRecognizer *recognizer, const char *data, int length);
|
||||
|
||||
|
||||
/** Same as above but the version with the short data for language bindings where you have
|
||||
* audio as array of shorts */
|
||||
int vosk_recognizer_accept_waveform_s(VoskRecognizer *recognizer, const short *data, int length);
|
||||
|
||||
|
||||
/** Same as above but the version with the float data for language bindings where you have
|
||||
* audio as array of floats */
|
||||
int vosk_recognizer_accept_waveform_f(VoskRecognizer *recognizer, const float *data, int length);
|
||||
|
||||
|
||||
/** Returns speech recognition result
|
||||
*
|
||||
* @returns the result in JSON format which contains decoded line, decoded
|
||||
* words, times in seconds and confidences. You can parse this result
|
||||
* with any json parser
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "result" : [{
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.110000,
|
||||
* "start" : 0.870000,
|
||||
* "word" : "what"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.530000,
|
||||
* "start" : 1.110000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.950000,
|
||||
* "start" : 1.530000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 2.340000,
|
||||
* "start" : 1.950000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 2.610000,
|
||||
* "start" : 2.340000,
|
||||
* "word" : "one"
|
||||
* }],
|
||||
* "text" : "what zero zero zero one"
|
||||
* }
|
||||
* </pre>
|
||||
*/
|
||||
const char *vosk_recognizer_result(VoskRecognizer *recognizer);
|
||||
|
||||
|
||||
/** Returns partial speech recognition
|
||||
*
|
||||
* @returns partial speech recognition text which is not yet finalized.
|
||||
* result may change as recognizer process more data.
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "partial" : "cyril one eight zero"
|
||||
* }
|
||||
* </pre>
|
||||
*/
|
||||
const char *vosk_recognizer_partial_result(VoskRecognizer *recognizer);
|
||||
|
||||
|
||||
/** Returns speech recognition result. Same as result, but doesn't wait for silence
|
||||
* You usually call it in the end of the stream to get final bits of audio. It
|
||||
* flushes the feature pipeline, so all remaining audio chunks got processed.
|
||||
*
|
||||
* @returns speech result in JSON format.
|
||||
*/
|
||||
const char *vosk_recognizer_final_result(VoskRecognizer *recognizer);
|
||||
|
||||
|
||||
/** Releases recognizer object
|
||||
*
|
||||
* Underlying model is also unreferenced and if needed released */
|
||||
void vosk_recognizer_free(VoskRecognizer *recognizer);
|
||||
|
||||
/** Set log level for Kaldi messages
|
||||
*
|
||||
* @param log_level the level
|
||||
* 0 - default value to print info and error messages but no debug
|
||||
* less than 0 - don't print info messages
|
||||
* greather than 0 - more verbose mode
|
||||
*/
|
||||
void vosk_set_log_level(int log_level);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif /* _VOSK_API_H_ */
|
||||
#endif /* VOSK_API_H */
|
||||
|
||||
@@ -0,0 +1,36 @@
|
||||
//
|
||||
// Vosk.swift
|
||||
// VoskApiTest
|
||||
//
|
||||
// Created by Niсkolay Shmyrev on 01.03.20.
|
||||
// Copyright © 2020-2021 Alpha Cephei. All rights reserved.
|
||||
//
|
||||
|
||||
import Foundation
|
||||
|
||||
public final class VoskModel {
|
||||
|
||||
var model : OpaquePointer!
|
||||
var spkModel : OpaquePointer!
|
||||
|
||||
init() {
|
||||
|
||||
// Set to -1 to disable logs
|
||||
vosk_set_log_level(0);
|
||||
|
||||
if let resourcePath = Bundle.main.resourcePath {
|
||||
let modelPath = resourcePath + "/vosk-model-small-en-us-0.15"
|
||||
let spkModelPath = resourcePath + "/vosk-model-spk-0.4"
|
||||
|
||||
model = vosk_model_new(modelPath)
|
||||
spkModel = vosk_spk_model_new(spkModelPath)
|
||||
}
|
||||
}
|
||||
|
||||
deinit {
|
||||
vosk_model_free(model)
|
||||
vosk_spk_model_free(spkModel)
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -1,64 +0,0 @@
|
||||
KALDI_ROOT ?= $(HOME)/kaldi
|
||||
CFLAGS := -g -O2 -DPIC -fPIC -Wno-unused-function
|
||||
CPPFLAGS := -I$(JAVA_HOME)/include -I$(JAVA_HOME)/include/linux -I$(KALDI_ROOT)/src -I$(KALDI_ROOT)/tools/openfst/include -I../src
|
||||
|
||||
KALDI_LIBS = \
|
||||
${KALDI_ROOT}/src/online2/kaldi-online2.a \
|
||||
${KALDI_ROOT}/src/decoder/kaldi-decoder.a \
|
||||
${KALDI_ROOT}/src/ivector/kaldi-ivector.a \
|
||||
${KALDI_ROOT}/src/gmm/kaldi-gmm.a \
|
||||
${KALDI_ROOT}/src/nnet3/kaldi-nnet3.a \
|
||||
${KALDI_ROOT}/src/tree/kaldi-tree.a \
|
||||
${KALDI_ROOT}/src/feat/kaldi-feat.a \
|
||||
${KALDI_ROOT}/src/lat/kaldi-lat.a \
|
||||
${KALDI_ROOT}/src/hmm/kaldi-hmm.a \
|
||||
${KALDI_ROOT}/src/transform/kaldi-transform.a \
|
||||
${KALDI_ROOT}/src/cudamatrix/kaldi-cudamatrix.a \
|
||||
${KALDI_ROOT}/src/matrix/kaldi-matrix.a \
|
||||
${KALDI_ROOT}/src/fstext/kaldi-fstext.a \
|
||||
${KALDI_ROOT}/src/util/kaldi-util.a \
|
||||
${KALDI_ROOT}/src/base/kaldi-base.a \
|
||||
${KALDI_ROOT}/tools/openfst/lib/libfst.a \
|
||||
${KALDI_ROOT}/tools/openfst/lib/libfstngram.a \
|
||||
${KALDI_ROOT}/tools/OpenBLAS/libopenblas.a \
|
||||
-lgfortran
|
||||
|
||||
all: libvosk_jni.so
|
||||
|
||||
VOSK_SOURCES = \
|
||||
vosk_wrap.cc \
|
||||
../src/kaldi_recognizer.cc \
|
||||
../src/kaldi_recognizer.h \
|
||||
../src/model.cc \
|
||||
../src/model.h \
|
||||
../src/spk_model.cc \
|
||||
../src/spk_model.h \
|
||||
../src/vosk_api.cc \
|
||||
../src/vosk_api.h
|
||||
|
||||
libvosk_jni.so: $(VOSK_SOURCES)
|
||||
$(CXX) -shared -o $@ $(CPPFLAGS) $(CFLAGS) $(VOSK_SOURCES) $(KALDI_LIBS)
|
||||
|
||||
vosk_wrap.cc: ../src/vosk.i
|
||||
mkdir -p org/kaldi
|
||||
swig -c++ -I../src \
|
||||
-java -package org.kaldi \
|
||||
-outdir org/kaldi -o $@ $<
|
||||
|
||||
clean:
|
||||
$(RM) *.so *_wrap.cc *_wrap.o test/*.class
|
||||
$(RM) -r org model-en
|
||||
|
||||
model-en:
|
||||
wget https://github.com/alphacep/kaldi-android-demo/releases/download/2020-01/alphacep-model-android-en-us-0.3.tar.gz
|
||||
tar xf alphacep-model-android-en-us-0.3.tar.gz && rm alphacep-model-android-en-us-0.3.tar.gz
|
||||
mv alphacep-model-android-en-us-0.3 model-en
|
||||
|
||||
model-spk:
|
||||
wget https://github.com/alphacep/kaldi-android-demo/releases/download/2020-01/alphacep-spk-model-0.3.tar.gz
|
||||
tar xf alphacep-spk-model-0.3.tar.gz && rm alphacep-spk-model-0.3.tar.gz
|
||||
mv alphacep-spk-model-0.3 model-spk
|
||||
|
||||
run: model-en model-spk
|
||||
javac test/*.java org/kaldi/*.java
|
||||
java -Djava.library.path=. -cp . test.DecoderTest
|
||||
+3
-17
@@ -1,19 +1,5 @@
|
||||
Java API sample
|
||||
Java Vosk API using jnr-ffi
|
||||
|
||||
Doesn't work on Windows or Mac yet, help to prepare the packaged jars is welcome.
|
||||
Still needs classes to wrap C code
|
||||
|
||||
For now to try it:
|
||||
|
||||
On Linux you can do
|
||||
|
||||
1. Build recent kaldi
|
||||
1. `git clone https://github.com/alphacep/vosk-api`
|
||||
1. `cd vosk-api/java`
|
||||
1. `export KALDI_ROOT=<KALDI_ROOT>`
|
||||
1. `export JAVA_HOME=<JAVA_HOME>`
|
||||
1. `make`
|
||||
1. `make run`
|
||||
|
||||
For details of the code you can check:
|
||||
|
||||
https://github.com/alphacep/vosk-api/blob/master/java/test/DecoderTest.java
|
||||
Run with simple gradle build. Unpack model and put libvosk library in current folder.
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
apply plugin: "application"
|
||||
|
||||
mainClassName = "test.DecoderTest"
|
||||
applicationDefaultJvmArgs = ['-Djna.library.path=.']
|
||||
|
||||
repositories {
|
||||
mavenCentral()
|
||||
}
|
||||
|
||||
dependencies {
|
||||
implementation group: 'net.java.dev.jna', name: 'jna', version: '4.5.0'
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
package test;
|
||||
|
||||
import java.io.File;
|
||||
import java.io.FileInputStream;
|
||||
import java.io.IOException;
|
||||
import java.net.URL;
|
||||
import java.nio.*;
|
||||
|
||||
import com.sun.jna.Library;
|
||||
import com.sun.jna.Native;
|
||||
import com.sun.jna.Platform;
|
||||
import com.sun.jna.Pointer;
|
||||
|
||||
public class DecoderTest {
|
||||
public interface LibVosk extends Library {
|
||||
static LibVosk INSTANCE = (LibVosk) Native.loadLibrary("vosk", LibVosk.class);
|
||||
|
||||
void vosk_set_log_level(int level);
|
||||
Pointer vosk_model_new(String path);
|
||||
void vosk_model_free(Pointer model);
|
||||
Pointer vosk_recognizer_new(Pointer model, float sample_rate);
|
||||
boolean vosk_recognizer_accept_waveform(Pointer recognizer, byte[] data, int len);
|
||||
String vosk_recognizer_result(Pointer recognizer);
|
||||
String vosk_recognizer_final_result(Pointer recognizer);
|
||||
String vosk_recognizer_partial_result(Pointer recognizer);
|
||||
void vosk_recognizer_free(Pointer recognizer);
|
||||
}
|
||||
|
||||
public static void main(String[] args) throws IOException {
|
||||
LibVosk.INSTANCE.vosk_set_log_level(0);
|
||||
Pointer model = LibVosk.INSTANCE.vosk_model_new("model");
|
||||
|
||||
FileInputStream ais = new FileInputStream(new File("../python/example/test.wav"));
|
||||
Pointer rec = LibVosk.INSTANCE.vosk_recognizer_new(model, 16000.0f);
|
||||
|
||||
int nbytes;
|
||||
byte[] b = new byte[4096];
|
||||
while ((nbytes = ais.read(b)) >= 0) {
|
||||
if (LibVosk.INSTANCE.vosk_recognizer_accept_waveform(rec, b, nbytes)) {
|
||||
System.out.println(LibVosk.INSTANCE.vosk_recognizer_result(rec));
|
||||
} else {
|
||||
System.out.println(LibVosk.INSTANCE.vosk_recognizer_partial_result(rec));
|
||||
}
|
||||
}
|
||||
System.out.println(LibVosk.INSTANCE.vosk_recognizer_final_result(rec));
|
||||
LibVosk.INSTANCE.vosk_recognizer_free(rec);
|
||||
LibVosk.INSTANCE.vosk_model_free(model);
|
||||
}
|
||||
}
|
||||
@@ -1,37 +0,0 @@
|
||||
package test;
|
||||
|
||||
import java.io.File;
|
||||
import java.io.FileInputStream;
|
||||
import java.io.FileOutputStream;
|
||||
import java.io.DataOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.net.URL;
|
||||
import java.nio.*;
|
||||
|
||||
import org.kaldi.KaldiRecognizer;
|
||||
import org.kaldi.Model;
|
||||
import org.kaldi.SpkModel;
|
||||
|
||||
public class DecoderTest {
|
||||
static {
|
||||
System.loadLibrary("vosk_jni");
|
||||
}
|
||||
|
||||
public static void main(String args[]) throws IOException {
|
||||
FileInputStream ais = new FileInputStream(new File("../python/example/test.wav"));
|
||||
Model model = new Model("model-en");
|
||||
SpkModel spkModel = new SpkModel("model-spk");
|
||||
KaldiRecognizer rec = new KaldiRecognizer(model, spkModel, 16000.0f);
|
||||
|
||||
int nbytes;
|
||||
byte[] b = new byte[4096];
|
||||
while ((nbytes = ais.read(b)) >= 0) {
|
||||
if (rec.AcceptWaveform(b)) {
|
||||
System.out.println(rec.Result());
|
||||
} else {
|
||||
System.out.println(rec.PartialResult());
|
||||
}
|
||||
}
|
||||
System.out.println(rec.FinalResult());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,3 @@
|
||||
demo/model
|
||||
demo/model-spk
|
||||
demo/test.wav
|
||||
@@ -0,0 +1,33 @@
|
||||
This is an FFI-NAPI wrapper for the Vosk library.
|
||||
|
||||
## Usage
|
||||
|
||||
It mostly follows Vosk interface, some methods are not yet fully implemented.
|
||||
|
||||
To use it you need to compile libvosk library, see Python module build
|
||||
instructions for details. You can find prebuilt library inside python
|
||||
wheel.
|
||||
|
||||
## About
|
||||
|
||||
Vosk is an offline open source speech recognition toolkit. It enables
|
||||
speech recognition models for 17 languages and dialects - English, Indian
|
||||
English, German, French, Spanish, Portuguese, Chinese, Russian, Turkish,
|
||||
Vietnamese, Italian, Dutch, Catalan, Arabic, Greek, Farsi, Filipino.
|
||||
|
||||
Vosk models are small (50 Mb) but provide continuous large vocabulary
|
||||
transcription, zero-latency response with streaming API, reconfigurable
|
||||
vocabulary and speaker identification.
|
||||
|
||||
Vosk supplies speech recognition for chatbots, smart home appliances,
|
||||
virtual assistants. It can also create subtitles for movies,
|
||||
transcription for lectures and interviews.
|
||||
|
||||
Vosk scales from small devices like Raspberry Pi or Android smartphone to
|
||||
big clusters.
|
||||
|
||||
# Documentation
|
||||
|
||||
For installation instructions, examples and documentation visit [Vosk
|
||||
Website](https://alphacephei.com/vosk). See also our project on
|
||||
[Github](https://github.com/alphacep/vosk-api).
|
||||
@@ -0,0 +1,30 @@
|
||||
var vosk = require('..')
|
||||
|
||||
const fs = require("fs");
|
||||
const { Readable } = require("stream");
|
||||
const wav = require("wav");
|
||||
|
||||
vosk.setLogLevel(0);
|
||||
const model = new vosk.Model("model");
|
||||
const rec = new vosk.Recognizer(model, 16000.0);
|
||||
|
||||
const wfStream = fs.createReadStream("test.wav", {'highWaterMark': 4096});
|
||||
const wfReader = new wav.Reader();
|
||||
|
||||
wfReader.on('format', async ({ audioFormat, sampleRate, channels }) => {
|
||||
if (audioFormat != 1 || channels != 1) {
|
||||
console.error("Audio file must be WAV format mono PCM.");
|
||||
process.exit(1);
|
||||
}
|
||||
for await (const data of new Readable().wrap(wfReader)) {
|
||||
const end_of_speech = rec.acceptWaveform(data);
|
||||
if (end_of_speech) {
|
||||
console.log(rec.result());
|
||||
}
|
||||
}
|
||||
console.log(rec.finalResult(rec));
|
||||
rec.free();
|
||||
model.free();
|
||||
});
|
||||
|
||||
wfStream.pipe(wfReader);
|
||||
+76
-2
@@ -1,3 +1,77 @@
|
||||
exports.printMsg = function() {
|
||||
console.log("This is a message from the Vosk package");
|
||||
'use strict'
|
||||
|
||||
const os = require('os');
|
||||
const path = require('path');
|
||||
const ffi = require('ffi-napi');
|
||||
const ref = require('ref-napi');
|
||||
|
||||
const vosk_model = ref.types.void;
|
||||
const vosk_model_ptr = ref.refType(vosk_model);
|
||||
const vosk_spk_model = ref.types.void;
|
||||
const vosk_spk_model_ptr = ref.refType(vosk_spk_model);
|
||||
const vosk_recognizer = ref.types.void;
|
||||
const vosk_recognizer_ptr = ref.refType(vosk_recognizer);
|
||||
|
||||
var soname;
|
||||
if (os.platform == 'win32') {
|
||||
soname = path.join(__dirname, "lib", "win-x86_64", "libvosk.dll")
|
||||
} else {
|
||||
soname = path.join(__dirname, "lib", "linux-x86_64", "libvosk.so")
|
||||
}
|
||||
|
||||
const libvosk = ffi.Library(soname, {
|
||||
'vosk_set_log_level': [ 'void', [ 'int' ] ],
|
||||
'vosk_model_new': [ vosk_model_ptr, [ 'string' ] ],
|
||||
'vosk_model_free': [ 'void', [ vosk_model_ptr ] ],
|
||||
'vosk_recognizer_new': [ vosk_recognizer_ptr, [ vosk_model_ptr, 'float' ] ],
|
||||
'vosk_recognizer_free': [ 'void', [ vosk_recognizer_ptr ] ],
|
||||
'vosk_recognizer_accept_waveform': [ 'bool', [ vosk_recognizer_ptr, 'pointer', 'int' ] ],
|
||||
'vosk_recognizer_result': ['string', [vosk_recognizer_ptr ] ],
|
||||
'vosk_recognizer_final_result': ['string', [vosk_recognizer_ptr ] ],
|
||||
});
|
||||
|
||||
function setLogLevel(level) {
|
||||
libvosk.vosk_set_log_level(level);
|
||||
}
|
||||
|
||||
function Model(model_path) {
|
||||
|
||||
this.handle = libvosk.vosk_model_new(model_path);
|
||||
|
||||
this.free = function() {
|
||||
libvosk.vosk_model_free(this.handle);
|
||||
}
|
||||
|
||||
this.getHandle = function() {
|
||||
return this.handle;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
function Recognizer(model, sample_rate) {
|
||||
this.handle = libvosk.vosk_recognizer_new(model.getHandle(), sample_rate);
|
||||
|
||||
this.free = function() {
|
||||
libvosk.vosk_recognizer_free(this.handle);
|
||||
}
|
||||
|
||||
this.acceptWaveform = function(data) {
|
||||
return libvosk.vosk_recognizer_accept_waveform(this.handle, data, data.length);
|
||||
}
|
||||
|
||||
this.result = function() {
|
||||
return libvosk.vosk_recognizer_result(this.handle);
|
||||
}
|
||||
|
||||
this.partialResult = function() {
|
||||
return libvosk.vosk_recognizer_partial_result(this.handle);
|
||||
}
|
||||
|
||||
this.finalResult = function() {
|
||||
return libvosk.vosk_recognizer_final_result(this.handle);
|
||||
}
|
||||
}
|
||||
|
||||
exports.setLogLevel = setLogLevel
|
||||
exports.Model = Model
|
||||
exports.Recognizer = Recognizer
|
||||
|
||||
+11
-4
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "vosk",
|
||||
"version": "0.1.0",
|
||||
"description": "Node binding for continuous voice recoginition through pocketsphinx.",
|
||||
"version": "0.3.21",
|
||||
"description": "Node binding for continuous offline voice recoginition with Vosk library.",
|
||||
"repository": {
|
||||
"type": "git",
|
||||
"url": "git://github.com/alphacep/vosk-api.git"
|
||||
@@ -13,6 +13,13 @@
|
||||
"voice"
|
||||
],
|
||||
"author": "Alpha Cephei Inc.",
|
||||
"license": "Apache 2.0",
|
||||
"engines": { "node" : ">= 12.x.x" }
|
||||
"license": "Apache-2.0",
|
||||
"engines": {
|
||||
"node": ">= 12.x.x"
|
||||
},
|
||||
"dependencies": {
|
||||
"ffi-napi": "^3.1.0",
|
||||
"ref-napi": "^3.0.0",
|
||||
"wav": "^1.0.2"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,50 +0,0 @@
|
||||
cmake_minimum_required(VERSION 3.12.0)
|
||||
project(vosk)
|
||||
|
||||
set(TOP_SRCDIR "${CMAKE_SOURCE_DIR}/..")
|
||||
if("x$ENV{WHEEL_FLAGS}" STREQUAL "x")
|
||||
find_package (Python COMPONENTS Interpreter Development)
|
||||
else()
|
||||
# docker case
|
||||
set(Python_INCLUDE_DIRS "")
|
||||
set(TOP_SRCDIR "/io")
|
||||
endif()
|
||||
|
||||
set(KALDI_ROOT "$ENV{KALDI_ROOT}")
|
||||
set(CMAKE_C_FLAGS "${CMAKE_C_FLAGS} -O3 -DFST_NO_DYNAMIC_LINKING")
|
||||
set(CMAKE_C_FLAGS "${CMAKE_C_FLAGS} $ENV{WHEEL_FLAGS}")
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} ${CMAKE_C_FLAGS} -std=c++11")
|
||||
include_directories("${TOP_SRCDIR}/src" "${KALDI_ROOT}/src" "${KALDI_ROOT}/tools/openfst/include" ${Python_INCLUDE_DIRS})
|
||||
|
||||
find_package(SWIG REQUIRED)
|
||||
include(${SWIG_USE_FILE})
|
||||
|
||||
swig_add_library(vosk TYPE SHARED LANGUAGE Python OUTPUT_DIR "${CMAKE_LIBRARY_OUTPUT_DIRECTORY}" OUTFILE_DIR "."
|
||||
SOURCES "${TOP_SRCDIR}/src/kaldi_recognizer.cc"
|
||||
"${TOP_SRCDIR}/src/spk_model.cc"
|
||||
"${TOP_SRCDIR}/src/model.cc"
|
||||
"${TOP_SRCDIR}/src/vosk_api.cc"
|
||||
"${TOP_SRCDIR}/src/vosk.i")
|
||||
|
||||
swig_link_libraries(vosk
|
||||
${KALDI_ROOT}/src/online2/kaldi-online2.a
|
||||
${KALDI_ROOT}/src/decoder/kaldi-decoder.a
|
||||
${KALDI_ROOT}/src/ivector/kaldi-ivector.a
|
||||
${KALDI_ROOT}/src/gmm/kaldi-gmm.a
|
||||
${KALDI_ROOT}/src/nnet3/kaldi-nnet3.a
|
||||
${KALDI_ROOT}/src/tree/kaldi-tree.a
|
||||
${KALDI_ROOT}/src/feat/kaldi-feat.a
|
||||
${KALDI_ROOT}/src/lat/kaldi-lat.a
|
||||
${KALDI_ROOT}/src/hmm/kaldi-hmm.a
|
||||
${KALDI_ROOT}/src/transform/kaldi-transform.a
|
||||
${KALDI_ROOT}/src/cudamatrix/kaldi-cudamatrix.a
|
||||
${KALDI_ROOT}/src/matrix/kaldi-matrix.a
|
||||
${KALDI_ROOT}/src/fstext/kaldi-fstext.a
|
||||
${KALDI_ROOT}/src/util/kaldi-util.a
|
||||
${KALDI_ROOT}/src/base/kaldi-base.a
|
||||
${KALDI_ROOT}/tools/openfst/lib/libfst.a
|
||||
${KALDI_ROOT}/tools/openfst/lib/libfstngram.a
|
||||
${KALDI_ROOT}/tools/OpenBLAS/libopenblas.a
|
||||
-lgfortran -lstdc++)
|
||||
|
||||
set_target_properties(_vosk PROPERTIES LINK_FLAGS_RELEASE -s)
|
||||
+22
-2
@@ -1,3 +1,23 @@
|
||||
Python module for vosk-api
|
||||
This is a Python module for Vosk.
|
||||
|
||||
See for details https://github.com/alphacep/vosk-api
|
||||
Vosk is an offline open source speech recognition toolkit. It enables
|
||||
speech recognition models for 17 languages and dialects - English, Indian
|
||||
English, German, French, Spanish, Portuguese, Chinese, Russian, Turkish,
|
||||
Vietnamese, Italian, Dutch, Catalan, Arabic, Greek, Farsi, Filipino.
|
||||
|
||||
Vosk models are small (50 Mb) but provide continuous large vocabulary
|
||||
transcription, zero-latency response with streaming API, reconfigurable
|
||||
vocabulary and speaker identification.
|
||||
|
||||
Vosk supplies speech recognition for chatbots, smart home appliances,
|
||||
virtual assistants. It can also create subtitles for movies,
|
||||
transcription for lectures and interviews.
|
||||
|
||||
Vosk scales from small devices like Raspberry Pi or Android smartphone to
|
||||
big clusters.
|
||||
|
||||
# Documentation
|
||||
|
||||
For installation instructions, examples and documentation visit [Vosk
|
||||
Website](https://alphacephei.com/vosk). See also our project on
|
||||
[Github](https://github.com/alphacep/vosk-api).
|
||||
|
||||
@@ -1,76 +0,0 @@
|
||||
# From https://github.com/raydouglass/cmake_setuptools
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import shutil
|
||||
import sys
|
||||
from setuptools import Extension
|
||||
from setuptools.command.build_ext import build_ext
|
||||
from setuptools.command.build_py import build_py
|
||||
|
||||
CMAKE_EXE = os.environ.get('CMAKE_EXE', shutil.which('cmake'))
|
||||
|
||||
|
||||
def check_for_cmake():
|
||||
if not CMAKE_EXE:
|
||||
print('cmake executable not found. '
|
||||
'Set CMAKE_EXE environment or update your path')
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
class CMakeExtension(Extension):
|
||||
"""
|
||||
setuptools.Extension for cmake
|
||||
"""
|
||||
|
||||
def __init__(self, name, pkg_name, sourcedir=''):
|
||||
check_for_cmake()
|
||||
Extension.__init__(self, name, sources=[])
|
||||
self.sourcedir = os.path.abspath(sourcedir)
|
||||
self.pkg_name = pkg_name
|
||||
|
||||
|
||||
class CMakeBuildExt(build_ext):
|
||||
"""
|
||||
setuptools build_exit which builds using cmake & make
|
||||
You can add cmake args with the CMAKE_COMMON_VARIABLES environment variable
|
||||
"""
|
||||
|
||||
def build_extension(self, ext):
|
||||
check_for_cmake()
|
||||
if isinstance(ext, CMakeExtension):
|
||||
output_dir = os.path.abspath(
|
||||
os.path.dirname(self.get_ext_fullpath(ext.pkg_name + "/" + ext.name)))
|
||||
|
||||
build_type = 'Debug' if self.debug else 'Release'
|
||||
cmake_args = [CMAKE_EXE,
|
||||
ext.sourcedir,
|
||||
'-Wno-dev',
|
||||
'-DCMAKE_LIBRARY_OUTPUT_DIRECTORY=' + output_dir,
|
||||
'-DCMAKE_BUILD_TYPE=' + build_type]
|
||||
cmake_args.extend(
|
||||
[x for x in
|
||||
os.environ.get('CMAKE_COMMON_VARIABLES', '').split(' ')
|
||||
if x])
|
||||
|
||||
env = os.environ.copy()
|
||||
if not os.path.exists(self.build_temp):
|
||||
os.makedirs(self.build_temp)
|
||||
subprocess.check_call(cmake_args,
|
||||
cwd=self.build_temp,
|
||||
env=env)
|
||||
subprocess.check_call(['make', 'VERBOSE=1', ext.name],
|
||||
cwd=self.build_temp,
|
||||
env=env)
|
||||
print()
|
||||
else:
|
||||
super().build_extension(ext)
|
||||
|
||||
|
||||
|
||||
class CMakeBuildExtFirst(build_py):
|
||||
def run(self):
|
||||
self.run_command("build_ext")
|
||||
return super().run()
|
||||
|
||||
__all__ = ['CMakeBuildExt', 'CMakeExtension', 'CMakeBuildExtFirst']
|
||||
@@ -1,10 +1,10 @@
|
||||
#!/usr/bin/python3
|
||||
#!/usr/bin/env python3
|
||||
|
||||
from vosk import Model, KaldiRecognizer
|
||||
import sys
|
||||
import json
|
||||
|
||||
model = Model("model-en")
|
||||
model = Model("model")
|
||||
rec = KaldiRecognizer(model, 8000)
|
||||
|
||||
res = json.loads(rec.FinalResult())
|
||||
|
||||
Executable
+33
@@ -0,0 +1,33 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
from vosk import Model, KaldiRecognizer, SetLogLevel
|
||||
import sys
|
||||
import os
|
||||
import wave
|
||||
import subprocess
|
||||
|
||||
SetLogLevel(0)
|
||||
|
||||
if not os.path.exists("model"):
|
||||
print ("Please download the model from https://alphacephei.com/vosk/models and unpack as 'model' in the current folder.")
|
||||
exit (1)
|
||||
|
||||
sample_rate=16000
|
||||
model = Model("model")
|
||||
rec = KaldiRecognizer(model, sample_rate)
|
||||
|
||||
process = subprocess.Popen(['ffmpeg', '-loglevel', 'quiet', '-i',
|
||||
sys.argv[1],
|
||||
'-ar', str(sample_rate) , '-ac', '1', '-f', 's16le', '-'],
|
||||
stdout=subprocess.PIPE)
|
||||
|
||||
while True:
|
||||
data = process.stdout.read(4000)
|
||||
if len(data) == 0:
|
||||
break
|
||||
if rec.AcceptWaveform(data):
|
||||
print(rec.Result())
|
||||
else:
|
||||
print(rec.PartialResult())
|
||||
|
||||
print(rec.FinalResult())
|
||||
@@ -1,28 +1,90 @@
|
||||
#!/usr/bin/python3
|
||||
#!/usr/bin/env python3
|
||||
|
||||
from vosk import Model, KaldiRecognizer
|
||||
import argparse
|
||||
import os
|
||||
import queue
|
||||
import sounddevice as sd
|
||||
import vosk
|
||||
import sys
|
||||
|
||||
if not os.path.exists("model-en"):
|
||||
print ("Please download the model from https://github.com/alphacep/kaldi-android-demo/releases and unpack as 'model-en' in the current folder.")
|
||||
exit (1)
|
||||
q = queue.Queue()
|
||||
|
||||
import pyaudio
|
||||
def int_or_str(text):
|
||||
"""Helper function for argument parsing."""
|
||||
try:
|
||||
return int(text)
|
||||
except ValueError:
|
||||
return text
|
||||
|
||||
p = pyaudio.PyAudio()
|
||||
stream = p.open(format=pyaudio.paInt16, channels=1, rate=16000, input=True, frames_per_buffer=8000)
|
||||
stream.start_stream()
|
||||
def callback(indata, frames, time, status):
|
||||
"""This is called (from a separate thread) for each audio block."""
|
||||
if status:
|
||||
print(status, file=sys.stderr)
|
||||
q.put(indata)
|
||||
|
||||
model = Model("model-en")
|
||||
rec = KaldiRecognizer(model, 16000)
|
||||
parser = argparse.ArgumentParser(add_help=False)
|
||||
parser.add_argument(
|
||||
'-l', '--list-devices', action='store_true',
|
||||
help='show list of audio devices and exit')
|
||||
args, remaining = parser.parse_known_args()
|
||||
if args.list_devices:
|
||||
print(sd.query_devices())
|
||||
parser.exit(0)
|
||||
parser = argparse.ArgumentParser(
|
||||
description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
parents=[parser])
|
||||
parser.add_argument(
|
||||
'-f', '--filename', type=str, metavar='FILENAME',
|
||||
help='audio file to store recording to')
|
||||
parser.add_argument(
|
||||
'-m', '--model', type=str, metavar='MODEL_PATH',
|
||||
help='Path to the model')
|
||||
parser.add_argument(
|
||||
'-d', '--device', type=int_or_str,
|
||||
help='input device (numeric ID or substring)')
|
||||
parser.add_argument(
|
||||
'-r', '--samplerate', type=int, help='sampling rate')
|
||||
args = parser.parse_args(remaining)
|
||||
|
||||
while True:
|
||||
data = stream.read(2000)
|
||||
if len(data) == 0:
|
||||
break
|
||||
if rec.AcceptWaveform(data):
|
||||
print(rec.Result())
|
||||
try:
|
||||
if args.model is None:
|
||||
args.model = "model"
|
||||
if not os.path.exists(args.model):
|
||||
print ("Please download a model for your language from https://alphacephei.com/vosk/models")
|
||||
print ("and unpack as 'model' in the current folder.")
|
||||
parser.exit(0)
|
||||
if args.samplerate is None:
|
||||
device_info = sd.query_devices(args.device, 'input')
|
||||
# soundfile expects an int, sounddevice provides a float:
|
||||
args.samplerate = int(device_info['default_samplerate'])
|
||||
|
||||
model = vosk.Model(args.model)
|
||||
|
||||
if args.filename:
|
||||
dump_fn = open(args.filename, "wb")
|
||||
else:
|
||||
print(rec.PartialResult())
|
||||
dump_fn = None
|
||||
|
||||
print(rec.FinalResult())
|
||||
with sd.RawInputStream(samplerate=args.samplerate, blocksize = 8000, device=args.device, dtype='int16',
|
||||
channels=1, callback=callback):
|
||||
print('#' * 80)
|
||||
print('Press Ctrl+C to stop the recording')
|
||||
print('#' * 80)
|
||||
|
||||
rec = vosk.KaldiRecognizer(model, args.samplerate)
|
||||
while True:
|
||||
data = q.get()
|
||||
data = bytes(data)
|
||||
if rec.AcceptWaveform(data):
|
||||
print(rec.Result())
|
||||
else:
|
||||
print(rec.PartialResult())
|
||||
if dump_fn is not None:
|
||||
dump_fn.write(data)
|
||||
|
||||
except KeyboardInterrupt:
|
||||
print('\nDone')
|
||||
parser.exit(0)
|
||||
except Exception as e:
|
||||
parser.exit(type(e).__name__ + ': ' + str(e))
|
||||
|
||||
@@ -1,12 +1,14 @@
|
||||
#!/usr/bin/python3
|
||||
#!/usr/bin/env python3
|
||||
|
||||
from vosk import Model, KaldiRecognizer
|
||||
from vosk import Model, KaldiRecognizer, SetLogLevel
|
||||
import sys
|
||||
import os
|
||||
import wave
|
||||
|
||||
if not os.path.exists("model-en"):
|
||||
print ("Please download the model from https://github.com/alphacep/kaldi-android-demo/releases and unpack as 'model-en' in the current folder.")
|
||||
SetLogLevel(0)
|
||||
|
||||
if not os.path.exists("model"):
|
||||
print ("Please download the model from https://alphacephei.com/vosk/models and unpack as 'model' in the current folder.")
|
||||
exit (1)
|
||||
|
||||
wf = wave.open(sys.argv[1], "rb")
|
||||
@@ -14,11 +16,11 @@ if wf.getnchannels() != 1 or wf.getsampwidth() != 2 or wf.getcomptype() != "NONE
|
||||
print ("Audio file must be WAV format mono PCM.")
|
||||
exit (1)
|
||||
|
||||
model = Model("model-en")
|
||||
model = Model("model")
|
||||
rec = KaldiRecognizer(model, wf.getframerate())
|
||||
|
||||
while True:
|
||||
data = wf.readframes(1000)
|
||||
data = wf.readframes(4000)
|
||||
if len(data) == 0:
|
||||
break
|
||||
if rec.AcceptWaveform(data):
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
#!/usr/bin/python3
|
||||
#!/usr/bin/env python3
|
||||
|
||||
from vosk import Model, KaldiRecognizer, SpkModel
|
||||
import sys
|
||||
@@ -7,15 +7,15 @@ import json
|
||||
import os
|
||||
import numpy as np
|
||||
|
||||
model_path = "model-en"
|
||||
model_path = "model"
|
||||
spk_model_path = "model-spk"
|
||||
|
||||
if not os.path.exists(model_path):
|
||||
print ("Please download the model from https://github.com/alphacep/kaldi-android-demo/releases and unpack as {} in the current folder.".format(model_path))
|
||||
print ("Please download the model from https://alphacephei.com/vosk/models and unpack as {} in the current folder.".format(model_path))
|
||||
exit (1)
|
||||
|
||||
if not os.path.exists(spk_model_path):
|
||||
print ("Please download the speaker model from https://github.com/alphacep/kaldi-android-demo/releases and unpack as {} in the current folder.".format(spk_model_path))
|
||||
print ("Please download the speaker model from https://alphacephei.com/vosk/models and unpack as {} in the current folder.".format(spk_model_path))
|
||||
exit (1)
|
||||
|
||||
wf = wave.open(sys.argv[1], "rb")
|
||||
@@ -30,7 +30,7 @@ rec = KaldiRecognizer(model, spk_model, wf.getframerate())
|
||||
|
||||
# We compare speakers with cosine distance. We can keep one or several fingerprints for the speaker in a database
|
||||
# to distingusih among users.
|
||||
spk_sig = [5.64308, 4.23898, 1.119433, -0.810904, 2.115443, 2.328436, 6.135152, 1.348195, 2.60771, 1.020717, 4.324225, -0.873012, 6.123375, 4.903791, 0.064803, 4.66212, 3.502724, 2.535861, 5.452417, 7.081769, -0.823969, -5.167974, 8.568919, 4.159035, 5.314441, 3.688272, 5.730379, 4.463213, 7.227232, 3.538961, 3.316218, 1.269628, -1.902378, 3.512679, -1.947611, -1.520158, 3.80928, -2.721601, 5.359588, 2.942463, -7.474174, 3.788054, 0.303426, 4.951366, 1.72281, -1.867125, -3.574615, 3.622509, 4.803109, 2.829714, 1.528521, 6.408293, 0.820131, 5.066522, 2.836125, 2.867029, 3.725267, 0.505927, 1.462984, 5.001863, -3.838309, -2.45902, 3.992581, 4.451616, 2.865211, -1.148313, 4.996399, -3.473454, 2.876967, 3.940124, 7.553079, 0.373356, 1.396561, 2.686691, 2.094895, 0.913796, -0.286909, 3.540179, 4.904687, 0.84554, 7.585956, 1.017081, 0.168355, 6.672327, 4.092033, -4.240158, -2.017081, -0.813043, 6.468298, 4.115041, 2.231936, 2.370055, 4.972295, 5.58382, 6.022872, 2.706988, 5.248096, -1.918003, 8.259204, -0.900911, 1.961962, 2.349709, 3.290093, 3.344172, 3.307027, 4.203372, -0.315103, 5.61919, -3.229496, 3.777309, 4.328595, 1.461014, 2.622894, 0.315525, 5.447259, 5.407609, 5.339016, 1.604555, 5.359932, 0.090242, 0.535306, 4.724705, 4.692502, 0.5783, -5.436688, -4.915511, 1.959807, 2.825248]
|
||||
spk_sig = [-1.110417,0.09703002,1.35658,0.7798632,-0.305457,-0.339204,0.6186931,-0.4521213,0.3982236,-0.004530723,0.7651616,0.6500852,-0.6664245,0.1361499,0.1358056,-0.2887807,-0.1280468,-0.8208137,-1.620276,-0.4628615,0.7870904,-0.105754,0.9739769,-0.3258137,-0.7322628,-0.6212429,-0.5531687,-0.7796484,0.7035915,1.056094,-0.4941756,-0.6521456,-0.2238328,-0.003737517,0.2165709,1.200186,-0.7737719,0.492015,1.16058,0.6135428,-0.7183084,0.3153541,0.3458071,-1.418189,-0.9624157,0.4168292,-1.627305,0.2742135,-0.6166027,0.1962581,-0.6406527,0.4372789,-0.4296024,0.4898657,-0.9531326,-0.2945702,0.7879696,-1.517101,-0.9344181,-0.5049928,-0.005040941,-0.4637912,0.8223695,-1.079849,0.8871287,-0.9732434,-0.5548235,1.879138,-1.452064,-0.1975368,1.55047,0.5941782,-0.52897,1.368219,0.6782904,1.202505,-0.9256122,-0.9718158,-0.9570228,-0.5563112,-1.19049,-1.167985,2.606804,-2.261825,0.01340385,0.2526799,-1.125458,-1.575991,-0.363153,0.3270262,1.485984,-1.769565,1.541829,0.7293826,0.1743717,-0.4759418,1.523451,-2.487134,-1.824067,-0.626367,0.7448186,-1.425648,0.3524166,-0.9903384,3.339342,0.4563958,-0.2876643,1.521635,0.9508078,-0.1398541,0.3867955,-0.7550205,0.6568405,0.09419366,-1.583935,1.306094,-0.3501927,0.1794427,-0.3768163,0.9683866,-0.2442541,-1.696921,-1.8056,-0.6803037,-1.842043,0.3069353,0.9070363,-0.486526]
|
||||
|
||||
def cosine_dist(x, y):
|
||||
nx = np.array(x)
|
||||
@@ -38,15 +38,21 @@ def cosine_dist(x, y):
|
||||
return 1 - np.dot(nx, ny) / np.linalg.norm(nx) / np.linalg.norm(ny)
|
||||
|
||||
while True:
|
||||
data = wf.readframes(1000)
|
||||
data = wf.readframes(4000)
|
||||
if len(data) == 0:
|
||||
break
|
||||
if rec.AcceptWaveform(data):
|
||||
res = json.loads(rec.Result())
|
||||
print ("Text:", res['text'])
|
||||
print ("Speaker distance:", cosine_dist(spk_sig, res['spk']))
|
||||
if 'spk' in res:
|
||||
print ("X-vector:", res['spk'])
|
||||
print ("Speaker distance:", cosine_dist(spk_sig, res['spk']), "based on", res['spk_frames'], "frames")
|
||||
|
||||
print ("Note that second distance is not very reliable because utterance is too short. Utterances longer than 4 seconds give better xvector")
|
||||
|
||||
res = json.loads(rec.FinalResult())
|
||||
print ("Text:", res['text'])
|
||||
print ("Speaker distance:", cosine_dist(spk_sig, res['spk']))
|
||||
if 'spk' in res:
|
||||
print ("X-vector:", res['spk'])
|
||||
print ("Speaker distance:", cosine_dist(spk_sig, res['spk']), "based on", res['spk_frames'], "frames")
|
||||
|
||||
|
||||
Executable
+55
@@ -0,0 +1,55 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
from vosk import Model, KaldiRecognizer, SetLogLevel
|
||||
import sys
|
||||
import os
|
||||
import wave
|
||||
import subprocess
|
||||
import srt
|
||||
import json
|
||||
import datetime
|
||||
|
||||
SetLogLevel(-1)
|
||||
|
||||
if not os.path.exists("model"):
|
||||
print ("Please download the model from https://alphacephei.com/vosk/models and unpack as 'model' in the current folder.")
|
||||
exit (1)
|
||||
|
||||
sample_rate=16000
|
||||
model = Model("model")
|
||||
rec = KaldiRecognizer(model, sample_rate)
|
||||
|
||||
process = subprocess.Popen(['ffmpeg', '-loglevel', 'quiet', '-i',
|
||||
sys.argv[1],
|
||||
'-ar', str(sample_rate) , '-ac', '1', '-f', 's16le', '-'],
|
||||
stdout=subprocess.PIPE)
|
||||
|
||||
|
||||
WORDS_PER_LINE = 7
|
||||
|
||||
def transcribe():
|
||||
results = []
|
||||
subs = []
|
||||
while True:
|
||||
data = process.stdout.read(4000)
|
||||
if len(data) == 0:
|
||||
break
|
||||
if rec.AcceptWaveform(data):
|
||||
results.append(rec.Result())
|
||||
results.append(rec.FinalResult())
|
||||
|
||||
for i, res in enumerate(results):
|
||||
jres = json.loads(res)
|
||||
if not 'result' in jres:
|
||||
continue
|
||||
words = jres['result']
|
||||
for j in range(0, len(words), WORDS_PER_LINE):
|
||||
line = words[j : j + WORDS_PER_LINE]
|
||||
s = srt.Subtitle(index=len(subs),
|
||||
content=" ".join([l['word'] for l in line]),
|
||||
start=datetime.timedelta(seconds=line[0]['start']),
|
||||
end=datetime.timedelta(seconds=line[-1]['end']))
|
||||
subs.append(s)
|
||||
return subs
|
||||
|
||||
print (srt.compose(transcribe()))
|
||||
@@ -1,16 +1,16 @@
|
||||
#!/usr/bin/python3
|
||||
#!/usr/bin/env python3
|
||||
|
||||
from vosk import Model, KaldiRecognizer
|
||||
import sys
|
||||
import json
|
||||
import os
|
||||
|
||||
if not os.path.exists("model-en"):
|
||||
print ("Please download the model from https://github.com/alphacep/kaldi-android-demo/releases and unpack as 'model-en' in the current folder.")
|
||||
if not os.path.exists("model"):
|
||||
print ("Please download the model from https://alphacephei.com/vosk/models and unpack as 'model' in the current folder.")
|
||||
exit (1)
|
||||
|
||||
|
||||
model = Model("model-en")
|
||||
model = Model("model")
|
||||
|
||||
# Large vocabulary free form recognition
|
||||
rec = KaldiRecognizer(model, 16000)
|
||||
@@ -22,7 +22,7 @@ wf = open(sys.argv[1], "rb")
|
||||
wf.read(44) # skip header
|
||||
|
||||
while True:
|
||||
data = wf.read(2000)
|
||||
data = wf.read(4000)
|
||||
if len(data) == 0:
|
||||
break
|
||||
if rec.AcceptWaveform(data):
|
||||
|
||||
@@ -1,12 +1,12 @@
|
||||
#!/usr/bin/python3
|
||||
#!/usr/bin/env python3
|
||||
|
||||
from vosk import Model, KaldiRecognizer
|
||||
import sys
|
||||
import os
|
||||
import wave
|
||||
|
||||
if not os.path.exists("model-en"):
|
||||
print ("Please download the model from https://github.com/alphacep/kaldi-android-demo/releases and unpack as 'model-en' in the current folder.")
|
||||
if not os.path.exists("model"):
|
||||
print ("Please download the model from https://alphacephei.com/vosk/models and unpack as 'model' in the current folder.")
|
||||
exit (1)
|
||||
|
||||
wf = wave.open(sys.argv[1], "rb")
|
||||
@@ -14,12 +14,13 @@ if wf.getnchannels() != 1 or wf.getsampwidth() != 2 or wf.getcomptype() != "NONE
|
||||
print ("Audio file must be WAV format mono PCM.")
|
||||
exit (1)
|
||||
|
||||
model = Model("model-en")
|
||||
# You can also specify the possible word list
|
||||
rec = KaldiRecognizer(model, wf.getframerate(), "zero oh one two three four five six seven eight nine")
|
||||
model = Model("model")
|
||||
|
||||
# You can also specify the possible word or phrase list as JSON list, the order doesn't have to be strict
|
||||
rec = KaldiRecognizer(model, wf.getframerate(), '["oh one two three four five six seven eight nine zero", "[unk]"]')
|
||||
|
||||
while True:
|
||||
data = wf.readframes(1000)
|
||||
data = wf.readframes(4000)
|
||||
if len(data) == 0:
|
||||
break
|
||||
if rec.AcceptWaveform(data):
|
||||
|
||||
+49
-7
@@ -1,22 +1,59 @@
|
||||
import os
|
||||
import sys
|
||||
import setuptools
|
||||
from cmake import *
|
||||
import shutil
|
||||
import glob
|
||||
import platform
|
||||
|
||||
# Figure out environment for cross-compile
|
||||
vosk_source = os.getenv("VOSK_SOURCE", os.path.abspath(os.path.join(os.path.dirname(__file__), "..")))
|
||||
system = os.environ.get('VOSK_PLATFORM', platform.system())
|
||||
architecture = os.environ.get('VOSK_ARCHITECTURE', platform.architecture()[0])
|
||||
|
||||
# Copy precompmilled libraries
|
||||
for lib in glob.glob(os.path.join(vosk_source, "src/lib*.*")):
|
||||
print ("Adding library", lib)
|
||||
shutil.copy(lib, "vosk")
|
||||
|
||||
# Create OS-dependent, but Python-independent wheels.
|
||||
try:
|
||||
from wheel.bdist_wheel import bdist_wheel
|
||||
except ImportError:
|
||||
cmdclass = {}
|
||||
else:
|
||||
class bdist_wheel_tag_name(bdist_wheel):
|
||||
def get_tag(self):
|
||||
abi = 'none'
|
||||
if system == 'Darwin':
|
||||
oses = 'macosx_10_6_x86_64'
|
||||
elif system == 'Windows' and architecture == '32bit':
|
||||
oses = 'win32'
|
||||
elif system == 'Windows' and architecture == '64bit':
|
||||
oses = 'win_amd64'
|
||||
elif system == 'Linux' and architecture == '64bit':
|
||||
oses = 'linux_x86_64'
|
||||
elif system == 'Linux':
|
||||
oses = 'linux_' + architecture
|
||||
else:
|
||||
raise TypeError("Unknown build environment")
|
||||
return 'py3', abi, oses
|
||||
cmdclass = {'bdist_wheel': bdist_wheel_tag_name}
|
||||
|
||||
with open("README.md", "r") as fh:
|
||||
long_description = fh.read()
|
||||
|
||||
setuptools.setup(
|
||||
name="vosk", # Replace with your own username
|
||||
version="0.3.4",
|
||||
name="vosk",
|
||||
version="0.3.21",
|
||||
author="Alpha Cephei Inc",
|
||||
author_email="contact@alphacephei.com",
|
||||
description="API for Kaldi and Vosk",
|
||||
description="Offline open source speech recognition API based on Kaldi and Vosk",
|
||||
long_description=long_description,
|
||||
long_description_content_type="text/markdown",
|
||||
url="https://github.com/alphacep/vosk-api",
|
||||
packages=setuptools.find_packages(),
|
||||
ext_modules=[CMakeExtension('_vosk', 'vosk')],
|
||||
cmdclass={'build_ext': CMakeBuildExt, 'build_py' : CMakeBuildExtFirst},
|
||||
package_data = {'vosk': ['*.so', '*.dll', '*.dyld']},
|
||||
include_package_data=True,
|
||||
classifiers=[
|
||||
'Programming Language :: Python :: 3',
|
||||
'License :: OSI Approved :: Apache Software License',
|
||||
@@ -25,5 +62,10 @@ setuptools.setup(
|
||||
'Operating System :: MacOS :: MacOS X',
|
||||
'Topic :: Software Development :: Libraries :: Python Modules'
|
||||
],
|
||||
python_requires='>=3.4',
|
||||
cmdclass=cmdclass,
|
||||
python_requires='>=3',
|
||||
zip_safe=False, # Since we load so file from the filesystem, we can not run from zip file
|
||||
setup_requires=['cffi>=1.0'],
|
||||
install_requires=['cffi>=1.0'],
|
||||
cffi_modules=['vosk_builder.py:ffibuilder'],
|
||||
)
|
||||
|
||||
Executable
+35
@@ -0,0 +1,35 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
from multiprocessing.dummy import Pool
|
||||
from vosk import Model, KaldiRecognizer
|
||||
|
||||
import sys
|
||||
import os
|
||||
import wave
|
||||
import json
|
||||
|
||||
model = Model("model")
|
||||
|
||||
def recognize(line):
|
||||
uid, fn = line.split()
|
||||
wf = wave.open(fn, "rb")
|
||||
rec = KaldiRecognizer(model, wf.getframerate())
|
||||
|
||||
text = ""
|
||||
while True:
|
||||
data = wf.readframes(1000)
|
||||
if len(data) == 0:
|
||||
break
|
||||
if rec.AcceptWaveform(data):
|
||||
jres = json.loads(rec.Result())
|
||||
text = text + " " + jres['text']
|
||||
jres = json.loads(rec.FinalResult())
|
||||
text = text + " " + jres['text']
|
||||
return (uid + text)
|
||||
|
||||
def main():
|
||||
p = Pool(8)
|
||||
texts = p.map(recognize, open(sys.argv[1]).readlines())
|
||||
print ("\n".join(texts))
|
||||
|
||||
main()
|
||||
+71
-1
@@ -1 +1,71 @@
|
||||
from .vosk import KaldiRecognizer, Model, SpkModel
|
||||
import os
|
||||
import sys
|
||||
|
||||
from .vosk_cffi import ffi as _ffi
|
||||
|
||||
def open_dll():
|
||||
dlldir = os.path.abspath(os.path.dirname(__file__))
|
||||
if sys.platform == 'win32':
|
||||
# We want to load dependencies too
|
||||
os.environ["PATH"] = dlldir + os.pathsep + os.environ['PATH']
|
||||
if hasattr(os, 'add_dll_directory'):
|
||||
os.add_dll_directory(dlldir)
|
||||
return _ffi.dlopen("libvosk.dll")
|
||||
elif sys.platform == 'linux':
|
||||
return _ffi.dlopen(os.path.join(dlldir, "libvosk.so"))
|
||||
elif sys.platform == 'darwin':
|
||||
return _ffi.dlopen(os.path.join(dlldir, "libvosk.dyld"))
|
||||
else:
|
||||
raise TypeError("Unsupported platform")
|
||||
|
||||
_c = open_dll()
|
||||
|
||||
class Model(object):
|
||||
|
||||
def __init__(self, model_path):
|
||||
self._handle = _c.vosk_model_new(model_path.encode('utf-8'))
|
||||
|
||||
def __del__(self):
|
||||
_c.vosk_model_free(self._handle)
|
||||
|
||||
def vosk_model_find_word(self, word):
|
||||
return _c.vosk_model_find_word(self._handle, word.encode('utf-8'))
|
||||
|
||||
class SpkModel(object):
|
||||
|
||||
def __init__(self, model_path):
|
||||
self._handle = _c.vosk_spk_model_new(model_path.encode('utf-8'))
|
||||
|
||||
def __del__(self):
|
||||
_c.vosk_spk_model_free(self._handle)
|
||||
|
||||
class KaldiRecognizer(object):
|
||||
|
||||
def __init__(self, *args):
|
||||
if len(args) == 2:
|
||||
self._handle = _c.vosk_recognizer_new(args[0]._handle, args[1])
|
||||
elif len(args) == 3 and type(args[1]) is SpkModel:
|
||||
self._handle = _c.vosk_recognizer_new_spk(args[0]._handle, args[1]._handle, args[2])
|
||||
elif len(args) == 3 and type(args[2]) is str:
|
||||
self._handle = _c.vosk_recognizer_new_grm(args[0]._handle, args[1], args[2].encode('utf-8'))
|
||||
else:
|
||||
raise TypeError("Unknown arguments")
|
||||
|
||||
def __del__(self):
|
||||
_c.vosk_recognizer_free(self._handle)
|
||||
|
||||
def AcceptWaveform(self, data):
|
||||
return _c.vosk_recognizer_accept_waveform(self._handle, data, len(data))
|
||||
|
||||
def Result(self):
|
||||
return _ffi.string(_c.vosk_recognizer_result(self._handle)).decode('utf-8')
|
||||
|
||||
def PartialResult(self):
|
||||
return _ffi.string(_c.vosk_recognizer_partial_result(self._handle)).decode('utf-8')
|
||||
|
||||
def FinalResult(self):
|
||||
return _ffi.string(_c.vosk_recognizer_final_result(self._handle)).decode('utf-8')
|
||||
|
||||
|
||||
def SetLogLevel(level):
|
||||
return _c.vosk_set_log_level(level)
|
||||
|
||||
@@ -0,0 +1,14 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
import os
|
||||
from cffi import FFI
|
||||
|
||||
vosk_root=os.environ.get("VOSK_SOURCE", "..")
|
||||
cpp_command = "cpp " + vosk_root + "/src/vosk_api.h"
|
||||
|
||||
ffibuilder = FFI()
|
||||
ffibuilder.set_source("vosk.vosk_cffi", None)
|
||||
ffibuilder.cdef(os.popen(cpp_command).read())
|
||||
|
||||
if __name__ == '__main__':
|
||||
ffibuilder.compile(verbose=True)
|
||||
@@ -0,0 +1,47 @@
|
||||
KALDI_ROOT?=$(HOME)/travis/kaldi
|
||||
OPENFST_ROOT?=$(KALDI_ROOT)/tools/openfst
|
||||
OPENBLAS_ROOT?=$(KALDI_ROOT)/tools/OpenBLAS/install
|
||||
EXT?=so
|
||||
CXX?=g++
|
||||
VOSK_SOURCES= \
|
||||
kaldi_recognizer.cc \
|
||||
language_model.cc \
|
||||
model.cc \
|
||||
spk_model.cc \
|
||||
vosk_api.cc
|
||||
|
||||
CFLAGS=-g -O2 -std=c++17 -fPIC -DFST_NO_DYNAMIC_LINKING -I. -I$(KALDI_ROOT)/src -I$(OPENFST_ROOT)/include -I$(OPENBLAS_ROOT)/include
|
||||
LIBS= \
|
||||
$(KALDI_ROOT)/src/online2/kaldi-online2.a \
|
||||
$(KALDI_ROOT)/src/decoder/kaldi-decoder.a \
|
||||
$(KALDI_ROOT)/src/ivector/kaldi-ivector.a \
|
||||
$(KALDI_ROOT)/src/gmm/kaldi-gmm.a \
|
||||
$(KALDI_ROOT)/src/nnet3/kaldi-nnet3.a \
|
||||
$(KALDI_ROOT)/src/tree/kaldi-tree.a \
|
||||
$(KALDI_ROOT)/src/feat/kaldi-feat.a \
|
||||
$(KALDI_ROOT)/src/lat/kaldi-lat.a \
|
||||
$(KALDI_ROOT)/src/lm/kaldi-lm.a \
|
||||
$(KALDI_ROOT)/src/hmm/kaldi-hmm.a \
|
||||
$(KALDI_ROOT)/src/transform/kaldi-transform.a \
|
||||
$(KALDI_ROOT)/src/cudamatrix/kaldi-cudamatrix.a \
|
||||
$(KALDI_ROOT)/src/matrix/kaldi-matrix.a \
|
||||
$(KALDI_ROOT)/src/fstext/kaldi-fstext.a \
|
||||
$(KALDI_ROOT)/src/util/kaldi-util.a \
|
||||
$(KALDI_ROOT)/src/base/kaldi-base.a \
|
||||
$(OPENBLAS_ROOT)/lib/libopenblas.a \
|
||||
$(OPENBLAS_ROOT)/lib/liblapack.a \
|
||||
$(OPENBLAS_ROOT)/lib/libblas.a \
|
||||
$(OPENBLAS_ROOT)/lib/libf2c.a \
|
||||
$(OPENFST_ROOT)/lib/libfst.a \
|
||||
$(OPENFST_ROOT)/lib/libfstngram.a
|
||||
|
||||
all: libvosk.$(EXT)
|
||||
|
||||
libvosk.$(EXT): $(VOSK_SOURCES:.cc=.o)
|
||||
$(CXX) --shared -s -o $@ $^ $(LIBS) -lm -latomic
|
||||
|
||||
%.o: %.cc
|
||||
$(CXX) $(CFLAGS) -c -o $@ $<
|
||||
|
||||
clean:
|
||||
rm -f *.o *.so *.dll
|
||||
+27
-27
@@ -44,7 +44,7 @@ namespace {
|
||||
case '\t': output += "\\t"; break;
|
||||
default : output += str[i]; break;
|
||||
}
|
||||
return std::move( output );
|
||||
return output;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -291,10 +291,10 @@ class JSON
|
||||
/// Functions for getting primitives from the JSON object.
|
||||
bool IsNull() const { return Type == Class::Null; }
|
||||
|
||||
string ToString() const { bool b; return std::move( ToString( b ) ); }
|
||||
string ToString() const { bool b; return ToString(b); }
|
||||
string ToString( bool &ok ) const {
|
||||
ok = (Type == Class::String);
|
||||
return ok ? std::move( json_escape( *Internal.String ) ): string("");
|
||||
return ok ? json_escape( *Internal.String ) : string("");
|
||||
}
|
||||
|
||||
double ToFloat() const { bool b; return ToFloat( b ); }
|
||||
@@ -425,18 +425,18 @@ class JSON
|
||||
};
|
||||
|
||||
JSON Array() {
|
||||
return std::move( JSON::Make( JSON::Class::Array ) );
|
||||
return JSON::Make( JSON::Class::Array );
|
||||
}
|
||||
|
||||
template <typename... T>
|
||||
JSON Array( T... args ) {
|
||||
JSON arr = JSON::Make( JSON::Class::Array );
|
||||
arr.append( args... );
|
||||
return std::move( arr );
|
||||
return arr;
|
||||
}
|
||||
|
||||
JSON Object() {
|
||||
return std::move( JSON::Make( JSON::Class::Object ) );
|
||||
return JSON::Make( JSON::Class::Object );
|
||||
}
|
||||
|
||||
std::ostream& operator<<( std::ostream &os, const JSON &json ) {
|
||||
@@ -457,7 +457,7 @@ namespace {
|
||||
++offset;
|
||||
consume_ws( str, offset );
|
||||
if( str[offset] == '}' ) {
|
||||
++offset; return std::move( Object );
|
||||
++offset; return Object;
|
||||
}
|
||||
|
||||
while( true ) {
|
||||
@@ -484,7 +484,7 @@ namespace {
|
||||
}
|
||||
}
|
||||
|
||||
return std::move( Object );
|
||||
return Object;
|
||||
}
|
||||
|
||||
JSON parse_array( const string &str, size_t &offset ) {
|
||||
@@ -494,7 +494,7 @@ namespace {
|
||||
++offset;
|
||||
consume_ws( str, offset );
|
||||
if( str[offset] == ']' ) {
|
||||
++offset; return std::move( Array );
|
||||
++offset; return Array;
|
||||
}
|
||||
|
||||
while( true ) {
|
||||
@@ -509,11 +509,11 @@ namespace {
|
||||
}
|
||||
else {
|
||||
std::cerr << "ERROR: Array: Expected ',' or ']', found '" << str[offset] << "'\n";
|
||||
return std::move( JSON::Make( JSON::Class::Array ) );
|
||||
return JSON::Make( JSON::Class::Array );
|
||||
}
|
||||
}
|
||||
|
||||
return std::move( Array );
|
||||
return Array;
|
||||
}
|
||||
|
||||
JSON parse_string( const string &str, size_t &offset ) {
|
||||
@@ -538,7 +538,7 @@ namespace {
|
||||
val += c;
|
||||
else {
|
||||
std::cerr << "ERROR: String: Expected hex character in unicode escape, found '" << c << "'\n";
|
||||
return std::move( JSON::Make( JSON::Class::String ) );
|
||||
return JSON::Make( JSON::Class::String );
|
||||
}
|
||||
}
|
||||
offset += 4;
|
||||
@@ -551,7 +551,7 @@ namespace {
|
||||
}
|
||||
++offset;
|
||||
String = val;
|
||||
return std::move( String );
|
||||
return String;
|
||||
}
|
||||
|
||||
JSON parse_number( const string &str, size_t &offset ) {
|
||||
@@ -580,7 +580,7 @@ namespace {
|
||||
exp_str += c;
|
||||
else if( !isspace( c ) && c != ',' && c != ']' && c != '}' ) {
|
||||
std::cerr << "ERROR: Number: Expected a number for exponent, found '" << c << "'\n";
|
||||
return std::move( JSON::Make( JSON::Class::Null ) );
|
||||
return JSON::Make( JSON::Class::Null );
|
||||
}
|
||||
else
|
||||
break;
|
||||
@@ -589,7 +589,7 @@ namespace {
|
||||
}
|
||||
else if( !isspace( c ) && c != ',' && c != ']' && c != '}' ) {
|
||||
std::cerr << "ERROR: Number: unexpected character '" << c << "'\n";
|
||||
return std::move( JSON::Make( JSON::Class::Null ) );
|
||||
return JSON::Make( JSON::Class::Null );
|
||||
}
|
||||
--offset;
|
||||
|
||||
@@ -601,7 +601,7 @@ namespace {
|
||||
else
|
||||
Number = std::stol( val );
|
||||
}
|
||||
return std::move( Number );
|
||||
return Number;
|
||||
}
|
||||
|
||||
JSON parse_bool( const string &str, size_t &offset ) {
|
||||
@@ -612,20 +612,20 @@ namespace {
|
||||
Bool = false;
|
||||
else {
|
||||
std::cerr << "ERROR: Bool: Expected 'true' or 'false', found '" << str.substr( offset, 5 ) << "'\n";
|
||||
return std::move( JSON::Make( JSON::Class::Null ) );
|
||||
return JSON::Make( JSON::Class::Null );
|
||||
}
|
||||
offset += (Bool.ToBool() ? 4 : 5);
|
||||
return std::move( Bool );
|
||||
return Bool;
|
||||
}
|
||||
|
||||
JSON parse_null( const string &str, size_t &offset ) {
|
||||
JSON Null;
|
||||
if( str.substr( offset, 4 ) != "null" ) {
|
||||
std::cerr << "ERROR: Null: Expected 'null', found '" << str.substr( offset, 4 ) << "'\n";
|
||||
return std::move( JSON::Make( JSON::Class::Null ) );
|
||||
return JSON::Make( JSON::Class::Null );
|
||||
}
|
||||
offset += 4;
|
||||
return std::move( Null );
|
||||
return Null;
|
||||
}
|
||||
|
||||
JSON parse_next( const string &str, size_t &offset ) {
|
||||
@@ -633,14 +633,14 @@ namespace {
|
||||
consume_ws( str, offset );
|
||||
value = str[offset];
|
||||
switch( value ) {
|
||||
case '[' : return std::move( parse_array( str, offset ) );
|
||||
case '{' : return std::move( parse_object( str, offset ) );
|
||||
case '\"': return std::move( parse_string( str, offset ) );
|
||||
case '[' : return parse_array( str, offset );
|
||||
case '{' : return parse_object( str, offset );
|
||||
case '\"': return parse_string( str, offset );
|
||||
case 't' :
|
||||
case 'f' : return std::move( parse_bool( str, offset ) );
|
||||
case 'n' : return std::move( parse_null( str, offset ) );
|
||||
case 'f' : return parse_bool( str, offset );
|
||||
case 'n' : return parse_null( str, offset );
|
||||
default : if( ( value <= '9' && value >= '0' ) || value == '-' )
|
||||
return std::move( parse_number( str, offset ) );
|
||||
return parse_number( str, offset );
|
||||
}
|
||||
std::cerr << "ERROR: Parse: Unknown starting character '" << value << "'\n";
|
||||
return JSON();
|
||||
@@ -649,7 +649,7 @@ namespace {
|
||||
|
||||
JSON JSON::Load( const string &str ) {
|
||||
size_t offset = 0;
|
||||
return std::move( parse_next( str, offset ) );
|
||||
return parse_next( str, offset );
|
||||
}
|
||||
|
||||
} // End Namespace json
|
||||
|
||||
+299
-111
@@ -1,4 +1,4 @@
|
||||
// Copyright 2019 Alpha Cephei Inc.
|
||||
// Copyright 2019-2020 Alpha Cephei Inc.
|
||||
//
|
||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||
// you may not use this file except in compliance with the License.
|
||||
@@ -16,128 +16,214 @@
|
||||
#include "json.h"
|
||||
#include "fstext/fstext-utils.h"
|
||||
#include "lat/sausages.h"
|
||||
#include "language_model.h"
|
||||
|
||||
using namespace fst;
|
||||
using namespace kaldi::nnet3;
|
||||
|
||||
KaldiRecognizer::KaldiRecognizer(Model &model, float sample_frequency) : model_(model), spk_model_(0), sample_frequency_(sample_frequency) {
|
||||
KaldiRecognizer::KaldiRecognizer(Model *model, float sample_frequency) : model_(model), spk_model_(0), sample_frequency_(sample_frequency) {
|
||||
|
||||
feature_pipeline_ = new kaldi::OnlineNnet2FeaturePipeline (model_.feature_info_);
|
||||
silence_weighting_ = new kaldi::OnlineSilenceWeighting(*model_.trans_model_, model_.feature_info_.silence_weighting_config, 3);
|
||||
model_->Ref();
|
||||
|
||||
feature_pipeline_ = new kaldi::OnlineNnet2FeaturePipeline (model_->feature_info_);
|
||||
silence_weighting_ = new kaldi::OnlineSilenceWeighting(*model_->trans_model_, model_->feature_info_.silence_weighting_config, 3);
|
||||
|
||||
g_fst_ = NULL;
|
||||
decode_fst_ = NULL;
|
||||
|
||||
if (!model_.hclg_fst_) {
|
||||
if (model_.hcl_fst_ && model_.g_fst_) {
|
||||
decode_fst_ = LookaheadComposeFst(*model_.hcl_fst_, *model_.g_fst_, model_.disambig_);
|
||||
if (!model_->hclg_fst_) {
|
||||
if (model_->hcl_fst_ && model_->g_fst_) {
|
||||
decode_fst_ = LookaheadComposeFst(*model_->hcl_fst_, *model_->g_fst_, model_->disambig_);
|
||||
} else {
|
||||
KALDI_ERR << "Can't create decoding graph";
|
||||
}
|
||||
}
|
||||
|
||||
decoder_ = new kaldi::SingleUtteranceNnet3Decoder(model_.nnet3_decoding_config_,
|
||||
*model_.trans_model_,
|
||||
*model_.decodable_info_,
|
||||
model_.hclg_fst_ ? *model.hclg_fst_ : *decode_fst_,
|
||||
decoder_ = new kaldi::SingleUtteranceNnet3Decoder(model_->nnet3_decoding_config_,
|
||||
*model_->trans_model_,
|
||||
*model_->decodable_info_,
|
||||
model_->hclg_fst_ ? *model_->hclg_fst_ : *decode_fst_,
|
||||
feature_pipeline_);
|
||||
|
||||
frame_offset_ = 0;
|
||||
input_finalized_ = false;
|
||||
spk_feature_ = NULL;
|
||||
|
||||
InitState();
|
||||
InitRescoring();
|
||||
}
|
||||
|
||||
KaldiRecognizer::KaldiRecognizer(Model &model, float sample_frequency, char const *grammar) : model_(model), spk_model_(0), sample_frequency_(sample_frequency)
|
||||
KaldiRecognizer::KaldiRecognizer(Model *model, float sample_frequency, char const *grammar) : model_(model), spk_model_(0), sample_frequency_(sample_frequency)
|
||||
{
|
||||
feature_pipeline_ = new kaldi::OnlineNnet2FeaturePipeline (model_.feature_info_);
|
||||
silence_weighting_ = new kaldi::OnlineSilenceWeighting(*model_.trans_model_, model_.feature_info_.silence_weighting_config, 3);
|
||||
model_->Ref();
|
||||
|
||||
g_fst_ = new StdVectorFst();
|
||||
if (model_.hcl_fst_) {
|
||||
g_fst_->AddState();
|
||||
g_fst_->SetStart(0);
|
||||
g_fst_->AddState();
|
||||
g_fst_->SetFinal(1, fst::TropicalWeight::One());
|
||||
g_fst_->AddArc(1, StdArc(0, 0, fst::TropicalWeight::One(), 0));
|
||||
feature_pipeline_ = new kaldi::OnlineNnet2FeaturePipeline (model_->feature_info_);
|
||||
silence_weighting_ = new kaldi::OnlineSilenceWeighting(*model_->trans_model_, model_->feature_info_.silence_weighting_config, 3);
|
||||
|
||||
// Create simple word loop FST
|
||||
std::stringstream ss(grammar);
|
||||
std::string token;
|
||||
if (model_->hcl_fst_) {
|
||||
json::JSON obj;
|
||||
obj = json::JSON::Load(grammar);
|
||||
|
||||
while (std::getline(ss, token, ' ')) {
|
||||
int32 id = model_.word_syms_->Find(token);
|
||||
g_fst_->AddArc(0, StdArc(id, id, fst::TropicalWeight::One(), 1));
|
||||
if (obj.length() <= 0) {
|
||||
KALDI_WARN << "Expecting array of strings, got: '" << grammar << "'";
|
||||
} else {
|
||||
KALDI_LOG << obj;
|
||||
|
||||
LanguageModelOptions opts;
|
||||
|
||||
opts.ngram_order = 2;
|
||||
opts.discount = 0.5;
|
||||
|
||||
LanguageModelEstimator estimator(opts);
|
||||
for (int i = 0; i < obj.length(); i++) {
|
||||
bool ok;
|
||||
string line = obj[i].ToString(ok);
|
||||
if (!ok) {
|
||||
KALDI_ERR << "Expecting array of strings, got: '" << obj << "'";
|
||||
}
|
||||
|
||||
std::vector<int32> sentence;
|
||||
stringstream ss(line);
|
||||
string token;
|
||||
while (getline(ss, token, ' ')) {
|
||||
int32 id = model_->word_syms_->Find(token);
|
||||
if (id == kNoSymbol) {
|
||||
KALDI_WARN << "Ignoring word missing in vocabulary: '" << token << "'";
|
||||
} else {
|
||||
sentence.push_back(id);
|
||||
}
|
||||
}
|
||||
estimator.AddCounts(sentence);
|
||||
}
|
||||
g_fst_ = new StdVectorFst();
|
||||
estimator.Estimate(g_fst_);
|
||||
|
||||
decode_fst_ = LookaheadComposeFst(*model_->hcl_fst_, *g_fst_, model_->disambig_);
|
||||
}
|
||||
ArcSort(g_fst_, ILabelCompare<StdArc>());
|
||||
|
||||
decode_fst_ = LookaheadComposeFst(*model_.hcl_fst_, *g_fst_, model_.disambig_);
|
||||
} else {
|
||||
decode_fst_ = NULL;
|
||||
KALDI_ERR << "Can't create decoding graph";
|
||||
KALDI_WARN << "Runtime graphs are not supported by this model";
|
||||
}
|
||||
|
||||
decoder_ = new kaldi::SingleUtteranceNnet3Decoder(model_.nnet3_decoding_config_,
|
||||
*model_.trans_model_,
|
||||
*model_.decodable_info_,
|
||||
model_.hclg_fst_ ? *model.hclg_fst_ : *decode_fst_,
|
||||
decoder_ = new kaldi::SingleUtteranceNnet3Decoder(model_->nnet3_decoding_config_,
|
||||
*model_->trans_model_,
|
||||
*model_->decodable_info_,
|
||||
model_->hclg_fst_ ? *model_->hclg_fst_ : *decode_fst_,
|
||||
feature_pipeline_);
|
||||
|
||||
frame_offset_ = 0;
|
||||
input_finalized_ = false;
|
||||
spk_feature_ = NULL;
|
||||
|
||||
InitState();
|
||||
InitRescoring();
|
||||
}
|
||||
|
||||
KaldiRecognizer::KaldiRecognizer(Model *model, SpkModel *spk_model, float sample_frequency) : model_(model), spk_model_(spk_model), sample_frequency_(sample_frequency) {
|
||||
|
||||
KaldiRecognizer::KaldiRecognizer(Model &model, SpkModel *spk_model, float sample_frequency) : model_(model), spk_model_(spk_model), sample_frequency_(sample_frequency) {
|
||||
feature_pipeline_ = new kaldi::OnlineNnet2FeaturePipeline (model_.feature_info_);
|
||||
silence_weighting_ = new kaldi::OnlineSilenceWeighting(*model_.trans_model_, model_.feature_info_.silence_weighting_config, 3);
|
||||
model_->Ref();
|
||||
spk_model->Ref();
|
||||
|
||||
feature_pipeline_ = new kaldi::OnlineNnet2FeaturePipeline (model_->feature_info_);
|
||||
silence_weighting_ = new kaldi::OnlineSilenceWeighting(*model_->trans_model_, model_->feature_info_.silence_weighting_config, 3);
|
||||
|
||||
decode_fst_ = NULL;
|
||||
g_fst_ = NULL;
|
||||
|
||||
if (!model_.hclg_fst_) {
|
||||
if (model_.hcl_fst_ && model_.g_fst_) {
|
||||
decode_fst_ = LookaheadComposeFst(*model_.hcl_fst_, *model_.g_fst_, model_.disambig_);
|
||||
if (!model_->hclg_fst_) {
|
||||
if (model_->hcl_fst_ && model_->g_fst_) {
|
||||
decode_fst_ = LookaheadComposeFst(*model_->hcl_fst_, *model_->g_fst_, model_->disambig_);
|
||||
} else {
|
||||
KALDI_ERR << "Can't create decoding graph";
|
||||
}
|
||||
}
|
||||
|
||||
decoder_ = new kaldi::SingleUtteranceNnet3Decoder(model_.nnet3_decoding_config_,
|
||||
*model_.trans_model_,
|
||||
*model_.decodable_info_,
|
||||
model_.hclg_fst_ ? *model.hclg_fst_ : *decode_fst_,
|
||||
decoder_ = new kaldi::SingleUtteranceNnet3Decoder(model_->nnet3_decoding_config_,
|
||||
*model_->trans_model_,
|
||||
*model_->decodable_info_,
|
||||
model_->hclg_fst_ ? *model_->hclg_fst_ : *decode_fst_,
|
||||
feature_pipeline_);
|
||||
|
||||
frame_offset_ = 0;
|
||||
input_finalized_ = false;
|
||||
|
||||
spk_feature_ = new OnlineMfcc(spk_model_->spkvector_mfcc_opts);
|
||||
|
||||
InitState();
|
||||
InitRescoring();
|
||||
}
|
||||
|
||||
KaldiRecognizer::~KaldiRecognizer() {
|
||||
delete decoder_;
|
||||
delete feature_pipeline_;
|
||||
delete silence_weighting_;
|
||||
delete decoder_;
|
||||
delete g_fst_;
|
||||
delete decode_fst_;
|
||||
delete spk_feature_;
|
||||
delete lm_fst_;
|
||||
|
||||
model_->Unref();
|
||||
if (spk_model_)
|
||||
spk_model_->Unref();
|
||||
}
|
||||
|
||||
void KaldiRecognizer::InitState()
|
||||
{
|
||||
frame_offset_ = 0;
|
||||
samples_processed_ = 0;
|
||||
samples_round_start_ = 0;
|
||||
|
||||
state_ = RECOGNIZER_INITIALIZED;
|
||||
}
|
||||
|
||||
void KaldiRecognizer::InitRescoring()
|
||||
{
|
||||
if (model_->std_lm_fst_) {
|
||||
fst::CacheOptions cache_opts(true, 50000);
|
||||
fst::ArcMapFstOptions mapfst_opts(cache_opts);
|
||||
fst::StdToLatticeMapper<kaldi::BaseFloat> mapper;
|
||||
lm_fst_ = new fst::ArcMapFst<fst::StdArc, kaldi::LatticeArc, fst::StdToLatticeMapper<kaldi::BaseFloat> >(*model_->std_lm_fst_, mapper, mapfst_opts);
|
||||
} else {
|
||||
lm_fst_ = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
void KaldiRecognizer::CleanUp()
|
||||
{
|
||||
delete silence_weighting_;
|
||||
silence_weighting_ = new kaldi::OnlineSilenceWeighting(*model_.trans_model_, model_.feature_info_.silence_weighting_config, 3);
|
||||
silence_weighting_ = new kaldi::OnlineSilenceWeighting(*model_->trans_model_, model_->feature_info_.silence_weighting_config, 3);
|
||||
|
||||
frame_offset_ += decoder_->NumFramesDecoded();
|
||||
decoder_->InitDecoding(frame_offset_);
|
||||
if (decoder_)
|
||||
frame_offset_ += decoder_->NumFramesDecoded();
|
||||
|
||||
// Each 10 minutes we drop the pipeline to save frontend memory in continuous processing
|
||||
// here we drop few frames remaining in the feature pipeline but hope it will not
|
||||
// cause a huge accuracy drop since it happens not very frequently.
|
||||
|
||||
// Also restart if we retrieved final result already
|
||||
|
||||
if (decoder_ == NULL || state_ == RECOGNIZER_FINALIZED || frame_offset_ > 20000) {
|
||||
samples_round_start_ += samples_processed_;
|
||||
samples_processed_ = 0;
|
||||
frame_offset_ = 0;
|
||||
|
||||
delete decoder_;
|
||||
delete feature_pipeline_;
|
||||
|
||||
feature_pipeline_ = new kaldi::OnlineNnet2FeaturePipeline (model_->feature_info_);
|
||||
decoder_ = new kaldi::SingleUtteranceNnet3Decoder(model_->nnet3_decoding_config_,
|
||||
*model_->trans_model_,
|
||||
*model_->decodable_info_,
|
||||
model_->hclg_fst_ ? *model_->hclg_fst_ : *decode_fst_,
|
||||
feature_pipeline_);
|
||||
|
||||
if (spk_model_) {
|
||||
delete spk_feature_;
|
||||
spk_feature_ = new OnlineMfcc(spk_model_->spkvector_mfcc_opts);
|
||||
}
|
||||
} else {
|
||||
decoder_->InitDecoding(frame_offset_);
|
||||
}
|
||||
}
|
||||
|
||||
void KaldiRecognizer::UpdateSilenceWeights()
|
||||
{
|
||||
if (silence_weighting_->Active() && feature_pipeline_->NumFramesReady() > 0 &&
|
||||
feature_pipeline_->IvectorFeature() != NULL) {
|
||||
std::vector<std::pair<int32, BaseFloat> > delta_weights;
|
||||
vector<pair<int32, BaseFloat> > delta_weights;
|
||||
silence_weighting_->ComputeCurrentTraceback(decoder_->Decoder());
|
||||
silence_weighting_->GetDeltaWeights(feature_pipeline_->NumFramesReady(),
|
||||
frame_offset_ * 3,
|
||||
@@ -175,27 +261,32 @@ bool KaldiRecognizer::AcceptWaveform(const float *fdata, int len)
|
||||
|
||||
bool KaldiRecognizer::AcceptWaveform(Vector<BaseFloat> &wdata)
|
||||
{
|
||||
if (input_finalized_) {
|
||||
// Cleanup if we finalized previous utterance or the whole feature pipeline
|
||||
if (!(state_ == RECOGNIZER_RUNNING || state_ == RECOGNIZER_INITIALIZED)) {
|
||||
CleanUp();
|
||||
input_finalized_ = false;
|
||||
}
|
||||
state_ = RECOGNIZER_RUNNING;
|
||||
|
||||
feature_pipeline_->AcceptWaveform(sample_frequency_, wdata);
|
||||
UpdateSilenceWeights();
|
||||
decoder_->AdvanceDecoding();
|
||||
int step = static_cast<int>(sample_frequency_ * 0.2);
|
||||
for (int i = 0; i < wdata.Dim(); i+= step) {
|
||||
SubVector<BaseFloat> r = wdata.Range(i, std::min(step, wdata.Dim() - i));
|
||||
feature_pipeline_->AcceptWaveform(sample_frequency_, r);
|
||||
UpdateSilenceWeights();
|
||||
decoder_->AdvanceDecoding();
|
||||
}
|
||||
samples_processed_ += wdata.Dim();
|
||||
|
||||
if (spk_feature_) {
|
||||
spk_feature_->AcceptWaveform(sample_frequency_, wdata);
|
||||
}
|
||||
|
||||
if (decoder_->EndpointDetected(model_.endpoint_config_)) {
|
||||
if (decoder_->EndpointDetected(model_->endpoint_config_)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
// Computes an xvector from a chunk of speech features.
|
||||
static void RunNnetComputation(const MatrixBase<BaseFloat> &features,
|
||||
const nnet3::Nnet &nnet, nnet3::CachingOptimizingCompiler *compiler,
|
||||
@@ -212,7 +303,7 @@ static void RunNnetComputation(const MatrixBase<BaseFloat> &features,
|
||||
output_spec.indexes.resize(1);
|
||||
request.outputs.resize(1);
|
||||
request.outputs[0].Swap(&output_spec);
|
||||
std::shared_ptr<const nnet3::NnetComputation> computation = compiler->Compile(request);
|
||||
shared_ptr<const nnet3::NnetComputation> computation = compiler->Compile(request);
|
||||
nnet3::Nnet *nnet_to_update = NULL; // we're not doing any update.
|
||||
nnet3::NnetComputer computer(nnet3::NnetComputeOptions(), *computation,
|
||||
nnet, nnet_to_update);
|
||||
@@ -225,17 +316,48 @@ static void RunNnetComputation(const MatrixBase<BaseFloat> &features,
|
||||
xvector->CopyFromVec(cu_output.Row(0));
|
||||
}
|
||||
|
||||
#define MIN_SPK_FEATS 50
|
||||
|
||||
void KaldiRecognizer::GetSpkVector(Vector<BaseFloat> &xvector)
|
||||
bool KaldiRecognizer::GetSpkVector(Vector<BaseFloat> &out_xvector, int *num_spk_frames)
|
||||
{
|
||||
vector<int32> nonsilence_frames;
|
||||
if (silence_weighting_->Active() && feature_pipeline_->NumFramesReady() > 0) {
|
||||
silence_weighting_->ComputeCurrentTraceback(decoder_->Decoder(), true);
|
||||
silence_weighting_->GetNonsilenceFrames(feature_pipeline_->NumFramesReady(),
|
||||
frame_offset_ * 3,
|
||||
&nonsilence_frames);
|
||||
}
|
||||
|
||||
int num_frames = spk_feature_->NumFramesReady() - frame_offset_ * 3;
|
||||
Matrix<BaseFloat> mfcc(num_frames, spk_feature_->Dim());
|
||||
|
||||
// Not very efficient, would be nice to have faster search
|
||||
int num_nonsilence_frames = 0;
|
||||
Vector<BaseFloat> feat(spk_feature_->Dim());
|
||||
|
||||
for (int i = 0; i < num_frames; ++i) {
|
||||
Vector<BaseFloat> feat(spk_feature_->Dim());
|
||||
if (std::find(nonsilence_frames.begin(),
|
||||
nonsilence_frames.end(), i / 3) == nonsilence_frames.end()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
spk_feature_->GetFrame(i + frame_offset_ * 3, &feat);
|
||||
mfcc.CopyRowFromVec(feat, i);
|
||||
mfcc.CopyRowFromVec(feat, num_nonsilence_frames);
|
||||
num_nonsilence_frames++;
|
||||
}
|
||||
|
||||
*num_spk_frames = num_nonsilence_frames;
|
||||
|
||||
// Don't extract vector if not enough data
|
||||
if (num_nonsilence_frames < MIN_SPK_FEATS) {
|
||||
return false;
|
||||
}
|
||||
|
||||
mfcc.Resize(num_nonsilence_frames, spk_feature_->Dim(), kCopyData);
|
||||
|
||||
SlidingWindowCmnOptions cmvn_opts;
|
||||
cmvn_opts.center = true;
|
||||
cmvn_opts.cmn_window = 300;
|
||||
Matrix<BaseFloat> features(mfcc.NumRows(), mfcc.NumCols(), kUndefined);
|
||||
SlidingWindowCmn(cmvn_opts, mfcc, &features);
|
||||
|
||||
@@ -243,112 +365,178 @@ void KaldiRecognizer::GetSpkVector(Vector<BaseFloat> &xvector)
|
||||
nnet3::CachingOptimizingCompilerOptions compiler_config;
|
||||
nnet3::CachingOptimizingCompiler compiler(spk_model_->speaker_nnet, opts.optimize_config, compiler_config);
|
||||
|
||||
Vector<BaseFloat> xvector;
|
||||
RunNnetComputation(features, spk_model_->speaker_nnet, &compiler, &xvector);
|
||||
|
||||
// Whiten the vector with global mean and transform and normalize mean
|
||||
xvector.AddVec(-1.0, spk_model_->mean);
|
||||
|
||||
out_xvector.Resize(spk_model_->transform.NumRows(), kSetZero);
|
||||
out_xvector.AddMatVec(1.0, spk_model_->transform, kNoTrans, xvector, 0.0);
|
||||
|
||||
BaseFloat norm = out_xvector.Norm(2.0);
|
||||
BaseFloat ratio = norm / sqrt(out_xvector.Dim()); // how much larger it is
|
||||
// than it would be, in
|
||||
// expectation, if normally
|
||||
out_xvector.Scale(1.0 / ratio);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
const char* KaldiRecognizer::Result()
|
||||
const char* KaldiRecognizer::GetResult()
|
||||
{
|
||||
|
||||
if (!input_finalized_) {
|
||||
decoder_->FinalizeDecoding();
|
||||
input_finalized_ = true;
|
||||
}
|
||||
|
||||
if (decoder_->NumFramesDecoded() == 0) {
|
||||
last_result_ = "{\"text\": \"\"}";
|
||||
return last_result_.c_str();
|
||||
return StoreReturn("{\"text\": \"\"}");
|
||||
}
|
||||
|
||||
kaldi::CompactLattice clat;
|
||||
decoder_->GetLattice(true, &clat);
|
||||
fst::ScaleLattice(fst::LatticeScale(8.0, 10.0), &clat);
|
||||
|
||||
if (model_->std_lm_fst_) {
|
||||
Lattice lat1;
|
||||
|
||||
ConvertLattice(clat, &lat1);
|
||||
fst::ScaleLattice(fst::GraphLatticeScale(-1.0), &lat1);
|
||||
fst::ArcSort(&lat1, fst::OLabelCompare<kaldi::LatticeArc>());
|
||||
kaldi::Lattice composed_lat;
|
||||
fst::Compose(lat1, *lm_fst_, &composed_lat);
|
||||
fst::Invert(&composed_lat);
|
||||
kaldi::CompactLattice determinized_lat;
|
||||
DeterminizeLattice(composed_lat, &determinized_lat);
|
||||
fst::ScaleLattice(fst::GraphLatticeScale(-1), &determinized_lat);
|
||||
fst::ArcSort(&determinized_lat, fst::OLabelCompare<kaldi::CompactLatticeArc>());
|
||||
|
||||
kaldi::ConstArpaLmDeterministicFst const_arpa_fst(model_->const_arpa_);
|
||||
kaldi::CompactLattice composed_clat;
|
||||
kaldi::ComposeCompactLatticeDeterministic(determinized_lat, &const_arpa_fst, &composed_clat);
|
||||
kaldi::Lattice composed_lat1;
|
||||
ConvertLattice(composed_clat, &composed_lat1);
|
||||
fst::Invert(&composed_lat1);
|
||||
DeterminizeLattice(composed_lat1, &clat);
|
||||
}
|
||||
|
||||
fst::ScaleLattice(fst::GraphLatticeScale(0.9), &clat); // Apply rescoring weight
|
||||
CompactLattice aligned_lat;
|
||||
if (model_.winfo_) {
|
||||
WordAlignLattice(clat, *model_.trans_model_, *model_.winfo_, 0, &aligned_lat);
|
||||
if (model_->winfo_) {
|
||||
WordAlignLattice(clat, *model_->trans_model_, *model_->winfo_, 0, &aligned_lat);
|
||||
} else {
|
||||
aligned_lat = clat;
|
||||
}
|
||||
|
||||
MinimumBayesRisk mbr(aligned_lat);
|
||||
const std::vector<BaseFloat> &conf = mbr.GetOneBestConfidences();
|
||||
const std::vector<int32> &words = mbr.GetOneBest();
|
||||
const std::vector<std::pair<BaseFloat, BaseFloat> > × =
|
||||
const vector<BaseFloat> &conf = mbr.GetOneBestConfidences();
|
||||
const vector<int32> &words = mbr.GetOneBest();
|
||||
const vector<pair<BaseFloat, BaseFloat> > × =
|
||||
mbr.GetOneBestTimes();
|
||||
|
||||
int size = words.size();
|
||||
|
||||
json::JSON obj;
|
||||
std::stringstream text;
|
||||
stringstream text;
|
||||
|
||||
// Create JSON object
|
||||
for (int i = 0; i < size; i++) {
|
||||
json::JSON word;
|
||||
word["word"] = model_.word_syms_->Find(words[i]);
|
||||
word["start"] = (frame_offset_ + times[i].first) * 0.03;
|
||||
word["end"] = (frame_offset_ + times[i].second) * 0.03;
|
||||
word["word"] = model_->word_syms_->Find(words[i]);
|
||||
word["start"] = samples_round_start_ / sample_frequency_ + (frame_offset_ + times[i].first) * 0.03;
|
||||
word["end"] = samples_round_start_ / sample_frequency_ + (frame_offset_ + times[i].second) * 0.03;
|
||||
word["conf"] = conf[i];
|
||||
obj["result"].append(word);
|
||||
|
||||
if (i) {
|
||||
text << " ";
|
||||
}
|
||||
text << model_.word_syms_->Find(words[i]);
|
||||
text << model_->word_syms_->Find(words[i]);
|
||||
}
|
||||
obj["text"] = text.str();
|
||||
|
||||
if (spk_model_) {
|
||||
Vector<BaseFloat> xvector;
|
||||
GetSpkVector(xvector);
|
||||
for (int i = 0; i < xvector.Dim(); i++) {
|
||||
obj["spk"].append(xvector(i));
|
||||
int num_spk_frames;
|
||||
if (GetSpkVector(xvector, &num_spk_frames)) {
|
||||
for (int i = 0; i < xvector.Dim(); i++) {
|
||||
obj["spk"].append(xvector(i));
|
||||
}
|
||||
obj["spk_frames"] = num_spk_frames;
|
||||
}
|
||||
}
|
||||
|
||||
last_result_ = obj.dump();
|
||||
return last_result_.c_str();
|
||||
return StoreReturn(obj.dump());
|
||||
}
|
||||
|
||||
|
||||
const char* KaldiRecognizer::PartialResult()
|
||||
{
|
||||
if (state_ != RECOGNIZER_RUNNING) {
|
||||
return StoreReturn("{\"text\": \"\"}");
|
||||
}
|
||||
|
||||
json::JSON res;
|
||||
|
||||
if (decoder_->NumFramesDecoded() == 0) {
|
||||
res["partial"] = "";
|
||||
last_result_ = res.dump();
|
||||
return last_result_.c_str();
|
||||
return StoreReturn(res.dump());
|
||||
}
|
||||
|
||||
kaldi::Lattice lat;
|
||||
decoder_->GetBestPath(false, &lat);
|
||||
std::vector<kaldi::int32> alignment, words;
|
||||
vector<kaldi::int32> alignment, words;
|
||||
LatticeWeight weight;
|
||||
GetLinearSymbolSequence(lat, &alignment, &words, &weight);
|
||||
|
||||
std::ostringstream text;
|
||||
ostringstream text;
|
||||
for (size_t i = 0; i < words.size(); i++) {
|
||||
if (i) {
|
||||
text << " ";
|
||||
}
|
||||
text << model_.word_syms_->Find(words[i]);
|
||||
text << model_->word_syms_->Find(words[i]);
|
||||
}
|
||||
res["partial"] = text.str();
|
||||
|
||||
last_result_ = res.dump();
|
||||
return last_result_.c_str();
|
||||
return StoreReturn(res.dump());
|
||||
}
|
||||
|
||||
const char* KaldiRecognizer::Result()
|
||||
{
|
||||
if (state_ != RECOGNIZER_RUNNING) {
|
||||
return StoreReturn("{\"text\": \"\"}");
|
||||
}
|
||||
decoder_->FinalizeDecoding();
|
||||
state_ = RECOGNIZER_ENDPOINT;
|
||||
return GetResult();
|
||||
}
|
||||
|
||||
const char* KaldiRecognizer::FinalResult()
|
||||
{
|
||||
if (!input_finalized_) {
|
||||
feature_pipeline_->InputFinished();
|
||||
UpdateSilenceWeights();
|
||||
decoder_->AdvanceDecoding();
|
||||
decoder_->FinalizeDecoding();
|
||||
input_finalized_ = true;
|
||||
return Result();
|
||||
} else {
|
||||
last_result_ = "{\"text\": \"\"}";
|
||||
return last_result_.c_str();
|
||||
if (state_ != RECOGNIZER_RUNNING) {
|
||||
return StoreReturn("{\"text\": \"\"}");
|
||||
}
|
||||
|
||||
feature_pipeline_->InputFinished();
|
||||
UpdateSilenceWeights();
|
||||
decoder_->AdvanceDecoding();
|
||||
decoder_->FinalizeDecoding();
|
||||
state_ = RECOGNIZER_FINALIZED;
|
||||
GetResult();
|
||||
|
||||
// Free some memory while we are finalized, next
|
||||
// iteration will reinitialize them anyway
|
||||
delete decoder_;
|
||||
delete feature_pipeline_;
|
||||
delete silence_weighting_;
|
||||
delete spk_feature_;
|
||||
|
||||
feature_pipeline_ = NULL;
|
||||
silence_weighting_ = NULL;
|
||||
decoder_ = NULL;
|
||||
spk_feature_ = NULL;
|
||||
|
||||
return last_result_.c_str();
|
||||
}
|
||||
|
||||
// Store result in recognizer and return as const string
|
||||
const char *KaldiRecognizer::StoreReturn(const string &res)
|
||||
{
|
||||
last_result_ = res;
|
||||
return last_result_.c_str();
|
||||
}
|
||||
|
||||
+29
-7
@@ -12,6 +12,9 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
#ifndef VOSK_KALDI_RECOGNIZER_H
|
||||
#define VOSK_KALDI_RECOGNIZER_H
|
||||
|
||||
#include "base/kaldi-common.h"
|
||||
#include "util/common-utils.h"
|
||||
#include "fstext/fstext-lib.h"
|
||||
@@ -29,11 +32,18 @@
|
||||
|
||||
using namespace kaldi;
|
||||
|
||||
enum KaldiRecognizerState {
|
||||
RECOGNIZER_INITIALIZED,
|
||||
RECOGNIZER_RUNNING,
|
||||
RECOGNIZER_ENDPOINT,
|
||||
RECOGNIZER_FINALIZED
|
||||
};
|
||||
|
||||
class KaldiRecognizer {
|
||||
public:
|
||||
KaldiRecognizer(Model &model, float sample_frequency);
|
||||
KaldiRecognizer(Model &model, SpkModel *spk_model, float sample_frequency);
|
||||
KaldiRecognizer(Model &model, float sample_frequency, char const *grammar);
|
||||
KaldiRecognizer(Model *model, float sample_frequency);
|
||||
KaldiRecognizer(Model *model, SpkModel *spk_model, float sample_frequency);
|
||||
KaldiRecognizer(Model *model, float sample_frequency, char const *grammar);
|
||||
~KaldiRecognizer();
|
||||
bool AcceptWaveform(const char *data, int len);
|
||||
bool AcceptWaveform(const short *sdata, int len);
|
||||
@@ -43,12 +53,16 @@ class KaldiRecognizer {
|
||||
const char* PartialResult();
|
||||
|
||||
private:
|
||||
void InitState();
|
||||
void InitRescoring();
|
||||
void CleanUp();
|
||||
void UpdateSilenceWeights();
|
||||
bool AcceptWaveform(Vector<BaseFloat> &wdata);
|
||||
void GetSpkVector(Vector<BaseFloat> &xvector);
|
||||
bool GetSpkVector(Vector<BaseFloat> &out_xvector, int *frames);
|
||||
const char *GetResult();
|
||||
const char *StoreReturn(const string &res);
|
||||
|
||||
Model &model_;
|
||||
Model *model_;
|
||||
SingleUtteranceNnet3Decoder *decoder_;
|
||||
fst::LookaheadFst<fst::StdArc, int32> *decode_fst_;
|
||||
fst::StdVectorFst *g_fst_; // dynamically constructed grammar
|
||||
@@ -58,8 +72,16 @@ class KaldiRecognizer {
|
||||
SpkModel *spk_model_;
|
||||
OnlineBaseFeature *spk_feature_;
|
||||
|
||||
fst::ArcMapFst<fst::StdArc, kaldi::LatticeArc, fst::StdToLatticeMapper<kaldi::BaseFloat> > *lm_fst_;
|
||||
|
||||
float sample_frequency_;
|
||||
int32 frame_offset_;
|
||||
bool input_finalized_;
|
||||
std::string last_result_;
|
||||
|
||||
int64 samples_processed_;
|
||||
int64 samples_round_start_;
|
||||
|
||||
KaldiRecognizerState state_;
|
||||
string last_result_;
|
||||
};
|
||||
|
||||
#endif /* VOSK_KALDI_RECOGNIZER_H */
|
||||
|
||||
@@ -0,0 +1,211 @@
|
||||
// Copyright 2015 Johns Hopkins University (author: Daniel Povey)
|
||||
|
||||
// See ../../COPYING for clarification regarding multiple authors
|
||||
//
|
||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||
// you may not use this file except in compliance with the License.
|
||||
// You may obtain a copy of the License at
|
||||
//
|
||||
// http://www.apache.org/licenses/LICENSE-2.0
|
||||
//
|
||||
// THIS CODE IS PROVIDED *AS IS* BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
|
||||
// KIND, EITHER EXPRESS OR IMPLIED, INCLUDING WITHOUT LIMITATION ANY IMPLIED
|
||||
// WARRANTIES OR CONDITIONS OF TITLE, FITNESS FOR A PARTICULAR PURPOSE,
|
||||
// MERCHANTABLITY OR NON-INFRINGEMENT.
|
||||
// See the Apache 2 License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
// A modified version from chain/language-model.cc for static backoff
|
||||
|
||||
#include <algorithm>
|
||||
#include <numeric>
|
||||
|
||||
#include "language_model.h"
|
||||
|
||||
using namespace kaldi;
|
||||
|
||||
void LanguageModelEstimator::AddCounts(const std::vector<int32> &sentence) {
|
||||
KALDI_ASSERT(opts_.ngram_order >= 2 && "--ngram-order must be >= 2");
|
||||
int32 order = opts_.ngram_order;
|
||||
// 0 is used for left-context at the beginning of the file.. treat it as BOS.
|
||||
std::vector<int32> history(0);
|
||||
std::vector<int32>::const_iterator iter = sentence.begin(),
|
||||
end = sentence.end();
|
||||
for (; iter != end; ++iter) {
|
||||
KALDI_ASSERT(*iter != 0);
|
||||
IncrementCount(history, *iter);
|
||||
history.push_back(*iter);
|
||||
if (history.size() >= order)
|
||||
history.erase(history.begin());
|
||||
}
|
||||
// Probability of end of sentence. This will end up getting ignored later, but
|
||||
// it still makes a difference for probability-normalization reasons.
|
||||
IncrementCount(history, 0);
|
||||
}
|
||||
|
||||
void LanguageModelEstimator::IncrementCount(const std::vector<int32> &history,
|
||||
int32 next_phone) {
|
||||
int32 lm_state_index = FindOrCreateLmStateIndexForHistory(history);
|
||||
if (lm_states_[lm_state_index].tot_count == 0) {
|
||||
num_active_lm_states_++;
|
||||
}
|
||||
lm_states_[lm_state_index].AddCount(next_phone, 1);
|
||||
}
|
||||
|
||||
void LanguageModelEstimator::SetParentCounts() {
|
||||
int32 num_lm_states = lm_states_.size();
|
||||
for (int32 l = 0; l < num_lm_states; l++) {
|
||||
int32 l_iter = lm_states_[l].backoff_lmstate_index;
|
||||
while (l_iter != -1) {
|
||||
lm_states_[l_iter].Add(lm_states_[l]);
|
||||
l_iter = lm_states_[l_iter].backoff_lmstate_index;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int32 LanguageModelEstimator::FindLmStateIndexForHistory(
|
||||
const std::vector<int32> &hist) const {
|
||||
MapType::const_iterator iter = hist_to_lmstate_index_.find(hist);
|
||||
if (iter == hist_to_lmstate_index_.end())
|
||||
return -1;
|
||||
else
|
||||
return iter->second;
|
||||
}
|
||||
|
||||
int32 LanguageModelEstimator::FindNonzeroLmStateIndexForHistory(
|
||||
std::vector<int32> hist) const {
|
||||
while (1) {
|
||||
int32 l = FindLmStateIndexForHistory(hist);
|
||||
if (l == -1 || lm_states_[l].tot_count == 0) {
|
||||
// no such state or state has zero count.
|
||||
if (hist.empty())
|
||||
KALDI_ERR << "Error looking up LM state index for history "
|
||||
<< "(likely code bug)";
|
||||
hist.erase(hist.begin()); // back off.
|
||||
} else {
|
||||
return l;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int32 LanguageModelEstimator::FindOrCreateLmStateIndexForHistory(
|
||||
const std::vector<int32> &hist) {
|
||||
MapType::const_iterator iter = hist_to_lmstate_index_.find(hist);
|
||||
if (iter != hist_to_lmstate_index_.end())
|
||||
return iter->second;
|
||||
int32 ans = lm_states_.size(); // index of next element
|
||||
// next statement relies on default construct of LmState.
|
||||
lm_states_.resize(lm_states_.size() + 1);
|
||||
lm_states_.back().history = hist;
|
||||
hist_to_lmstate_index_[hist] = ans;
|
||||
|
||||
// make sure backoff_lmstate_index is set
|
||||
if (hist.size() > 0) {
|
||||
std::vector<int32> backoff_hist(hist.begin() + 1,
|
||||
hist.end());
|
||||
int32 backoff_lm_state = FindOrCreateLmStateIndexForHistory(backoff_hist);
|
||||
lm_states_[ans].backoff_lmstate_index = backoff_lm_state;
|
||||
}
|
||||
return ans;
|
||||
}
|
||||
|
||||
void LanguageModelEstimator::LmState::AddCount(int32 phone, int32 count) {
|
||||
std::map<int32, int32>::iterator iter = phone_to_count.find(phone);
|
||||
if (iter == phone_to_count.end())
|
||||
phone_to_count[phone] = count;
|
||||
else
|
||||
iter->second += count;
|
||||
tot_count += count;
|
||||
}
|
||||
|
||||
void LanguageModelEstimator::LmState::Add(const LmState &other) {
|
||||
KALDI_ASSERT(&other != this);
|
||||
std::map<int32, int32>::const_iterator iter = other.phone_to_count.begin(),
|
||||
end = other.phone_to_count.end();
|
||||
for (; iter != end; ++iter)
|
||||
AddCount(iter->first, iter->second);
|
||||
}
|
||||
|
||||
int32 LanguageModelEstimator::AssignFstStates() {
|
||||
int32 num_lm_states = lm_states_.size();
|
||||
int32 current_fst_state = 0;
|
||||
for (int32 l = 0; l < num_lm_states; l++) {
|
||||
if (lm_states_[l].tot_count != 0) {
|
||||
lm_states_[l].fst_state = current_fst_state++;
|
||||
}
|
||||
}
|
||||
KALDI_ASSERT(current_fst_state == num_active_lm_states_);
|
||||
return current_fst_state;
|
||||
}
|
||||
|
||||
void LanguageModelEstimator::Estimate(fst::StdVectorFst *fst) {
|
||||
KALDI_LOG << "Estimating language model with ngram-order="
|
||||
<< opts_.ngram_order << ", discount="
|
||||
<< opts_.discount;
|
||||
SetParentCounts();
|
||||
int32 num_fst_states = AssignFstStates();
|
||||
OutputToFst(num_fst_states, fst);
|
||||
}
|
||||
|
||||
int32 LanguageModelEstimator::FindInitialFstState() const {
|
||||
std::vector<int32> history(0);
|
||||
int32 l = FindNonzeroLmStateIndexForHistory(history);
|
||||
KALDI_ASSERT(l != -1 && lm_states_[l].fst_state != -1);
|
||||
return lm_states_[l].fst_state;
|
||||
}
|
||||
|
||||
void LanguageModelEstimator::OutputToFst(
|
||||
int32 num_states,
|
||||
fst::StdVectorFst *fst) const {
|
||||
KALDI_ASSERT(num_states == num_active_lm_states_);
|
||||
fst->DeleteStates();
|
||||
for (int32 i = 0; i < num_states; i++)
|
||||
fst->AddState();
|
||||
fst->SetStart(FindInitialFstState());
|
||||
|
||||
int64 tot_count = 0;
|
||||
double tot_logprob = 0.0;
|
||||
|
||||
int32 num_lm_states = lm_states_.size();
|
||||
// note: not all lm-states end up being 'active'.
|
||||
for (int32 l = 0; l < num_lm_states; l++) {
|
||||
const LmState &lm_state = lm_states_[l];
|
||||
if (lm_state.fst_state == -1) {
|
||||
continue;
|
||||
}
|
||||
int32 state_count = lm_state.tot_count;
|
||||
KALDI_ASSERT(state_count != 0);
|
||||
std::map<int32, int32>::const_iterator
|
||||
iter = lm_state.phone_to_count.begin(),
|
||||
end = lm_state.phone_to_count.end();
|
||||
for (; iter != end; ++iter) {
|
||||
int32 phone = iter->first, count = iter->second;
|
||||
BaseFloat logprob = log(count * opts_.discount / state_count);
|
||||
tot_count += count;
|
||||
tot_logprob += logprob * count;
|
||||
if (phone == 0) { // Go to final state
|
||||
fst->SetFinal(lm_state.fst_state, fst::TropicalWeight(-logprob));
|
||||
} else { // It becomes a transition.
|
||||
std::vector<int32> next_history(lm_state.history);
|
||||
next_history.push_back(phone);
|
||||
int32 dest_lm_state = FindNonzeroLmStateIndexForHistory(next_history),
|
||||
dest_fst_state = lm_states_[dest_lm_state].fst_state;
|
||||
KALDI_ASSERT(dest_fst_state != -1);
|
||||
fst->AddArc(lm_state.fst_state,
|
||||
fst::StdArc(phone, phone, fst::TropicalWeight(-logprob),
|
||||
dest_fst_state));
|
||||
}
|
||||
}
|
||||
if (lm_state.backoff_lmstate_index >= 0) {
|
||||
fst->AddArc(lm_state.fst_state, fst::StdArc(0, 0, fst::TropicalWeight(-log(1 - opts_.discount)), lm_states_[lm_state.backoff_lmstate_index].fst_state));
|
||||
}
|
||||
}
|
||||
fst::Connect(fst);
|
||||
// Make sure that Connect does not delete any states.
|
||||
int32 num_states_connected = fst->NumStates();
|
||||
KALDI_ASSERT(num_states_connected == num_states);
|
||||
// arc-sort. ilabel or olabel doesn't matter, it's an acceptor.
|
||||
fst::ArcSort(fst, fst::ILabelCompare<fst::StdArc>());
|
||||
KALDI_LOG << "Created language model with " << num_states
|
||||
<< " states and " << fst::NumArcs(*fst) << " arcs.";
|
||||
}
|
||||
@@ -0,0 +1,150 @@
|
||||
// Copyright 2015 Johns Hopkins University (Author: Daniel Povey)
|
||||
|
||||
// See ../../COPYING for clarification regarding multiple authors
|
||||
//
|
||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||
// you may not use this file except in compliance with the License.
|
||||
// You may obtain a copy of the License at
|
||||
//
|
||||
// http://www.apache.org/licenses/LICENSE-2.0
|
||||
//
|
||||
// THIS CODE IS PROVIDED *AS IS* BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
|
||||
// KIND, EITHER EXPRESS OR IMPLIED, INCLUDING WITHOUT LIMITATION ANY IMPLIED
|
||||
// WARRANTIES OR CONDITIONS OF TITLE, FITNESS FOR A PARTICULAR PURPOSE,
|
||||
// MERCHANTABLITY OR NON-INFRINGEMENT.
|
||||
// See the Apache 2 License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
|
||||
#ifndef VOSK_LANGUAGE_MODEL_H
|
||||
#define VOSK_LANGUAGE_MODEL_H
|
||||
|
||||
#include <vector>
|
||||
#include <map>
|
||||
|
||||
#include "base/kaldi-common.h"
|
||||
#include "util/common-utils.h"
|
||||
#include "fstext/fstext-lib.h"
|
||||
#include "lat/kaldi-lattice.h"
|
||||
|
||||
using namespace kaldi;
|
||||
|
||||
// Very simply lm construction with absolute discounting
|
||||
|
||||
struct LanguageModelOptions {
|
||||
int32 ngram_order; // you might want to tune this
|
||||
BaseFloat discount; // discount for backoff
|
||||
|
||||
LanguageModelOptions():
|
||||
ngram_order(3),
|
||||
discount(0.5)
|
||||
{ }
|
||||
|
||||
void Register(OptionsItf *opts) {
|
||||
opts->Register("ngram-order", &ngram_order, "n-gram order for the phone "
|
||||
"language model used for the 'denominator model'");
|
||||
opts->Register("discount", &discount, "Discount for backoff");
|
||||
}
|
||||
};
|
||||
|
||||
class LanguageModelEstimator {
|
||||
public:
|
||||
LanguageModelEstimator(LanguageModelOptions &opts): opts_(opts),
|
||||
num_active_lm_states_(0) {
|
||||
KALDI_ASSERT(opts.ngram_order >= 1);
|
||||
}
|
||||
|
||||
// Adds counts for this sentence. Basically does: for each n-gram in the
|
||||
// sentence, count[n-gram] += 1. The only constraint on 'sentence' is that it
|
||||
// should contain no zeros.
|
||||
void AddCounts(const std::vector<int32> &sentence);
|
||||
|
||||
// Estimates the LM and outputs it as an FST. Note: there is
|
||||
// no concept here of backoff arcs.
|
||||
void Estimate(fst::StdVectorFst *fst);
|
||||
|
||||
protected:
|
||||
struct LmState {
|
||||
// the phone history associated with this state (length can vary).
|
||||
std::vector<int32> history;
|
||||
|
||||
// maps from
|
||||
std::map<int32, int32> phone_to_count;
|
||||
|
||||
// total count of this state. As we back off states to lower-order states
|
||||
// (and note that this is a hard backoff where we completely remove un-needed
|
||||
// states) this tot_count may become zero.
|
||||
int32 tot_count;
|
||||
|
||||
// LM-state index of the backoff LM state (if it exists, else -1)...
|
||||
// provided for convenience.
|
||||
int32 backoff_lmstate_index;
|
||||
|
||||
// this is only set after we decide on the FST state numbering (at the end).
|
||||
// If not set, it's -1.
|
||||
int32 fst_state;
|
||||
|
||||
void AddCount(int32 phone, int32 count);
|
||||
|
||||
// Add the contents of another LmState.
|
||||
void Add(const LmState &other);
|
||||
|
||||
LmState(): tot_count(0), backoff_lmstate_index(-1),
|
||||
fst_state(-1) { }
|
||||
LmState(const LmState &other):
|
||||
history(other.history), phone_to_count(other.phone_to_count),
|
||||
tot_count(other.tot_count),
|
||||
backoff_lmstate_index(other.backoff_lmstate_index),
|
||||
fst_state(other.fst_state) { }
|
||||
};
|
||||
|
||||
// maps from history to int32
|
||||
typedef unordered_map<std::vector<int32>, int32, VectorHasher<int32> > MapType;
|
||||
|
||||
LanguageModelOptions opts_;
|
||||
|
||||
MapType hist_to_lmstate_index_;
|
||||
std::vector<LmState> lm_states_; // indexed by lmstate_index, the LmStates.
|
||||
|
||||
// Keeps track of the number of lm states that have nonzero counts.
|
||||
int32 num_active_lm_states_;
|
||||
|
||||
|
||||
// adds the counts for this ngram (called from AddCounts()).
|
||||
inline void IncrementCount(const std::vector<int32> &history,
|
||||
int32 next_phone);
|
||||
|
||||
// sets up tot_count_with_parents in all the lm-states
|
||||
void SetParentCounts();
|
||||
|
||||
// Finds and returns an LM-state index for a history -- or -1 if it doesn't
|
||||
// exist. No backoff is done.
|
||||
int32 FindLmStateIndexForHistory(const std::vector<int32> &hist) const;
|
||||
|
||||
// Finds and returns an LM-state index for a history -- and creates one if
|
||||
// it doesn't exist -- and also creates any backoff states needed, down
|
||||
// to history-length no_prune_ngram_order - 1.
|
||||
int32 FindOrCreateLmStateIndexForHistory(const std::vector<int32> &hist);
|
||||
|
||||
// Finds and returns the most specific LM-state index for a history or
|
||||
// backed-off versions of it, that exists and has nonzero count. Will die if
|
||||
// there is no such history. [e.g. if there is no unigram backoff state,
|
||||
// which generally speaking there won't be.]
|
||||
int32 FindNonzeroLmStateIndexForHistory(std::vector<int32> hist) const;
|
||||
|
||||
// after all backoff has been done, assigns FST state indexes to all states
|
||||
// that exist and have nonzero count. Returns the number of states.
|
||||
int32 AssignFstStates();
|
||||
|
||||
// find the FST index of the initial-state, and returns it.
|
||||
int32 FindInitialFstState() const;
|
||||
|
||||
// Write to an FST
|
||||
void OutputToFst(
|
||||
int32 num_fst_states,
|
||||
fst::StdVectorFst *fst) const;
|
||||
|
||||
};
|
||||
|
||||
#endif
|
||||
|
||||
+215
-65
@@ -14,19 +14,7 @@
|
||||
|
||||
|
||||
//
|
||||
// Possible model layout:
|
||||
//
|
||||
// * Default kaldi model with HCLG.fst
|
||||
//
|
||||
// * Lookahead model with const G.fst
|
||||
//
|
||||
// * Lookahead model with ngram G.fst
|
||||
//
|
||||
// * File disambig_tid.int required only for lookadhead models
|
||||
//
|
||||
// * File word_boundary.int is required if we want to have precise word timing information
|
||||
// otherwise we don't do any word alignment. Optionally lexicon alignment can be done
|
||||
// with corresponding C++ code inside kaldi recognizer.
|
||||
// For details of possible model layout see doc/models.md section model-structure
|
||||
|
||||
#include "model.h"
|
||||
|
||||
@@ -45,24 +33,101 @@ static FstRegisterer<NGramFst<StdArc>> NGramFst_StdArc_registerer;
|
||||
|
||||
#ifdef __ANDROID__
|
||||
#include <android/log.h>
|
||||
static void AndroidLogHandler(const LogMessageEnvelope &env, const char *message)
|
||||
static void KaldiLogHandler(const LogMessageEnvelope &env, const char *message)
|
||||
{
|
||||
__android_log_print(ANDROID_LOG_VERBOSE, "KaldiDemo", message, 1);
|
||||
int priority;
|
||||
if (env.severity > GetVerboseLevel())
|
||||
return;
|
||||
|
||||
if (env.severity > LogMessageEnvelope::kInfo) {
|
||||
priority = ANDROID_LOG_VERBOSE;
|
||||
} else {
|
||||
switch (env.severity) {
|
||||
case LogMessageEnvelope::kInfo:
|
||||
priority = ANDROID_LOG_INFO;
|
||||
break;
|
||||
case LogMessageEnvelope::kWarning:
|
||||
priority = ANDROID_LOG_WARN;
|
||||
break;
|
||||
case LogMessageEnvelope::kAssertFailed:
|
||||
priority = ANDROID_LOG_FATAL;
|
||||
break;
|
||||
case LogMessageEnvelope::kError:
|
||||
default: // If not the ERROR, it still an error!
|
||||
priority = ANDROID_LOG_ERROR;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
std::stringstream full_message;
|
||||
full_message << env.func << "():" << env.file << ':'
|
||||
<< env.line << ") " << message;
|
||||
|
||||
__android_log_print(priority, "VoskAPI", "%s", full_message.str().c_str());
|
||||
}
|
||||
#else
|
||||
static void KaldiLogHandler(const LogMessageEnvelope &env, const char *message)
|
||||
{
|
||||
if (env.severity > GetVerboseLevel())
|
||||
return;
|
||||
|
||||
// Modified default Kaldi logging so we can disable LOG messages.
|
||||
std::stringstream full_message;
|
||||
if (env.severity > LogMessageEnvelope::kInfo) {
|
||||
full_message << "VLOG[" << env.severity << "] (";
|
||||
} else {
|
||||
switch (env.severity) {
|
||||
case LogMessageEnvelope::kInfo:
|
||||
full_message << "LOG (";
|
||||
break;
|
||||
case LogMessageEnvelope::kWarning:
|
||||
full_message << "WARNING (";
|
||||
break;
|
||||
case LogMessageEnvelope::kAssertFailed:
|
||||
full_message << "ASSERTION_FAILED (";
|
||||
break;
|
||||
case LogMessageEnvelope::kError:
|
||||
default: // If not the ERROR, it still an error!
|
||||
full_message << "ERROR (";
|
||||
break;
|
||||
}
|
||||
}
|
||||
// Add other info from the envelope and the message text.
|
||||
full_message << "VoskAPI" << ':'
|
||||
<< env.func << "():" << env.file << ':'
|
||||
<< env.line << ") " << message;
|
||||
|
||||
// Print the complete message to stderr.
|
||||
full_message << "\n";
|
||||
std::cerr << full_message.str();
|
||||
}
|
||||
#endif
|
||||
|
||||
Model::Model(const char *model_path) {
|
||||
Model::Model(const char *model_path) : model_path_str_(model_path) {
|
||||
|
||||
#ifdef __ANDROID__
|
||||
SetLogHandler(AndroidLogHandler);
|
||||
#endif
|
||||
SetLogHandler(KaldiLogHandler);
|
||||
|
||||
const char *usage = "Read the docs";
|
||||
struct stat buffer;
|
||||
string am_path = model_path_str_ + "/am/final.mdl";
|
||||
if (stat(am_path.c_str(), &buffer) == 0) {
|
||||
ConfigureV2();
|
||||
} else {
|
||||
ConfigureV1();
|
||||
}
|
||||
|
||||
ReadDataFiles();
|
||||
|
||||
ref_cnt_ = 1;
|
||||
}
|
||||
|
||||
// Old model layout without model configuration file
|
||||
|
||||
void Model::ConfigureV1()
|
||||
{
|
||||
const char *extra_args[] = {
|
||||
"--min-active=200",
|
||||
"--max-active=3000",
|
||||
"--beam=10.0",
|
||||
"--lattice-beam=2.0",
|
||||
"--max-active=7000",
|
||||
"--beam=13.0",
|
||||
"--lattice-beam=6.0",
|
||||
"--acoustic-scale=1.0",
|
||||
|
||||
"--frame-subsampling-factor=3",
|
||||
@@ -71,55 +136,78 @@ Model::Model(const char *model_path) {
|
||||
"--endpoint.rule2.min-trailing-silence=0.5",
|
||||
"--endpoint.rule3.min-trailing-silence=1.0",
|
||||
"--endpoint.rule4.min-trailing-silence=2.0",
|
||||
};
|
||||
std::string model_path_str(model_path);
|
||||
|
||||
kaldi::ParseOptions po(usage);
|
||||
"--print-args=false",
|
||||
};
|
||||
|
||||
kaldi::ParseOptions po("");
|
||||
nnet3_decoding_config_.Register(&po);
|
||||
endpoint_config_.Register(&po);
|
||||
decodable_opts_.Register(&po);
|
||||
|
||||
std::vector<const char*> args;
|
||||
vector<const char*> args;
|
||||
args.push_back("vosk");
|
||||
args.insert(args.end(), extra_args, extra_args + sizeof(extra_args) / sizeof(extra_args[0]));
|
||||
po.Read(args.size(), args.data());
|
||||
|
||||
nnet3_rxfilename_ = model_path_str_ + "/final.mdl";
|
||||
hclg_fst_rxfilename_ = model_path_str_ + "/HCLG.fst";
|
||||
hcl_fst_rxfilename_ = model_path_str_ + "/HCLr.fst";
|
||||
g_fst_rxfilename_ = model_path_str_ + "/Gr.fst";
|
||||
disambig_rxfilename_ = model_path_str_ + "/disambig_tid.int";
|
||||
word_syms_rxfilename_ = model_path_str_ + "/words.txt";
|
||||
winfo_rxfilename_ = model_path_str_ + "/word_boundary.int";
|
||||
carpa_rxfilename_ = model_path_str_ + "/rescore/G.carpa";
|
||||
std_fst_rxfilename_ = model_path_str_ + "/rescore/G.fst";
|
||||
final_ie_rxfilename_ = model_path_str_ + "/ivector/final.ie";
|
||||
mfcc_conf_rxfilename_ = model_path_str_ + "/mfcc.conf";
|
||||
global_cmvn_stats_rxfilename_ = model_path_str_ + "/global_cmvn.stats";
|
||||
}
|
||||
|
||||
void Model::ConfigureV2()
|
||||
{
|
||||
kaldi::ParseOptions po("something");
|
||||
nnet3_decoding_config_.Register(&po);
|
||||
endpoint_config_.Register(&po);
|
||||
decodable_opts_.Register(&po);
|
||||
po.ReadConfigFile(model_path_str_ + "/conf/model.conf");
|
||||
|
||||
|
||||
nnet3_rxfilename_ = model_path_str_ + "/am/final.mdl";
|
||||
hclg_fst_rxfilename_ = model_path_str_ + "/graph/HCLG.fst";
|
||||
hcl_fst_rxfilename_ = model_path_str_ + "/graph/HCLr.fst";
|
||||
g_fst_rxfilename_ = model_path_str_ + "/graph/Gr.fst";
|
||||
disambig_rxfilename_ = model_path_str_ + "/graph/disambig_tid.int";
|
||||
word_syms_rxfilename_ = model_path_str_ + "/graph/words.txt";
|
||||
winfo_rxfilename_ = model_path_str_ + "/graph/phones/word_boundary.int";
|
||||
carpa_rxfilename_ = model_path_str_ + "/rescore/G.carpa";
|
||||
std_fst_rxfilename_ = model_path_str_ + "/rescore/G.fst";
|
||||
final_ie_rxfilename_ = model_path_str_ + "/ivector/final.ie";
|
||||
mfcc_conf_rxfilename_ = model_path_str_ + "/conf/mfcc.conf";
|
||||
global_cmvn_stats_rxfilename_ = model_path_str_ + "/am/global_cmvn.stats";
|
||||
}
|
||||
|
||||
void Model::ReadDataFiles()
|
||||
{
|
||||
struct stat buffer;
|
||||
|
||||
KALDI_LOG << "Decoding params beam=" << nnet3_decoding_config_.beam <<
|
||||
" max-active=" << nnet3_decoding_config_.max_active <<
|
||||
" lattice-beam=" << nnet3_decoding_config_.lattice_beam;
|
||||
KALDI_LOG << "Silence phones " << endpoint_config_.silence_phones;
|
||||
|
||||
feature_info_.feature_type = "mfcc";
|
||||
ReadConfigFromFile(model_path_str + "/mfcc.conf", &feature_info_.mfcc_opts);
|
||||
ReadConfigFromFile(mfcc_conf_rxfilename_, &feature_info_.mfcc_opts);
|
||||
feature_info_.mfcc_opts.frame_opts.allow_downsample = true; // It is safe to downsample
|
||||
|
||||
feature_info_.silence_weighting_config.silence_weight = 1e-3;
|
||||
feature_info_.silence_weighting_config.silence_phones_str = "1:2:3:4:5:6:7:8:9:10";
|
||||
|
||||
OnlineIvectorExtractionConfig ivector_extraction_opts;
|
||||
ivector_extraction_opts.splice_config_rxfilename = model_path_str + "/ivector/splice.conf";
|
||||
ivector_extraction_opts.cmvn_config_rxfilename = model_path_str + "/ivector/online_cmvn.conf";
|
||||
ivector_extraction_opts.lda_mat_rxfilename = model_path_str + "/ivector/final.mat";
|
||||
ivector_extraction_opts.global_cmvn_stats_rxfilename = model_path_str + "/ivector/global_cmvn.stats";
|
||||
ivector_extraction_opts.diag_ubm_rxfilename = model_path_str + "/ivector/final.dubm";
|
||||
ivector_extraction_opts.ivector_extractor_rxfilename = model_path_str + "/ivector/final.ie";
|
||||
ivector_extraction_opts.num_gselect = 5;
|
||||
ivector_extraction_opts.min_post = 0.025;
|
||||
ivector_extraction_opts.posterior_scale = 0.1;
|
||||
ivector_extraction_opts.max_remembered_frames = 1000;
|
||||
ivector_extraction_opts.max_count = 100;
|
||||
ivector_extraction_opts.ivector_period = 200;
|
||||
feature_info_.use_ivectors = true;
|
||||
feature_info_.ivector_extractor_info.Init(ivector_extraction_opts);
|
||||
|
||||
std::string nnet3_rxfilename = model_path_str + "/final.mdl";
|
||||
std::string hclg_fst_rxfilename = model_path_str + "/HCLG.fst";
|
||||
std::string hcl_fst_rxfilename = model_path_str + "/HCLr.fst";
|
||||
std::string g_fst_rxfilename = model_path_str + "/Gr.fst";
|
||||
std::string disambig_rxfilename = model_path_str + "/disambig_tid.int";
|
||||
std::string word_syms_rxfilename = model_path_str + "/words.txt";
|
||||
std::string winfo_rxfilename = model_path_str + "/word_boundary.int";
|
||||
feature_info_.silence_weighting_config.silence_phones_str = endpoint_config_.silence_phones;
|
||||
|
||||
trans_model_ = new kaldi::TransitionModel();
|
||||
nnet_ = new kaldi::nnet3::AmNnetSimple();
|
||||
{
|
||||
bool binary;
|
||||
kaldi::Input ki(nnet3_rxfilename, &binary);
|
||||
kaldi::Input ki(nnet3_rxfilename_, &binary);
|
||||
trans_model_->Read(ki.Stream(), binary);
|
||||
nnet_->Read(ki.Stream(), binary);
|
||||
SetBatchnormTestMode(true, &(nnet_->GetNnet()));
|
||||
@@ -129,16 +217,42 @@ Model::Model(const char *model_path) {
|
||||
decodable_info_ = new nnet3::DecodableNnetSimpleLoopedInfo(decodable_opts_,
|
||||
nnet_);
|
||||
|
||||
struct stat buffer;
|
||||
if (stat(hclg_fst_rxfilename.c_str(), &buffer) == 0) {
|
||||
hclg_fst_ = fst::ReadFstKaldiGeneric(hclg_fst_rxfilename);
|
||||
if (stat(final_ie_rxfilename_.c_str(), &buffer) == 0) {
|
||||
KALDI_LOG << "Loading i-vector extractor from " << final_ie_rxfilename_;
|
||||
|
||||
OnlineIvectorExtractionConfig ivector_extraction_opts;
|
||||
ivector_extraction_opts.splice_config_rxfilename = model_path_str_ + "/ivector/splice.conf";
|
||||
ivector_extraction_opts.cmvn_config_rxfilename = model_path_str_ + "/ivector/online_cmvn.conf";
|
||||
ivector_extraction_opts.lda_mat_rxfilename = model_path_str_ + "/ivector/final.mat";
|
||||
ivector_extraction_opts.global_cmvn_stats_rxfilename = model_path_str_ + "/ivector/global_cmvn.stats";
|
||||
ivector_extraction_opts.diag_ubm_rxfilename = model_path_str_ + "/ivector/final.dubm";
|
||||
ivector_extraction_opts.ivector_extractor_rxfilename = model_path_str_ + "/ivector/final.ie";
|
||||
ivector_extraction_opts.max_count = 100;
|
||||
|
||||
feature_info_.use_ivectors = true;
|
||||
feature_info_.ivector_extractor_info.Init(ivector_extraction_opts);
|
||||
} else {
|
||||
feature_info_.use_ivectors = false;
|
||||
}
|
||||
|
||||
if (stat(global_cmvn_stats_rxfilename_.c_str(), &buffer) == 0) {
|
||||
KALDI_LOG << "Reading CMVN stats from " << global_cmvn_stats_rxfilename_;
|
||||
feature_info_.use_cmvn = true;
|
||||
ReadKaldiObject(global_cmvn_stats_rxfilename_, &feature_info_.global_cmvn_stats);
|
||||
}
|
||||
|
||||
|
||||
if (stat(hclg_fst_rxfilename_.c_str(), &buffer) == 0) {
|
||||
KALDI_LOG << "Loading HCLG from " << hclg_fst_rxfilename_;
|
||||
hclg_fst_ = fst::ReadFstKaldiGeneric(hclg_fst_rxfilename_);
|
||||
hcl_fst_ = NULL;
|
||||
g_fst_ = NULL;
|
||||
} else {
|
||||
KALDI_LOG << "Loading HCL and G from " << hcl_fst_rxfilename_ << " " << g_fst_rxfilename_;
|
||||
hclg_fst_ = NULL;
|
||||
hcl_fst_ = fst::StdFst::Read(hcl_fst_rxfilename);
|
||||
g_fst_ = fst::StdFst::Read(g_fst_rxfilename);
|
||||
ReadIntegerVectorSimple(disambig_rxfilename, &disambig_);
|
||||
hcl_fst_ = fst::StdFst::Read(hcl_fst_rxfilename_);
|
||||
g_fst_ = fst::StdFst::Read(g_fst_rxfilename_);
|
||||
ReadIntegerVectorSimple(disambig_rxfilename_, &disambig_);
|
||||
}
|
||||
|
||||
word_syms_ = NULL;
|
||||
@@ -148,18 +262,54 @@ Model::Model(const char *model_path) {
|
||||
word_syms_ = g_fst_->OutputSymbols();
|
||||
}
|
||||
if (!word_syms_) {
|
||||
if (!(word_syms_ = fst::SymbolTable::ReadText(word_syms_rxfilename)))
|
||||
KALDI_LOG << "Loading words from " << word_syms_rxfilename_;
|
||||
if (!(word_syms_ = fst::SymbolTable::ReadText(word_syms_rxfilename_)))
|
||||
KALDI_ERR << "Could not read symbol table from file "
|
||||
<< word_syms_rxfilename;
|
||||
<< word_syms_rxfilename_;
|
||||
}
|
||||
KALDI_ASSERT(word_syms_);
|
||||
|
||||
if (stat(winfo_rxfilename.c_str(), &buffer) == 0) {
|
||||
if (stat(winfo_rxfilename_.c_str(), &buffer) == 0) {
|
||||
KALDI_LOG << "Loading winfo " << winfo_rxfilename_;
|
||||
kaldi::WordBoundaryInfoNewOpts opts;
|
||||
winfo_ = new kaldi::WordBoundaryInfo(opts, winfo_rxfilename);
|
||||
winfo_ = new kaldi::WordBoundaryInfo(opts, winfo_rxfilename_);
|
||||
} else {
|
||||
winfo_ = NULL;
|
||||
}
|
||||
|
||||
if (stat(carpa_rxfilename_.c_str(), &buffer) == 0) {
|
||||
KALDI_LOG << "Loading CARPA model from " << carpa_rxfilename_;
|
||||
std_lm_fst_ = fst::ReadFstKaldi(std_fst_rxfilename_);
|
||||
fst::Project(std_lm_fst_, fst::PROJECT_OUTPUT);
|
||||
if (std_lm_fst_->Properties(fst::kILabelSorted, true) == 0) {
|
||||
fst::ILabelCompare<fst::StdArc> ilabel_comp;
|
||||
fst::ArcSort(std_lm_fst_, ilabel_comp);
|
||||
}
|
||||
ReadKaldiObject(carpa_rxfilename_, &const_arpa_);
|
||||
} else {
|
||||
std_lm_fst_ = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
void Model::Ref()
|
||||
{
|
||||
ref_cnt_++;
|
||||
}
|
||||
|
||||
void Model::Unref()
|
||||
{
|
||||
ref_cnt_--;
|
||||
if (ref_cnt_ == 0) {
|
||||
delete this;
|
||||
}
|
||||
}
|
||||
|
||||
int Model::FindWord(const char *word)
|
||||
{
|
||||
if (!word_syms_)
|
||||
return -1;
|
||||
|
||||
return word_syms_->Find(word);
|
||||
}
|
||||
|
||||
Model::~Model() {
|
||||
|
||||
+32
-5
@@ -12,8 +12,8 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
#ifndef MODEL_H_
|
||||
#define MODEL_H_
|
||||
#ifndef VOSK_MODEL_H
|
||||
#define VOSK_MODEL_H
|
||||
|
||||
#include "base/kaldi-common.h"
|
||||
#include "fstext/fstext-lib.h"
|
||||
@@ -32,6 +32,7 @@
|
||||
#include "rnnlm/rnnlm-utils.h"
|
||||
|
||||
using namespace kaldi;
|
||||
using namespace std;
|
||||
|
||||
class KaldiRecognizer;
|
||||
|
||||
@@ -39,11 +40,32 @@ class Model {
|
||||
|
||||
public:
|
||||
Model(const char *model_path);
|
||||
~Model();
|
||||
void Ref();
|
||||
void Unref();
|
||||
int FindWord(const char *word);
|
||||
|
||||
protected:
|
||||
~Model();
|
||||
void ConfigureV1();
|
||||
void ConfigureV2();
|
||||
void ReadDataFiles();
|
||||
|
||||
friend class KaldiRecognizer;
|
||||
|
||||
string model_path_str_;
|
||||
string nnet3_rxfilename_;
|
||||
string hclg_fst_rxfilename_;
|
||||
string hcl_fst_rxfilename_;
|
||||
string g_fst_rxfilename_;
|
||||
string disambig_rxfilename_;
|
||||
string word_syms_rxfilename_;
|
||||
string winfo_rxfilename_;
|
||||
string carpa_rxfilename_;
|
||||
string std_fst_rxfilename_;
|
||||
string final_ie_rxfilename_;
|
||||
string mfcc_conf_rxfilename_;
|
||||
string global_cmvn_stats_rxfilename_;
|
||||
|
||||
kaldi::OnlineEndpointConfig endpoint_config_;
|
||||
kaldi::LatticeFasterDecoderConfig nnet3_decoding_config_;
|
||||
kaldi::nnet3::NnetSimpleLoopedComputationOptions decodable_opts_;
|
||||
@@ -54,11 +76,16 @@ protected:
|
||||
kaldi::nnet3::AmNnetSimple *nnet_;
|
||||
const fst::SymbolTable *word_syms_;
|
||||
kaldi::WordBoundaryInfo *winfo_;
|
||||
std::vector<int32> disambig_;
|
||||
vector<int32> disambig_;
|
||||
|
||||
fst::Fst<fst::StdArc> *hclg_fst_;
|
||||
fst::Fst<fst::StdArc> *hcl_fst_;
|
||||
fst::Fst<fst::StdArc> *g_fst_;
|
||||
|
||||
fst::VectorFst<fst::StdArc> *std_lm_fst_;
|
||||
kaldi::ConstArpaLm const_arpa_;
|
||||
|
||||
int ref_cnt_;
|
||||
};
|
||||
|
||||
#endif /* MODEL_H_ */
|
||||
#endif /* VOSK_MODEL_H */
|
||||
|
||||
@@ -24,4 +24,22 @@ SpkModel::SpkModel(const char *speaker_path) {
|
||||
SetBatchnormTestMode(true, &speaker_nnet);
|
||||
SetDropoutTestMode(true, &speaker_nnet);
|
||||
CollapseModel(nnet3::CollapseModelConfig(), &speaker_nnet);
|
||||
|
||||
ReadKaldiObject(speaker_path_str + "/mean.vec", &mean);
|
||||
ReadKaldiObject(speaker_path_str + "/transform.mat", &transform);
|
||||
|
||||
ref_cnt_ = 1;
|
||||
}
|
||||
|
||||
void SpkModel::Ref()
|
||||
{
|
||||
ref_cnt_++;
|
||||
}
|
||||
|
||||
void SpkModel::Unref()
|
||||
{
|
||||
ref_cnt_--;
|
||||
if (ref_cnt_ == 0) {
|
||||
delete this;
|
||||
}
|
||||
}
|
||||
|
||||
+11
-3
@@ -12,8 +12,8 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
#ifndef SPK_MODEL_H_
|
||||
#define SPK_MODEL_H_
|
||||
#ifndef VOSK_SPK_MODEL_H
|
||||
#define VOSK_SPK_MODEL_H
|
||||
|
||||
#include "base/kaldi-common.h"
|
||||
#include "online2/online-feature-pipeline.h"
|
||||
@@ -27,12 +27,20 @@ class SpkModel {
|
||||
|
||||
public:
|
||||
SpkModel(const char *spk_path);
|
||||
void Ref();
|
||||
void Unref();
|
||||
|
||||
protected:
|
||||
friend class KaldiRecognizer;
|
||||
~SpkModel() {};
|
||||
|
||||
kaldi::nnet3::Nnet speaker_nnet;
|
||||
kaldi::Vector<BaseFloat> mean;
|
||||
kaldi::Matrix<BaseFloat> transform;
|
||||
|
||||
MfccOptions spkvector_mfcc_opts;
|
||||
|
||||
int ref_cnt_;
|
||||
};
|
||||
|
||||
#endif /* SPK_MODEL_H_ */
|
||||
#endif /* VOSK_SPK_MODEL_H */
|
||||
|
||||
+30
-3
@@ -1,4 +1,8 @@
|
||||
%module(package="vosk") vosk
|
||||
#if SWIGPYTHON
|
||||
%module(package="vosk", "threads"=1) vosk
|
||||
#else
|
||||
%module Vosk
|
||||
#endif
|
||||
|
||||
%include <typemaps.i>
|
||||
|
||||
@@ -10,8 +14,6 @@
|
||||
%include <arrays_csharp.i>
|
||||
#endif
|
||||
|
||||
|
||||
|
||||
#if SWIGPYTHON
|
||||
%pybuffer_binary(const char *data, int len);
|
||||
#endif
|
||||
@@ -32,6 +34,11 @@ import java.nio.ByteOrder;
|
||||
return AcceptWaveform(bdata, bdata.length);
|
||||
}
|
||||
%}
|
||||
%pragma(java) jniclasscode=%{
|
||||
static {
|
||||
System.loadLibrary("vosk_jni");
|
||||
}
|
||||
%}
|
||||
#endif
|
||||
|
||||
#if SWIGCSHARP
|
||||
@@ -42,6 +49,14 @@ CSHARP_ARRAYS(char, byte)
|
||||
#endif
|
||||
|
||||
|
||||
#if SWIGJAVASCRIPT
|
||||
%begin %{
|
||||
#include <v8.h>
|
||||
#include <node.h>
|
||||
#include <node_buffer.h>
|
||||
%}
|
||||
#endif
|
||||
|
||||
%{
|
||||
#include "vosk_api.h"
|
||||
typedef struct VoskModel Model;
|
||||
@@ -60,6 +75,9 @@ typedef struct {} KaldiRecognizer;
|
||||
~Model() {
|
||||
vosk_model_free($self);
|
||||
}
|
||||
int vosk_model_find_word(const char* word) {
|
||||
return vosk_model_find_word($self, word);
|
||||
}
|
||||
}
|
||||
|
||||
%extend SpkModel {
|
||||
@@ -99,6 +117,12 @@ typedef struct {} KaldiRecognizer;
|
||||
bool AcceptWaveform(const char *data, int len) {
|
||||
return vosk_recognizer_accept_waveform($self, data, len);
|
||||
}
|
||||
#elif SWIGJAVASCRIPT
|
||||
bool AcceptWaveform(SWIG_Object ptr) {
|
||||
char* data = (char*) node::Buffer::Data(ptr);
|
||||
size_t length = node::Buffer::Length(ptr);
|
||||
return vosk_recognizer_accept_waveform($self, data, length);
|
||||
}
|
||||
#else
|
||||
int AcceptWaveform(const char *data, int len) {
|
||||
return vosk_recognizer_accept_waveform($self, data, len);
|
||||
@@ -115,3 +139,6 @@ typedef struct {} KaldiRecognizer;
|
||||
return vosk_recognizer_final_result($self);
|
||||
}
|
||||
}
|
||||
|
||||
%rename(SetLogLevel) vosk_set_log_level;
|
||||
void vosk_set_log_level(int level);
|
||||
|
||||
+15
-5
@@ -28,7 +28,12 @@ VoskModel *vosk_model_new(const char *model_path)
|
||||
|
||||
void vosk_model_free(VoskModel *model)
|
||||
{
|
||||
delete (Model *)model;
|
||||
((Model *)model)->Unref();
|
||||
}
|
||||
|
||||
int vosk_model_find_word(VoskModel *model, const char *word)
|
||||
{
|
||||
return (int) ((Model *)model)->FindWord(word);
|
||||
}
|
||||
|
||||
VoskSpkModel *vosk_spk_model_new(const char *model_path)
|
||||
@@ -38,22 +43,22 @@ VoskSpkModel *vosk_spk_model_new(const char *model_path)
|
||||
|
||||
void vosk_spk_model_free(VoskSpkModel *model)
|
||||
{
|
||||
delete (SpkModel *)model;
|
||||
((SpkModel *)model)->Unref();
|
||||
}
|
||||
|
||||
VoskRecognizer *vosk_recognizer_new(VoskModel *model, float sample_rate)
|
||||
{
|
||||
return (VoskRecognizer *)new KaldiRecognizer(*(Model *)model, sample_rate);
|
||||
return (VoskRecognizer *)new KaldiRecognizer((Model *)model, sample_rate);
|
||||
}
|
||||
|
||||
VoskRecognizer *vosk_recognizer_new_spk(VoskModel *model, VoskSpkModel *spk_model, float sample_rate)
|
||||
{
|
||||
return (VoskRecognizer *)new KaldiRecognizer(*(Model *)model, (SpkModel *)spk_model, sample_rate);
|
||||
return (VoskRecognizer *)new KaldiRecognizer((Model *)model, (SpkModel *)spk_model, sample_rate);
|
||||
}
|
||||
|
||||
VoskRecognizer *vosk_recognizer_new_grm(VoskModel *model, float sample_rate, const char *grammar)
|
||||
{
|
||||
return (VoskRecognizer *)new KaldiRecognizer(*(Model *)model, sample_rate, grammar);
|
||||
return (VoskRecognizer *)new KaldiRecognizer((Model *)model, sample_rate, grammar);
|
||||
}
|
||||
|
||||
int vosk_recognizer_accept_waveform(VoskRecognizer *recognizer, const char *data, int length)
|
||||
@@ -90,3 +95,8 @@ void vosk_recognizer_free(VoskRecognizer *recognizer)
|
||||
{
|
||||
delete (KaldiRecognizer *)(recognizer);
|
||||
}
|
||||
|
||||
void vosk_set_log_level(int log_level)
|
||||
{
|
||||
SetVerboseLevel(log_level);
|
||||
}
|
||||
|
||||
+174
-3
@@ -12,37 +12,208 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
/* This header contains the C API for Vosk speech recognition system */
|
||||
|
||||
#ifndef _VOSK_API_H_
|
||||
#define _VOSK_API_H_
|
||||
#ifndef VOSK_API_H
|
||||
#define VOSK_API_H
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/** Model stores all the data required for recognition
|
||||
* it contains static data and can be shared across processing
|
||||
* threads. */
|
||||
typedef struct VoskModel VoskModel;
|
||||
|
||||
|
||||
/** Speaker model is the same as model but contains the data
|
||||
* for speaker identification. */
|
||||
typedef struct VoskSpkModel VoskSpkModel;
|
||||
|
||||
|
||||
/** Recognizer object is the main object which processes data.
|
||||
* Each recognizer usually runs in own thread and takes audio as input.
|
||||
* Once audio is processed recognizer returns JSON object as a string
|
||||
* which represent decoded information - words, confidences, times, n-best lists,
|
||||
* speaker information and so on */
|
||||
typedef struct VoskRecognizer VoskRecognizer;
|
||||
|
||||
|
||||
/** Loads model data from the file and returns the model object
|
||||
*
|
||||
* @param model_path: the path of the model on the filesystem
|
||||
@ @returns model object */
|
||||
VoskModel *vosk_model_new(const char *model_path);
|
||||
|
||||
|
||||
/** Releases the model memory
|
||||
*
|
||||
* The model object is reference-counted so if some recognizer
|
||||
* depends on this model, model might still stay alive. When
|
||||
* last recognizer is released, model will be released too. */
|
||||
void vosk_model_free(VoskModel *model);
|
||||
|
||||
|
||||
/** Check if a word can be recognized by the model
|
||||
* @param word: the word
|
||||
* @returns the word symbol if @param word exists inside the model
|
||||
* or -1 otherwise.
|
||||
* Reminding that word symbol 0 is for <epsilon> */
|
||||
int vosk_model_find_word(VoskModel *model, const char *word);
|
||||
|
||||
|
||||
/** Loads speaker model data from the file and returns the model object
|
||||
*
|
||||
* @param model_path: the path of the model on the filesystem
|
||||
* @returns model object */
|
||||
VoskSpkModel *vosk_spk_model_new(const char *model_path);
|
||||
|
||||
|
||||
/** Releases the model memory
|
||||
*
|
||||
* The model object is reference-counted so if some recognizer
|
||||
* depends on this model, model might still stay alive. When
|
||||
* last recognizer is released, model will be released too. */
|
||||
void vosk_spk_model_free(VoskSpkModel *model);
|
||||
|
||||
/** Creates the recognizer object
|
||||
*
|
||||
* The recognizers process the speech and return text using shared model data
|
||||
* @param sample_rate The sample rate of the audio you going to feed into the recognizer
|
||||
* @returns recognizer object */
|
||||
VoskRecognizer *vosk_recognizer_new(VoskModel *model, float sample_rate);
|
||||
|
||||
|
||||
/** Creates the recognizer object with speaker recognition
|
||||
*
|
||||
* With the speaker recognition mode the recognizer not just recognize
|
||||
* text but also return speaker vectors one can use for speaker identification
|
||||
*
|
||||
* @param spk_model speaker model for speaker identification
|
||||
* @param sample_rate The sample rate of the audio you going to feed into the recognizer
|
||||
* @returns recognizer object */
|
||||
VoskRecognizer *vosk_recognizer_new_spk(VoskModel *model, VoskSpkModel *spk_model, float sample_rate);
|
||||
|
||||
|
||||
/** Creates the recognizer object with the phrase list
|
||||
*
|
||||
* Sometimes when you want to improve recognition accuracy and when you don't need
|
||||
* to recognize large vocabulary you can specify a list of phrases to recognize. This
|
||||
* will improve recognizer speed and accuracy but might return [unk] if user said
|
||||
* something different.
|
||||
*
|
||||
* Only recognizers with lookahead models support this type of quick configuration.
|
||||
* Precompiled HCLG graph models are not supported.
|
||||
*
|
||||
* @param sample_rate The sample rate of the audio you going to feed into the recognizer
|
||||
* @param grammar The string with the list of phrases to recognize as JSON array of strings,
|
||||
* for example "["one two three four five", "[unk]"]".
|
||||
*
|
||||
* @returns recognizer object */
|
||||
VoskRecognizer *vosk_recognizer_new_grm(VoskModel *model, float sample_rate, const char *grammar);
|
||||
|
||||
|
||||
/** Accept voice data
|
||||
*
|
||||
* accept and process new chunk of voice data
|
||||
*
|
||||
* @param data - audio data in PCM 16-bit mono format
|
||||
* @param length - length of the audio data
|
||||
* @returns true if silence is occured and you can retrieve a new utterance with result method */
|
||||
int vosk_recognizer_accept_waveform(VoskRecognizer *recognizer, const char *data, int length);
|
||||
|
||||
|
||||
/** Same as above but the version with the short data for language bindings where you have
|
||||
* audio as array of shorts */
|
||||
int vosk_recognizer_accept_waveform_s(VoskRecognizer *recognizer, const short *data, int length);
|
||||
|
||||
|
||||
/** Same as above but the version with the float data for language bindings where you have
|
||||
* audio as array of floats */
|
||||
int vosk_recognizer_accept_waveform_f(VoskRecognizer *recognizer, const float *data, int length);
|
||||
|
||||
|
||||
/** Returns speech recognition result
|
||||
*
|
||||
* @returns the result in JSON format which contains decoded line, decoded
|
||||
* words, times in seconds and confidences. You can parse this result
|
||||
* with any json parser
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "result" : [{
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.110000,
|
||||
* "start" : 0.870000,
|
||||
* "word" : "what"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.530000,
|
||||
* "start" : 1.110000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 1.950000,
|
||||
* "start" : 1.530000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 2.340000,
|
||||
* "start" : 1.950000,
|
||||
* "word" : "zero"
|
||||
* }, {
|
||||
* "conf" : 1.000000,
|
||||
* "end" : 2.610000,
|
||||
* "start" : 2.340000,
|
||||
* "word" : "one"
|
||||
* }],
|
||||
* "text" : "what zero zero zero one"
|
||||
* }
|
||||
* </pre>
|
||||
*/
|
||||
const char *vosk_recognizer_result(VoskRecognizer *recognizer);
|
||||
|
||||
|
||||
/** Returns partial speech recognition
|
||||
*
|
||||
* @returns partial speech recognition text which is not yet finalized.
|
||||
* result may change as recognizer process more data.
|
||||
*
|
||||
* <pre>
|
||||
* {
|
||||
* "partial" : "cyril one eight zero"
|
||||
* }
|
||||
* </pre>
|
||||
*/
|
||||
const char *vosk_recognizer_partial_result(VoskRecognizer *recognizer);
|
||||
|
||||
|
||||
/** Returns speech recognition result. Same as result, but doesn't wait for silence
|
||||
* You usually call it in the end of the stream to get final bits of audio. It
|
||||
* flushes the feature pipeline, so all remaining audio chunks got processed.
|
||||
*
|
||||
* @returns speech result in JSON format.
|
||||
*/
|
||||
const char *vosk_recognizer_final_result(VoskRecognizer *recognizer);
|
||||
|
||||
|
||||
/** Releases recognizer object
|
||||
*
|
||||
* Underlying model is also unreferenced and if needed released */
|
||||
void vosk_recognizer_free(VoskRecognizer *recognizer);
|
||||
|
||||
/** Set log level for Kaldi messages
|
||||
*
|
||||
* @param log_level the level
|
||||
* 0 - default value to print info and error messages but no debug
|
||||
* less than 0 - don't print info messages
|
||||
* greather than 0 - more verbose mode
|
||||
*/
|
||||
void vosk_set_log_level(int log_level);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif /* _VOSK_API_H_ */
|
||||
#endif /* VOSK_API_H */
|
||||
|
||||
+25
-61
@@ -1,5 +1,5 @@
|
||||
ARG DOCKCROSS_IMAGE=linux-armv7
|
||||
FROM dockcross/${DOCKCROSS_IMAGE}
|
||||
ARG DOCKCROSS_IMAGE=alphacep/dockcross-linux-armv7
|
||||
FROM ${DOCKCROSS_IMAGE}
|
||||
|
||||
LABEL description="A docker image for building portable Python linux binary wheels and Kaldi on other architectures"
|
||||
LABEL maintainer="contact@alphacephei.com"
|
||||
@@ -10,73 +10,37 @@ RUN apt-get update && \
|
||||
libffi-dev \
|
||||
libpcre3-dev \
|
||||
zlib1g-dev \
|
||||
automake \
|
||||
autoconf \
|
||||
libtool \
|
||||
cmake \
|
||||
python3 \
|
||||
python3-pip \
|
||||
python3-wheel \
|
||||
python3-setuptools \
|
||||
python3-cffi \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN cd /opt \
|
||||
&& wget -O swig-4.0.1.tar.gz https://sourceforge.net/projects/swig/files/swig/swig-4.0.1/swig-4.0.1.tar.gz/download \
|
||||
&& tar xf swig-4.0.1.tar.gz \
|
||||
&& cd swig-4.0.1 \
|
||||
&& CPP=/usr/bin/cpp CXX=/usr/bin/g++ CC=/usr/bin/gcc ./configure --prefix=/usr && make -j 10 && make install \
|
||||
&& cd .. \
|
||||
&& rm -rf swig-4.0.1.tar.gz swig-4.0.1
|
||||
|
||||
ARG OPENBLAS_ARCH=ARMV7
|
||||
ARG ARM_HARDWARE_OPTS="-mfloat-abi=hard -mfpu=neon"
|
||||
RUN cd /opt \
|
||||
&& export OPENFST_CONFIGURE="--enable-static --enable-shared --enable-far --enable-ngram-fsts --enable-lookahead-fsts --with-pic --disable-bin --host=${CROSS_TRIPLE} --build=x86-linux-gnu" \
|
||||
&& git clone -b lookahead --single-branch https://github.com/alphacep/kaldi \
|
||||
&& git clone -b lookahead-1.8.0 --single-branch https://github.com/alphacep/kaldi \
|
||||
&& cd kaldi/tools \
|
||||
&& git clone -b v0.3.7 --single-branch https://github.com/xianyi/OpenBLAS \
|
||||
&& make PREFIX=$(pwd)/OpenBLAS/install TARGET="${OPENBLAS_ARCH}" HOSTCC=gcc USE_LOCKING=1 USE_THREAD=0 -C OpenBLAS all install \
|
||||
&& sed -i 's:status=0:exit 0:g' extras/check_dependencies.sh \
|
||||
&& make -j 10 openfst \
|
||||
&& git clone -b v0.3.13 --single-branch https://github.com/xianyi/OpenBLAS \
|
||||
&& git clone -b v3.2.1 --single-branch https://github.com/alphacep/clapack \
|
||||
&& make -C OpenBLAS ONLY_CBLAS=1 TARGET="${OPENBLAS_ARCH}" HOSTCC=gcc USE_LOCKING=1 USE_THREAD=0 all \
|
||||
&& make -C OpenBLAS PREFIX=$(pwd)/OpenBLAS/install install \
|
||||
&& mkdir -p clapack/BUILD && cd clapack/BUILD && cmake .. && make -j 10 && find . -name "*.a" | xargs cp -t ../../OpenBLAS/install/lib \
|
||||
&& cd /opt/kaldi/tools \
|
||||
&& git clone --single-branch https://github.com/alphacep/openfst openfst \
|
||||
&& cd openfst \
|
||||
&& autoreconf -i \
|
||||
&& CFLAGS="-g -O3" ./configure --prefix=/opt/kaldi/tools/openfst --enable-static --enable-shared --enable-far --enable-ngram-fsts --enable-lookahead-fsts --with-pic --disable-bin --host=${CROSS_TRIPLE} --build=x86-linux-gnu \
|
||||
&& make -j 10 && make install \
|
||||
&& cd /opt/kaldi/src \
|
||||
&& sed -i "s:TARGET_ARCH=\"\`uname -m\`\":TARGET_ARCH=$(echo $CROSS_TRIPLE|cut -d - -f 1):g" configure \
|
||||
&& sed -i "s:-mfloat-abi=hard -mfpu=neon:${ARM_HARDWARE_OPTS}:g" makefiles/linux_openblas_arm.mk \
|
||||
&& sed -i "s: -O1 : -O3 :g" makefiles/linux_openblas_arm.mk \
|
||||
&& ./configure --mathlib=OPENBLAS --shared --use-cuda=no \
|
||||
&& make -j 10 online2 \
|
||||
&& ./configure --mathlib=OPENBLAS_CLAPACK --shared --use-cuda=no \
|
||||
&& make -j 10 online2 lm \
|
||||
&& find /opt/kaldi -name "*.o" -exec rm {} \;
|
||||
|
||||
RUN cd /opt \
|
||||
&& wget -q https://github.com/openssl/openssl/archive/OpenSSL_1_0_2u.tar.gz \
|
||||
&& tar xf OpenSSL_1_0_2u.tar.gz \
|
||||
&& cd openssl-OpenSSL_1_0_2u \
|
||||
&& CROSS_COMPILE= MACHINE="$(echo $CROSS_TRIPLE|cut -d - -f 1)" ./config --prefix=$CROSS_ROOT shared \
|
||||
&& make -j $(nproc) \
|
||||
&& make install \
|
||||
&& rm -rf /opt/openssl-OpenSSL_1_0_2u /opt/OpenSSL_1_0_2u.tar.gz
|
||||
|
||||
RUN cd /opt \
|
||||
&& wget -q https://github.com/python/cpython/archive/v3.7.6.tar.gz \
|
||||
&& tar xf v3.7.6.tar.gz \
|
||||
&& cp -r cpython-3.7.6 cpython-3.7.6-cross \
|
||||
&& cd /opt/cpython-3.7.6 \
|
||||
&& AR=/usr/bin/ar RANLIB=/usr/bin/ranlib CPP=/usr/bin/cpp CXX=/usr/bin/g++ CC=/usr/bin/gcc ./configure --prefix="/opt/python/cp3.7-cp3.7m" \
|
||||
&& make -j $(nproc) \
|
||||
&& make install \
|
||||
&& /opt/python/cp3.7-cp3.7m/bin/pip3 install -U pip \
|
||||
&& /opt/python/cp3.7-cp3.7m/bin/pip3 install -U wheel \
|
||||
&& cd /opt/cpython-3.7.6-cross \
|
||||
&& export PATH=/opt/python/cp3.7-cp3.7m/bin:$PATH \
|
||||
&& ./configure --prefix=$CROSS_ROOT --with-openssl=$CROSS_ROOT --host=${CROSS_TRIPLE} --build=x86-linux-gnu --disable-ipv6 ac_cv_file__dev_ptmx=no ac_cv_file__dev_ptc=no ac_cv_have_long_long_format=yes \
|
||||
&& make -j $(nproc) \
|
||||
&& make install \
|
||||
&& rm -rf /opt/cpython-3.7.6 /opt/cpython-3.7.6-cross /opt/v3.7.6.tar.gz
|
||||
|
||||
RUN cd /opt \
|
||||
&& wget -q https://github.com/python/cpython/archive/v3.6.10.tar.gz \
|
||||
&& tar xf v3.6.10.tar.gz \
|
||||
&& cp -r cpython-3.6.10 cpython-3.6.10-cross \
|
||||
&& cd /opt/cpython-3.6.10 \
|
||||
&& AR=/usr/bin/ar RANLIB=/usr/bin/ranlib CPP=/usr/bin/cpp CXX=/usr/bin/g++ CC=/usr/bin/gcc ./configure --prefix="/opt/python/cp3.6-cp3.6m" \
|
||||
&& make -j $(nproc) \
|
||||
&& make install \
|
||||
&& /opt/python/cp3.6-cp3.6m/bin/pip3 install -U pip \
|
||||
&& /opt/python/cp3.6-cp3.6m/bin/pip3 install -U wheel \
|
||||
&& cd /opt/cpython-3.6.10-cross \
|
||||
&& export PATH=/opt/python/cp3.6-cp3.6m/bin:$PATH \
|
||||
&& ./configure --prefix=$CROSS_ROOT --with-openssl=$CROSS_ROOT --host=${CROSS_TRIPLE} --build=x86-linux-gnu --disable-ipv6 ac_cv_file__dev_ptmx=no ac_cv_file__dev_ptc=no ac_cv_have_long_long_format=yes \
|
||||
&& make -j $(nproc) \
|
||||
&& make install \
|
||||
&& rm -rf /opt/cpython-3.6.10 /opt/cpython-3.6.10-cross /opt/v3.6.10.tar.gz
|
||||
|
||||
+23
-29
@@ -4,36 +4,30 @@ LABEL description="A docker image for building portable Python linux binary whee
|
||||
LABEL maintainer="contact@alphacephei.com"
|
||||
|
||||
RUN yum -y update && yum -y install \
|
||||
wget \
|
||||
openssl-devel \
|
||||
pcre-devel \
|
||||
devtoolset-8-libatomic-devel \
|
||||
automake \
|
||||
autoconf \
|
||||
libtool \
|
||||
cmake \
|
||||
&& yum clean all
|
||||
|
||||
RUN cd /opt \
|
||||
&& wget https://github.com/Kitware/CMake/releases/download/v3.16.2/cmake-3.16.2.tar.gz \
|
||||
&& tar xf cmake-3.16.2.tar.gz \
|
||||
&& cd cmake-3.16.2 \
|
||||
&& ./configure --prefix=/usr && make -j 10 && make install \
|
||||
&& cd .. \
|
||||
&& rm -rf cmake-3.16.2 cmake-3.16.2.tar.gz
|
||||
|
||||
RUN cd /opt \
|
||||
&& export OPENFST_CONFIGURE="--enable-static --enable-shared --enable-far --enable-ngram-fsts --enable-lookahead-fsts --with-pic --disable-bin" \
|
||||
&& git clone -b lookahead --single-branch https://github.com/alphacep/kaldi \
|
||||
&& cd kaldi/tools \
|
||||
&& git clone -b v0.3.7 --single-branch https://github.com/xianyi/OpenBLAS \
|
||||
&& make PREFIX=$(pwd)/OpenBLAS/install TARGET=NEHALEM USE_LOCKING=1 USE_THREAD=0 -C OpenBLAS all install \
|
||||
&& sed -i 's:status=0:exit 0:g' extras/check_dependencies.sh \
|
||||
&& make -j 10 openfst \
|
||||
&& cd ../src \
|
||||
&& ./configure --mathlib=OPENBLAS --shared --use-cuda=no \
|
||||
&& make -j 10 online2 \
|
||||
&& git clone -b lookahead-1.8.0 --single-branch https://github.com/alphacep/kaldi \
|
||||
&& cd /opt/kaldi/tools \
|
||||
&& git clone -b v0.3.13 --single-branch https://github.com/xianyi/OpenBLAS \
|
||||
&& git clone -b v3.2.1 --single-branch https://github.com/alphacep/clapack \
|
||||
&& make -C OpenBLAS ONLY_CBLAS=1 DYNAMIC_ARCH=1 TARGET=NEHALEM USE_LOCKING=1 USE_THREAD=0 all \
|
||||
&& make -C OpenBLAS PREFIX=$(pwd)/OpenBLAS/install install \
|
||||
&& mkdir -p clapack/BUILD && cd clapack/BUILD && cmake .. && make -j 10 && find . -name "*.a" | xargs cp -t ../../OpenBLAS/install/lib \
|
||||
&& cd /opt/kaldi/tools \
|
||||
&& git clone --single-branch https://github.com/alphacep/openfst openfst \
|
||||
&& cd openfst \
|
||||
&& autoreconf -i \
|
||||
&& CFLAGS="-g -O3" ./configure --prefix=/opt/kaldi/tools/openfst --enable-static --enable-shared --enable-far --enable-ngram-fsts --enable-lookahead-fsts --with-pic --disable-bin \
|
||||
&& make -j 10 && make install \
|
||||
&& cd /opt/kaldi/src \
|
||||
&& ./configure --mathlib=OPENBLAS_CLAPACK --shared --use-cuda=no \
|
||||
&& sed -i 's:-msse -msse2:-msse -msse2:g' kaldi.mk \
|
||||
&& sed -i 's: -O1 : -O3 :g' kaldi.mk \
|
||||
&& make -j $(nproc) online2 lm rnnlm \
|
||||
&& find /opt/kaldi -name "*.o" -exec rm {} \;
|
||||
|
||||
RUN cd /opt \
|
||||
&& wget -O swig-4.0.1.tar.gz https://sourceforge.net/projects/swig/files/swig/swig-4.0.1/swig-4.0.1.tar.gz/download \
|
||||
&& tar xf swig-4.0.1.tar.gz \
|
||||
&& cd swig-4.0.1 \
|
||||
&& ./configure --prefix=/usr && make -j 10 && make install \
|
||||
&& cd .. \
|
||||
&& rm -rf swig-4.0.1.tar.gz swig-4.0.1
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
FROM debian:10.4
|
||||
|
||||
RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends \
|
||||
ca-certificates \
|
||||
g++ \
|
||||
bzip2 \
|
||||
unzip \
|
||||
make \
|
||||
wget \
|
||||
git \
|
||||
python3 \
|
||||
python3-pip \
|
||||
python3-wheel \
|
||||
python3-setuptools \
|
||||
python3-cffi \
|
||||
zlib1g-dev \
|
||||
patch \
|
||||
cmake \
|
||||
xz-utils \
|
||||
automake \
|
||||
autoconf \
|
||||
libtool \
|
||||
pkg-config \
|
||||
sudo \
|
||||
g++-mingw-w64-i686 \
|
||||
g++-mingw-w64-x86-64 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN mkdir /opt/kaldi \
|
||||
&& git clone https://github.com/alphacep/openfst \
|
||||
&& cd openfst \
|
||||
&& autoreconf -i \
|
||||
&& CXX=x86_64-w64-mingw32-g++-posix CXXFLAGS="-O3 -ftree-vectorize -DFST_NO_DYNAMIC_LINKING" \
|
||||
./configure --prefix=/opt/kaldi/local \
|
||||
--enable-shared --enable-static --with-pic --disable-bin \
|
||||
--enable-lookahead-fsts --enable-ngram-fsts --host=x86_64-w64-mingw32 \
|
||||
&& make -j $(nproc) \
|
||||
&& make install
|
||||
|
||||
RUN cd /opt/kaldi \
|
||||
&& git clone -b v0.3.13 --single-branch https://github.com/xianyi/OpenBLAS \
|
||||
&& cd OpenBLAS \
|
||||
&& make HOSTCC=gcc BINARY=64 CC=x86_64-w64-mingw32-gcc ONLY_CBLAS=1 DYNAMIC_ARCH=1 TARGET=NEHALEM USE_LOCKING=1 USE_THREAD=0 -j $(nproc) \
|
||||
&& make PREFIX=/opt/kaldi/local install
|
||||
|
||||
RUN cd /opt/kaldi \
|
||||
&& git clone -b v3.2.1 --single-branch https://github.com/alphacep/clapack \
|
||||
&& mkdir clapack/BUILD \
|
||||
&& cd clapack/BUILD \
|
||||
&& cmake -DCMAKE_C_COMPILER_TARGET=x86_64-w64-mingw32 -DCMAKE_C_COMPILER=x86_64-w64-mingw32-gcc-posix -DCMAKE_SYSTEM_NAME=Windows -DCMAKE_CROSSCOMPILING=True .. \
|
||||
&& make -C F2CLIBS/libf2c \
|
||||
&& make -C BLAS \
|
||||
&& make -C SRC \
|
||||
&& find . -name *.a -exec cp {} /opt/kaldi/local/lib \;
|
||||
|
||||
RUN cd /opt/kaldi \
|
||||
&& git clone -b android-mix --single-branch https://github.com/alphacep/kaldi \
|
||||
&& cd kaldi/src \
|
||||
&& CXX=x86_64-w64-mingw32-g++-posix CXXFLAGS="-O3 -ftree-vectorize -DFST_NO_DYNAMIC_LINKING" ./configure --shared --mingw=yes --use-cuda=no \
|
||||
--mathlib=OPENBLAS_CLAPACK \
|
||||
--host=x86_64-w64-mingw32 --openblas-clapack-root=/opt/kaldi/local \
|
||||
--fst-root=/opt/kaldi/local --fst-version=1.8.0 \
|
||||
&& make depend -j \
|
||||
&& make -j $(nproc) online2 lm
|
||||
@@ -0,0 +1,64 @@
|
||||
FROM debian:10.4
|
||||
|
||||
RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends \
|
||||
ca-certificates \
|
||||
g++ \
|
||||
bzip2 \
|
||||
unzip \
|
||||
make \
|
||||
wget \
|
||||
git \
|
||||
python3 \
|
||||
python3-pip \
|
||||
python3-wheel \
|
||||
python3-setuptools \
|
||||
python3-cffi \
|
||||
zlib1g-dev \
|
||||
patch \
|
||||
cmake \
|
||||
automake \
|
||||
autoconf \
|
||||
libtool \
|
||||
pkg-config \
|
||||
sudo \
|
||||
g++-mingw-w64-i686 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN mkdir /opt/kaldi \
|
||||
&& git clone https://github.com/alphacep/openfst \
|
||||
&& cd openfst \
|
||||
&& autoreconf -i \
|
||||
&& CXX=i686-w64-mingw32-g++-posix CXXFLAGS="-msse2 -O3 -ftree-vectorize -DFST_NO_DYNAMIC_LINKING" \
|
||||
./configure --prefix=/opt/kaldi/local \
|
||||
--enable-shared --enable-static --with-pic \
|
||||
--disable-bin --enable-lookahead-fsts --enable-ngram-fsts \
|
||||
--host=i686-w64-mingw32 \
|
||||
&& make -j $(nproc) \
|
||||
&& make install
|
||||
|
||||
RUN cd /opt/kaldi \
|
||||
&& git clone -b v0.3.13 --single-branch https://github.com/xianyi/OpenBLAS \
|
||||
&& cd OpenBLAS \
|
||||
&& make HOSTCC=gcc BINARY=32 CC=i686-w64-mingw32-gcc ONLY_CBLAS=1 DYNAMIC_ARCH=1 TARGET=NEHALEM USE_LOCKING=1 USE_THREAD=0 -j $(nproc) \
|
||||
&& make PREFIX=/opt/kaldi/local install
|
||||
|
||||
RUN cd /opt/kaldi \
|
||||
&& git clone -b v3.2.1 --single-branch https://github.com/alphacep/clapack \
|
||||
&& mkdir clapack/BUILD \
|
||||
&& cd clapack/BUILD \
|
||||
&& cmake -DCMAKE_C_COMPILER_TARGET=i686-w64-mingw32 -DCMAKE_C_COMPILER=i686-w64-mingw32-gcc-posix -DCMAKE_SYSTEM_NAME=Windows -DCMAKE_CROSSCOMPILING=True .. \
|
||||
&& make -C F2CLIBS/libf2c \
|
||||
&& make -C BLAS \
|
||||
&& make -C SRC \
|
||||
&& find . -name *.a -exec cp {} /opt/kaldi/local/lib \;
|
||||
|
||||
RUN cd /opt/kaldi \
|
||||
&& git clone -b android-mix --single-branch https://github.com/alphacep/kaldi \
|
||||
&& cd kaldi/src \
|
||||
&& CXX=i686-w64-mingw32-g++-posix CXXFLAGS="-O3 -ftree-vectorize -DFST_NO_DYNAMIC_LINKING" ./configure --shared --mingw=yes --use-cuda=no \
|
||||
--mathlib=OPENBLAS_CLAPACK \
|
||||
--host=i686-w64-mingw32 --openblas-clapack-root=/opt/kaldi/local \
|
||||
--fst-root=/opt/kaldi/local --fst-version=1.8.0 \
|
||||
&& make depend -j \
|
||||
&& make -j $(nproc) online2 lm
|
||||
@@ -3,12 +3,8 @@
|
||||
set -e
|
||||
set -x
|
||||
|
||||
skip() {
|
||||
docker build --build-arg="DOCKCROSS_IMAGE=linux-armv7" --build-arg="OPENBLAS_ARCH=ARMV7" --file Dockerfile.dockcross --tag alphacep/kaldi-dockcross-armv7:latest .
|
||||
docker build --build-arg="DOCKCROSS_IMAGE=linux-armv6" --build-arg="OPENBLAS_ARCH=ARMV6" --build-arg="ARM_HARDWARE_OPTS=" --file Dockerfile.dockcross --tag alphacep/kaldi-dockcross-armv6:latest .
|
||||
docker build --build-arg="DOCKCROSS_IMAGE=linux-arm64" --build-arg="OPENBLAS_ARCH=ARMV8" --file Dockerfile.dockcross --tag alphacep/kaldi-dockcross-arm64:latest .
|
||||
}
|
||||
docker build --build-arg="DOCKCROSS_IMAGE=alphacep/dockcross-linux-armv7" --build-arg="OPENBLAS_ARCH=ARMV7" --file Dockerfile.dockcross --tag alphacep/kaldi-dockcross-armv7:latest .
|
||||
docker build --build-arg="DOCKCROSS_IMAGE=dockcross/linux-arm64" --build-arg="OPENBLAS_ARCH=ARMV8" --file Dockerfile.dockcross --tag alphacep/kaldi-dockcross-arm64:latest .
|
||||
|
||||
docker run --rm -v /home/shmyrev/travis/vosk-api/:/io alphacep/kaldi-dockcross-armv6 /io/travis/build-wheels-dockcross.sh
|
||||
docker run --rm -v /home/shmyrev/travis/vosk-api/:/io alphacep/kaldi-dockcross-armv7 /io/travis/build-wheels-dockcross.sh
|
||||
docker run --rm -v /home/shmyrev/travis/vosk-api/:/io alphacep/kaldi-dockcross-arm64 /io/travis/build-wheels-dockcross.sh
|
||||
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/bin/bash
|
||||
|
||||
set -e -x
|
||||
docker build --file Dockerfile.win --tag alphacep/kaldi-win:latest .
|
||||
docker run --rm -v `realpath ..`:/io alphacep/kaldi-win /io/travis/build-wheels-win.sh
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/bin/bash
|
||||
|
||||
set -e -x
|
||||
docker build --file Dockerfile.win32 --tag alphacep/kaldi-win32:latest .
|
||||
docker run --rm -v `realpath ..`:/io alphacep/kaldi-win32 /io/travis/build-wheels-win32.sh
|
||||
@@ -1,7 +1,5 @@
|
||||
#!/bin/bash
|
||||
|
||||
set -e
|
||||
set -x
|
||||
|
||||
set -e -x
|
||||
docker build --file Dockerfile.manylinux --tag alphacep/kaldi-manylinux:latest .
|
||||
docker run --rm -e PLAT=manylinux2010_x86_64 -v /home/shmyrev/travis/vosk-api/:/io alphacep/kaldi-manylinux /io/travis/build-wheels.sh
|
||||
docker run --rm -v `realpath ..`:/io alphacep/kaldi-manylinux /io/travis/build-wheels.sh
|
||||
|
||||
@@ -1,25 +1,26 @@
|
||||
#!/bin/bash
|
||||
set -e -x
|
||||
|
||||
ORIG_PATH=$PATH
|
||||
for pyver in 3.6 3.7; do
|
||||
# Build so file
|
||||
cd /opt
|
||||
git clone https://github.com/alphacep/vosk-api
|
||||
cd /opt/vosk-api/src
|
||||
KALDI_ROOT=/opt/kaldi make -j $(nproc)
|
||||
|
||||
export KALDI_ROOT=/opt/kaldi
|
||||
export WHEEL_FLAGS=`$CROSS_ROOT/bin/python${pyver}-config --cflags`
|
||||
export PATH=/opt/python/cp${pyver}-cp${pyver}m/bin:$ORIG_PATH
|
||||
echo $CROSS_TRIPLE
|
||||
case $CROSS_TRIPLE in
|
||||
*arm-*)
|
||||
export _PYTHON_HOST_PLATFORM=linux-armv6l
|
||||
;;
|
||||
*armv7-*)
|
||||
export _PYTHON_HOST_PLATFORM=linux-armv7l
|
||||
;;
|
||||
*aarch64-*)
|
||||
export _PYTHON_HOST_PLATFORM=linux-aarch64
|
||||
;;
|
||||
esac
|
||||
# Decide architecture name
|
||||
export VOSK_SOURCE=/opt/vosk-api
|
||||
case $CROSS_TRIPLE in
|
||||
*armv7-*)
|
||||
export VOSK_ARCHITECTURE=armv7l
|
||||
;;
|
||||
*aarch64-*)
|
||||
export VOSK_ARCHITECTURE=aarch64
|
||||
;;
|
||||
esac
|
||||
|
||||
pip3 wheel /io/python -w /io/wheelhouse
|
||||
# Copy library to output folder
|
||||
mkdir -p /io/wheelhouse/linux-$VOSK_ARCHITECTURE
|
||||
cp /opt/vosk-api/src/*.so /io/wheelhouse/linux-$VOSK_ARCHITECTURE
|
||||
|
||||
done
|
||||
# Build wheel
|
||||
pip3 wheel /opt/vosk-api/python --no-deps -w /io/wheelhouse
|
||||
|
||||
Executable
+23
@@ -0,0 +1,23 @@
|
||||
#!/bin/bash
|
||||
set -e -x
|
||||
|
||||
# Build libvosk
|
||||
cd /opt
|
||||
git clone https://github.com/alphacep/vosk-api
|
||||
cd vosk-api/src
|
||||
CXX=x86_64-w64-mingw32-g++-posix EXT=dll KALDI_ROOT=/opt/kaldi/kaldi OPENFST_ROOT=/opt/kaldi/local OPENBLAS_ROOT=/opt/kaldi/local make -j $(nproc)
|
||||
|
||||
# Collect dependencies
|
||||
cp /usr/lib/gcc/x86_64-w64-mingw32/*-posix/libstdc++-6.dll /opt/vosk-api/src
|
||||
cp /usr/lib/gcc/x86_64-w64-mingw32/*-posix/libgcc_s_seh-1.dll /opt/vosk-api/src
|
||||
cp /usr/x86_64-w64-mingw32/lib/libwinpthread-1.dll /opt/vosk-api/src
|
||||
|
||||
# Copy dlls to output folder
|
||||
mkdir -p /io/wheelhouse/win64
|
||||
cp /opt/vosk-api/src/*.dll /io/wheelhouse/win64
|
||||
|
||||
# Build wheel and put to the output folder
|
||||
export VOSK_SOURCE=/opt/vosk-api
|
||||
export VOSK_PLATFORM=Windows
|
||||
export VOSK_ARCHITECTURE=64bit
|
||||
python3 -m pip -v wheel /opt/vosk-api/python --no-deps -w /io/wheelhouse
|
||||
Executable
+23
@@ -0,0 +1,23 @@
|
||||
#!/bin/bash
|
||||
set -e -x
|
||||
|
||||
# Build libvosk
|
||||
cd /opt
|
||||
git clone https://github.com/alphacep/vosk-api
|
||||
cd vosk-api/src
|
||||
CXX=i686-w64-mingw32-g++-posix EXT=dll KALDI_ROOT=/opt/kaldi/kaldi OPENFST_ROOT=/opt/kaldi/local OPENBLAS_ROOT=/opt/kaldi/local make -j $(nproc)
|
||||
|
||||
# Copy dependencies
|
||||
cp /usr/lib/gcc/i686-w64-mingw32/*-posix/libstdc++-6.dll /opt/vosk-api/src
|
||||
cp /usr/lib/gcc/i686-w64-mingw32/*-posix/libgcc_s_sjlj-1.dll /opt/vosk-api/src
|
||||
cp /usr/i686-w64-mingw32/lib/libwinpthread-1.dll /opt/vosk-api/src
|
||||
|
||||
# Copy dlls to output folder
|
||||
mkdir -p /io/wheelhouse/win32
|
||||
cp /opt/vosk-api/src/*.dll /io/wheelhouse/win32
|
||||
|
||||
# Build wheel and put to the output folder
|
||||
export VOSK_SOURCE=/opt/vosk-api
|
||||
export VOSK_PLATFORM=Windows
|
||||
export VOSK_ARCHITECTURE=32bit
|
||||
python3 -m pip -v wheel /opt/vosk-api/python --no-deps -w /io/wheelhouse
|
||||
+16
-9
@@ -1,16 +1,23 @@
|
||||
#!/bin/bash
|
||||
set -e -x
|
||||
|
||||
export KALDI_ROOT=/opt/kaldi
|
||||
# Build libvosk
|
||||
cd /opt
|
||||
git clone https://github.com/alphacep/vosk-api
|
||||
cd vosk-api/src
|
||||
KALDI_ROOT=/opt/kaldi OPENFST_ROOT=/opt/kaldi/tools/openfst OPENBLAS_ROOT=/opt/kaldi/tools/OpenBLAS/install make -j $(nproc)
|
||||
|
||||
# Compile wheels
|
||||
for pypath in /opt/python/cp3[56789]*; do
|
||||
export WHEEL_FLAGS=`${pypath}/bin/python3-config --cflags`
|
||||
mkdir -p /opt/wheelhouse
|
||||
"${pypath}/bin/pip" wheel /io/python -w /opt/wheelhouse
|
||||
done
|
||||
# Copy dlls to output folder
|
||||
mkdir -p /io/wheelhouse/linux
|
||||
cp /opt/vosk-api/src/*.so /io/wheelhouse/linux
|
||||
|
||||
# Bundle external shared libraries into the wheels
|
||||
# Build wheel and put to the output folder
|
||||
mkdir -p /opt/wheelhouse
|
||||
export VOSK_SOURCE=/opt/vosk-api
|
||||
/opt/python/cp37*/bin/pip -v wheel /opt/vosk-api/python --no-deps -w /opt/wheelhouse
|
||||
|
||||
# Fix manylinux
|
||||
for whl in /opt/wheelhouse/*.whl; do
|
||||
auditwheel repair "$whl" --plat $PLAT -w /io/wheelhouse/
|
||||
cp $whl /io/wheelhouse
|
||||
auditwheel repair "$whl" --plat manylinux2010_x86_64 -w /io/wheelhouse
|
||||
done
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
exports.printMsg = function() {
|
||||
console.log("This is a message from the Vosk package");
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
{
|
||||
"name": "vosk-js",
|
||||
"version": "0.3.0",
|
||||
"description": "Node binding for continuous voice recoginition through vosk-api.",
|
||||
"repository": {
|
||||
"type": "git",
|
||||
"url": "git://github.com/alphacep/vosk-api.git"
|
||||
},
|
||||
"main": "index.js",
|
||||
"keywords": [
|
||||
"speech",
|
||||
"speech recognition",
|
||||
"voice"
|
||||
],
|
||||
"author": "Alpha Cephei Inc.",
|
||||
"license": "Apache 2.0",
|
||||
"engines": {
|
||||
"node": ">= 12.x.x"
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user