Compare commits

..

5 Commits

Author SHA1 Message Date
Nickolay Shmyrev 8dd2166443 Another timestamp bugfix 2020-06-29 14:11:05 +02:00
Nickolay Shmyrev 7f3f25e170 Release memory when final result is received to reduce memory pressure. 2020-06-24 18:37:53 +02:00
Nickolay Shmyrev f48e4624bf Actually show the distance 2020-06-23 22:14:39 +02:00
Nickolay V. Shmyrev 856c935e92 Merge pull request #153 from He1nr1chK/master
Added cosine distance function
2020-06-23 21:54:47 +03:00
He1nr1chK 444123b37e Merge pull request #1 from He1nr1chK/He1nr1chK-Nodejs
Added cosine distance function
2020-06-23 20:11:22 +02:00
130 changed files with 1973 additions and 4993 deletions
+28 -38
View File
@@ -1,24 +1,19 @@
# Object files
*.o
# Built application files
*.apk
*.ap_
# Temp files
nohup.out
# Java class files
*.class
# Gradle files
.gradle/
build/
gradlew
gradlew.bat
gradle
local.properties
# Android
android/build
android/lib/build
android/model-en/build
android/model-en/src/main/assets/model-en-us
android/repo
*.apk
*.ap_
# Local configuration file (sdk path, etc)
local.properties
# Cmake
.cxx
@@ -29,8 +24,12 @@ wheelhouse
__pycache__
*.egg-info
python/dist
python/build
python/vosk/*.so
python/vosk/*.cc
python/vosk/*.c
python/vosk/*.h
python/vosk/*.i
python/vosk/vosk.py
python/vosk/vosk_wrap.cpp
python/test/db
python/test/hyp
python/test/model
@@ -39,32 +38,23 @@ python/test/result.txt
python/test/wav.scp
# Java
*.class
java/lib/model
java/demo/model
java/lib/build
java/demo/build
*.so
java/org
java/*.cc
java/model-spk/
java/model/
# CSharp
*.dll
*.so
*.dylib
*.nupkg
csharp/demo/model
csharp/demo/test.wav
csharp/demo/bin
csharp/demo/obj
csharp/gen
csharp/*.exe
csharp/*.c
csharp/model/
csharp/test.wav
# Node
nodejs/demo/model
nodejs/demo/model-spk
nodejs/demo/test.wav
nodejs/vosk_wrap.cc
nodejs/example/model
nodejs/example/test.wav
nodejs/node_modules
nodejs/package-lock.json
# C
c/test_vosk
c/test_vosk_speaker
c/oprofile_data
c/model
c/test.wav
nodejs/build
+2 -1
View File
@@ -7,10 +7,11 @@ matrix:
services:
- docker
env: DOCKER_IMAGE=alphacep/kaldi-manylinux:latest
PLAT=manylinux2010_x86_64
install:
- docker pull $DOCKER_IMAGE
script:
- docker run --rm -v `pwd`:/io $DOCKER_IMAGE $PRE_CMD /io/travis/build-wheels.sh
- docker run --rm -e PLAT=$PLAT -v `pwd`:/io $DOCKER_IMAGE $PRE_CMD /io/travis/build-wheels.sh
- ls wheelhouse/
+20 -20
View File
@@ -1,26 +1,26 @@
# About
Vosk is an offline open source speech recognition toolkit. It enables
speech recognition models for 18 languages and dialects - English, Indian
English, German, French, Spanish, Portuguese, Chinese, Russian, Turkish,
Vietnamese, Italian, Dutch, Catalan, Arabic, Greek, Farsi, Filipino,
Ukrainian.
Vosk models are small (50 Mb) but provide continuous large vocabulary
transcription, zero-latency response with streaming API, reconfigurable
Vosk is an open source speech recognition toolkit which supports 9
languages - English, German, French, Spanish, Portuguese, Chinese,
Russian, Turkish, Vietnamese. Vosk works offline with small (50 Mb), but
accurate model, zero-latency response with streaming API, reconfigurable
vocabulary and speaker identification.
Speech recognition bindings implemented for various programming languages
like Python, Java, Node.JS, C#, C++ and others.
### Installation and usage
Vosk supplies speech recognition for chatbots, smart home appliances,
virtual assistants. It can also create subtitles for movies,
transcription for lectures and interviews.
For Vosk installation instructions, examples and turorial and documentation visit https://alphacephei.com/vosk
Vosk scales from small devices like Raspberry Pi or Android smartphone to
big clusters.
### Build
# Documentation
[![Build Status](https://travis-ci.com/alphacep/vosk-api.svg?branch=master)](https://travis-ci.com/alphacep/vosk-api)
For installation instructions, examples and documentation visit [Vosk
Website](https://alphacephei.com/vosk).
### Models for different languages
For information about models see [the documentation on available models](https://alphacephei.com/vosk/models.html).
### Contact Us
If you have any questions, feel free to:
* Post an issue here on github
* Send us an e-mail at [contact@alphacephei.com](mailto:contact@alphacephei.com)
* Join our group dedicated to speech recognition on Telegram [@speech_recognition](https://t.me/speech_recognition)
* We have a Wechat group which is pretty big, so it is invitation-only. Mail us to join the group and provide some information about yourself.
+67
View File
@@ -0,0 +1,67 @@
# Vosk CMake File
cmake_minimum_required(VERSION 3.4.1)
if ("x${ANDROID_ABI}" STREQUAL "xarmeabi-v7a")
set(OPENBLAS_ARCH "armv7")
set(KALDI_SUFFIX "arm_32")
elseif ("x${ANDROID_ABI}" STREQUAL "xarm64-v8a")
set(OPENBLAS_ARCH "armv8")
set(KALDI_SUFFIX "arm_64")
elseif ("x${ANDROID_ABI}" STREQUAL "xx86")
set(OPENBLAS_ARCH "atom")
set(KALDI_SUFFIX "x86")
else ("x${ANDROID_ABI}" STREQUAL "xarmeabi-v7a")
set(OPENBLAS_ARCH "atom")
set(KALDI_SUFFIX "x86_64")
endif ("x${ANDROID_ABI}" STREQUAL "xarmeabi-v7a")
set(KALDI_ROOT "${PROJECT_SOURCE_DIR}/build/kaldi_${KALDI_SUFFIX}/kaldi")
set(LIB_ROOT "${PROJECT_SOURCE_DIR}/build/kaldi_${KALDI_SUFFIX}/local")
set(API_SOURCES
"${PROJECT_SOURCE_DIR}/../src/kaldi_recognizer.cc"
"${PROJECT_SOURCE_DIR}/../src/kaldi_recognizer.h"
"${PROJECT_SOURCE_DIR}/../src/model.cc"
"${PROJECT_SOURCE_DIR}/../src/model.h"
"${PROJECT_SOURCE_DIR}/../src/spk_model.cc"
"${PROJECT_SOURCE_DIR}/../src/spk_model.h"
"${PROJECT_SOURCE_DIR}/../src/vosk_api.cc"
"${PROJECT_SOURCE_DIR}/../src/vosk_api.h"
)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -O3 -DFST_NO_DYNAMIC_LINKING")
add_library( kaldi_jni SHARED
build/generated-src/cpp/vosk_wrap.cc
${API_SOURCES}
)
include_directories("${PROJECT_SOURCE_DIR}/../src" "build/kaldi_${KALDI_SUFFIX}/kaldi/src" "build/kaldi_${KALDI_SUFFIX}/local/include")
target_link_libraries( kaldi_jni
${KALDI_ROOT}/src/online2/kaldi-online2.a
${KALDI_ROOT}/src/decoder/kaldi-decoder.a
${KALDI_ROOT}/src/ivector/kaldi-ivector.a
${KALDI_ROOT}/src/gmm/kaldi-gmm.a
${KALDI_ROOT}/src/nnet3/kaldi-nnet3.a
${KALDI_ROOT}/src/tree/kaldi-tree.a
${KALDI_ROOT}/src/feat/kaldi-feat.a
${KALDI_ROOT}/src/lat/kaldi-lat.a
${KALDI_ROOT}/src/lm/kaldi-lm.a
${KALDI_ROOT}/src/hmm/kaldi-hmm.a
${KALDI_ROOT}/src/transform/kaldi-transform.a
${KALDI_ROOT}/src/cudamatrix/kaldi-cudamatrix.a
${KALDI_ROOT}/src/matrix/kaldi-matrix.a
${KALDI_ROOT}/src/fstext/kaldi-fstext.a
${KALDI_ROOT}/src/util/kaldi-util.a
${KALDI_ROOT}/src/base/kaldi-base.a
${LIB_ROOT}/lib/libfst.a
${LIB_ROOT}/lib/libfstngram.a
${LIB_ROOT}/lib/libopenblas_${OPENBLAS_ARCH}-r0.3.7.a
${LIB_ROOT}/lib/libclapack.a
${LIB_ROOT}/lib/liblapack.a
${LIB_ROOT}/lib/libblas.a
${LIB_ROOT}/lib/libf2c.a
log
)
+21 -2
View File
@@ -1,3 +1,22 @@
Vosk library for Android
This is still work in progress, more to come
See for details https://alphacephei.com/vosk/android
## TODO
* Optimize graph construction, current one is below accuracy
* Load model from the AAR (mmap them in tflite style)
* Add decoding speed measurement
* Add wakeup word
* Add speakerid
* Integrate proper hardware optimized neural network library. Candidates are:
* https://github.com/XiaoMi/mace
* https://github.com/Tencent/ncnn
* https://developer.android.com/ndk/guides/neuralnetworks/ (since API level 27)
* https://github.com/google/XNNPACK
* Quantization for the models
@@ -1,6 +1,6 @@
#!/bin/bash
# Copyright 2019-2021 Alpha Cephei Inc.
# Copyright 2019 Alpha Cephei Inc.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
@@ -14,39 +14,70 @@
# See the License for the specific language governing permissions and
# limitations under the License.
if [ "x$ANDROID_NDK_HOME" == "x" ]; then
echo "ANDROID_NDK_HOME environment variable is undefined, define it with local.properties or with export"
if [ "x$ANDROID_SDK_HOME" == "x" ]; then
echo "ANDROID_SDK_HOME environment variable is undefined, define it with local.properties or with export"
exit 1
fi
if [ ! -d "$ANDROID_NDK_HOME" ]; then
echo "ANDROID_NDK_HOME ($ANDROID_NDK_HOME) is missing. Make sure you have ndk installed"
if [ ! -d "$ANDROID_SDK_HOME" ]; then
echo "ANDROID_SDK_HOME ($ANDROID_SDK_HOME) is missing. Make sure you have sdk installed"
exit 1
fi
if [ ! -d "$ANDROID_SDK_HOME/ndk-bundle" ]; then
echo "$ANDROID_SDK_HOME/ndk-bundle is missing. Make sure you have ndk installed within sdk"
exit 1
fi
set -x
OS_NAME=`echo $(uname -s) | tr '[:upper:]' '[:lower:]'`
ANDROID_NDK_HOME=$ANDROID_SDK_HOME/ndk-bundle
ANDROID_TOOLCHAIN_PATH=$ANDROID_NDK_HOME/toolchains/llvm/prebuilt/${OS_NAME}-x86_64
WORKDIR_BASE=`pwd`/build
WORKDIR_X86=`pwd`/build/kaldi_x86
WORKDIR_X86_64=`pwd`/build/kaldi_x86_64
WORKDIR_ARM32=`pwd`/build/kaldi_arm_32
WORKDIR_ARM64=`pwd`/build/kaldi_arm_64
PATH=$PATH:$ANDROID_NDK_HOME/toolchains/llvm/prebuilt/${OS_NAME}-x86_64/bin
OPENFST_VERSION=1.8.0
OPENFST_VERSION=1.6.7
for arch in armeabi-v7a arm64-v8a x86_64 x86; do
mkdir -p $WORKDIR_ARM64/local/lib $WORKDIR_ARM32/local/lib $WORKDIR_X86_64/local/lib $WORKDIR_X86/local/lib
WORKDIR=${WORKDIR_BASE}/kaldi_${arch}
# Build standalone CLAPACK since gfortran is missing
cd build
git clone https://github.com/simonlynen/android_libs
cd android_libs/lapack
sed -i.bak -e 's/APP_STL := gnustl_static/APP_STL := c++_static/g' jni/Application.mk && \
sed -i.bak -e 's/android-10/android-21/g' project.properties && \
sed -i.bak -e 's/APP_ABI := armeabi armeabi-v7a/APP_ABI := armeabi-v7a arm64-v8a x86_64 x86/g' jni/Application.mk && \
sed -i.bak -e 's/LOCAL_MODULE:= testlapack/#LOCAL_MODULE:= testlapack/g' jni/Android.mk && \
sed -i.bak -e 's/LOCAL_SRC_FILES:= testclapack.cpp/#LOCAL_SRC_FILES:= testclapack.cpp/g' jni/Android.mk && \
sed -i.bak -e 's/LOCAL_STATIC_LIBRARIES := lapack/#LOCAL_STATIC_LIBRARIES := lapack/g' jni/Android.mk && \
sed -i.bak -e 's/include $(BUILD_SHARED_LIBRARY)/#include $(BUILD_SHARED_LIBRARY)/g' jni/Android.mk && \
${ANDROID_NDK_HOME}/ndk-build && \
cp obj/local/armeabi-v7a/*.a ${WORKDIR_ARM32}/local/lib && \
cp obj/local/arm64-v8a/*.a ${WORKDIR_ARM64}/local/lib
cp obj/local/x86_64/*.a ${WORKDIR_X86_64}/local/lib
cp obj/local/x86/*.a ${WORKDIR_X86}/local/lib
# Architecture-specific part
for arch in arm32 arm64 x86_64 x86; do
#for arch in x86_64; do
case $arch in
armeabi-v7a)
arm32)
BLAS_ARCH=ARMV7
WORKDIR=$WORKDIR_ARM32
HOST=arm-linux-androideabi
AR=arm-linux-androideabi-ar
CC=armv7a-linux-androideabi21-clang
CXX=armv7a-linux-androideabi21-clang++
ARCHFLAGS="-mfloat-abi=softfp -mfpu=neon"
;;
arm64-v8a)
arm64)
BLAS_ARCH=ARMV8
WORKDIR=$WORKDIR_ARM64
HOST=aarch64-linux-android
AR=aarch64-linux-android-ar
CC=aarch64-linux-android21-clang
@@ -55,6 +86,7 @@ case $arch in
;;
x86_64)
BLAS_ARCH=ATOM
WORKDIR=$WORKDIR_X86_64
HOST=x86_64-linux-android
AR=x86_64-linux-android-ar
CC=x86_64-linux-android21-clang
@@ -63,6 +95,7 @@ case $arch in
;;
x86)
BLAS_ARCH=ATOM
WORKDIR=$WORKDIR_X86
HOST=i686-linux-android
AR=i686-linux-android-ar
CC=i686-linux-android21-clang
@@ -71,27 +104,12 @@ case $arch in
;;
esac
mkdir -p $WORKDIR/local/lib
# openblas first
cd $WORKDIR
git clone -b v0.3.13 --single-branch https://github.com/xianyi/OpenBLAS
git clone -b v0.3.7 --single-branch https://github.com/xianyi/OpenBLAS
make -C OpenBLAS TARGET=$BLAS_ARCH ONLY_CBLAS=1 AR=$AR CC=$CC HOSTCC=gcc ARM_SOFTFP_ABI=1 USE_THREAD=0 NUM_THREADS=1 -j4
make -C OpenBLAS install PREFIX=$WORKDIR/local
# CLAPACK
cd $WORKDIR
git clone -b v3.2.1 --single-branch https://github.com/alphacep/clapack
mkdir -p clapack/BUILD && cd clapack/BUILD
cmake -DCMAKE_C_FLAGS=$ARCHFLAGS -DCMAKE_C_COMPILER_TARGET=$HOST \
-DCMAKE_C_COMPILER=$CC -DCMAKE_SYSTEM_NAME=Generic -DCMAKE_AR=$ANDROID_NDK_HOME/toolchains/llvm/prebuilt/${OS_NAME}-x86_64/bin/$AR \
-DCMAKE_TRY_COMPILE_TARGET_TYPE=STATIC_LIBRARY \
-DCMAKE_CROSSCOMPILING=True ..
make -j 8 -C F2CLIBS/libf2c
make -j 8 -C BLAS/SRC
make -j 8 -C SRC
find . -name "*.a" | xargs cp -t $WORKDIR/local/lib
# tools directory --> we'll only compile OpenFST
cd $WORKDIR
git clone https://github.com/alphacep/openfst
@@ -107,23 +125,17 @@ make install
cd $WORKDIR
git clone -b android-mix --single-branch https://github.com/alphacep/kaldi
cd $WORKDIR/kaldi/src
if [ "`uname`" == "Darwin" ]; then
sed -i.bak -e 's/libfst.dylib/libfst.a/' configure
fi
CXX=$CXX CXXFLAGS="$ARCHFLAGS -O3 -DFST_NO_DYNAMIC_LINKING" ./configure --use-cuda=no \
--mathlib=OPENBLAS_CLAPACK --shared \
--mathlib=OPENBLAS --shared \
--android-incdir=${ANDROID_TOOLCHAIN_PATH}/sysroot/usr/include \
--host=$HOST --openblas-root=${WORKDIR}/local \
--fst-root=${WORKDIR}/local --fst-version=${OPENFST_VERSION}
make -j 8 depend
cd $WORKDIR/kaldi/src
make -j 8 online2 lm rnnlm
# Vosk-api
cd $WORKDIR
#rm -rf vosk-api
git clone -b master --single-branch https://github.com/alphacep/vosk-api
cd vosk-api/src
make -j 8 KALDI_ROOT=${WORKDIR}/kaldi OPENFST_ROOT=${WORKDIR}/local OPENBLAS_ROOT=${WORKDIR}/local CXX=$CXX EXTRA_LDFLAGS="-llog -static-libstdc++"
# Copy JNI library to sources
cp $WORKDIR/vosk-api/src/libvosk.so $WORKDIR/../../src/main/jniLibs/$arch/libvosk.so
make -j 8 online2 lm
done
+50 -40
View File
@@ -4,55 +4,65 @@ buildscript {
jcenter()
}
dependencies {
classpath 'com.android.tools.build:gradle:4.1.3'
classpath 'com.android.tools.build:gradle:3.5.3'
}
}
allprojects {
version = '0.3.30'
repositories {
google()
jcenter()
}
}
subprojects {
apply plugin: 'com.android.library'
apply plugin: 'com.android.library'
apply plugin: 'maven-publish'
repositories {
google()
jcenter()
}
publishing {
publications {
aar(MavenPublication) {
groupId 'com.alphacephei'
version version
pom {
url = 'http://www.alphacephei.com.com/vosk/'
licenses {
license {
name = 'The Apache License, Version 2.0'
url = 'http://www.apache.org/licenses/LICENSE-2.0.txt'
}
}
developers {
developer {
id = 'com.alphacephei'
name = 'Alpha Cephei Inc'
email = 'contact@alphacephei.com'
}
}
scm {
connection = 'scm:git:git://github.com/alphacep/vosk-api.git'
url = 'https://github.com/alphacep/vosk-api/'
}
}
android {
compileSdkVersion 29
defaultConfig {
minSdkVersion 21
targetSdkVersion 29
versionCode 5
versionName "5.2"
setProperty("archivesBaseName", "kaldi-android-$versionName")
externalNativeBuild {
cmake {
arguments "-DCMAKE_VERBOSE_MAKEFILE=ON", "-DANDROID_ARM_NEON=TRUE", "-DCMAKE_CXX_FLAGS_RELEASE=-O3"
}
}
repositories {
maven {
url = "$rootDir/repo"
}
ndk {
abiFilters 'armeabi-v7a', 'arm64-v8a', 'x86_64', 'x86'
}
}
sourceSets {
main {
java.srcDirs = ['src/main/java', 'build/generated-src']
}
}
externalNativeBuild {
cmake {
path "CMakeLists.txt"
}
}
}
task swig {
doLast {
mkdir 'build/generated-src/java'
mkdir 'build/generated-src/cpp'
exec {
commandLine 'swig',
"-c++",
"-java", "-package", "org.kaldi",
"-outdir", "build/generated-src/java", "-o", "build/generated-src/cpp/vosk_wrap.cc",
"../src/vosk.i"
}
}
}
task kaldi(type: Exec) {
commandLine './build-kaldi.sh'
environment ANDROID_SDK_HOME: android.getSdkDirectory()
}
preBuild.dependsOn kaldi, swig
-42
View File
@@ -1,42 +0,0 @@
def archiveName = "vosk-android"
def pomName = "Vosk Android"
def pomDescription = "Vosk speech recognition library for Android"
android {
compileSdkVersion 29
defaultConfig {
minSdkVersion 21
targetSdkVersion 29
versionCode 6
versionName = version
archivesBaseName = archiveName
}
compileOptions {
sourceCompatibility JavaVersion.VERSION_1_8
targetCompatibility JavaVersion.VERSION_1_8
}
}
task buildVosk(type: Exec) {
commandLine './build-vosk.sh'
environment ANDROID_NDK_HOME: android.getSdkDirectory()
}
dependencies {
implementation 'net.java.dev.jna:jna:4.4.0@aar'
}
//preBuild.dependsOn buildVosk
publishing {
publications {
aar(MavenPublication) {
artifactId = archiveName
artifact("$buildDir/outputs/aar/$archiveName-release.aar")
pom {
name = pomName
description = pomDescription
}
}
}
}
-1
View File
@@ -1 +0,0 @@
include 'model-en'
@@ -1,60 +0,0 @@
package org.vosk;
import com.sun.jna.Native;
import com.sun.jna.Library;
import com.sun.jna.Platform;
import com.sun.jna.Pointer;
import java.io.File;
import java.io.InputStream;
import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.StandardCopyOption;
public class LibVosk {
static {
Native.register(LibVosk.class, "vosk");
}
public static native void vosk_set_log_level(int level);
public static native Pointer vosk_model_new(String path);
public static native void vosk_model_free(Pointer model);
public static native Pointer vosk_spk_model_new(String path);
public static native void vosk_spk_model_free(Pointer model);
public static native Pointer vosk_recognizer_new(Model model, float sample_rate);
public static native Pointer vosk_recognizer_new_spk(Pointer model, float sample_rate, Pointer spk_model);
public static native Pointer vosk_recognizer_new_grm(Pointer model, float sample_rate, String grammar);
public static native void vosk_recognizer_set_max_alternatives(Pointer recognizer, int max_alternatives);
public static native void vosk_recognizer_set_words(Pointer recognizer, boolean words);
public static native void vosk_recognizer_set_spk_model(Pointer recognizer, Pointer spk_model);
public static native boolean vosk_recognizer_accept_waveform(Pointer recognizer, byte[] data, int len);
public static native boolean vosk_recognizer_accept_waveform_s(Pointer recognizer, short[] data, int len);
public static native boolean vosk_recognizer_accept_waveform_f(Pointer recognizer, float[] data, int len);
public static native String vosk_recognizer_result(Pointer recognizer);
public static native String vosk_recognizer_final_result(Pointer recognizer);
public static native String vosk_recognizer_partial_result(Pointer recognizer);
public static native void vosk_recognizer_reset(Pointer recognizer);
public static native void vosk_recognizer_free(Pointer recognizer);
public static void setLogLevel(LogLevel loglevel) {
vosk_set_log_level(loglevel.getValue());
}
}
@@ -1,17 +0,0 @@
package org.vosk;
public enum LogLevel {
WARNINGS(-1), // Print warning and errors
INFO(0), // Print info, along with warning and error messages, but no debug
DEBUG(1); // Print debug info
private final int value;
LogLevel(int value) {
this.value = value;
}
public int getValue() {
return this.value;
}
}
@@ -1,17 +0,0 @@
package org.vosk;
import com.sun.jna.PointerType;
public class Model extends PointerType implements AutoCloseable {
public Model() {
}
public Model(String path) {
super(LibVosk.vosk_model_new(path));
}
@Override
public void close() {
LibVosk.vosk_model_free(this.getPointer());
}
}
@@ -1,62 +0,0 @@
package org.vosk;
import com.sun.jna.PointerType;
public class Recognizer extends PointerType implements AutoCloseable {
public Recognizer(Model model, float sampleRate) {
super(LibVosk.vosk_recognizer_new(model, sampleRate));
}
public Recognizer(Model model, float sampleRate, SpeakerModel spkModel) {
super(LibVosk.vosk_recognizer_new_spk(model.getPointer(), sampleRate, spkModel.getPointer()));
}
public Recognizer(Model model, float sampleRate, String grammar) {
super(LibVosk.vosk_recognizer_new_grm(model.getPointer(), sampleRate, grammar));
}
public void setMaxAlternatives(int maxAlternatives) {
LibVosk.vosk_recognizer_set_max_alternatives(this.getPointer(), maxAlternatives);
}
public void setWords(boolean words) {
LibVosk.vosk_recognizer_set_words(this.getPointer(), words);
}
public void setSpeakerModel(SpeakerModel spkModel) {
LibVosk.vosk_recognizer_set_spk_model(this.getPointer(), spkModel.getPointer());
}
public boolean acceptWaveForm(byte[] data, int len) {
return LibVosk.vosk_recognizer_accept_waveform(this.getPointer(), data, len);
}
public boolean acceptWaveForm(short[] data, int len) {
return LibVosk.vosk_recognizer_accept_waveform_s(this.getPointer(), data, len);
}
public boolean acceptWaveForm(float[] data, int len) {
return LibVosk.vosk_recognizer_accept_waveform_f(this.getPointer(), data, len);
}
public String getResult() {
return LibVosk.vosk_recognizer_result(this.getPointer());
}
public String getPartialResult() {
return LibVosk.vosk_recognizer_partial_result(this.getPointer());
}
public String getFinalResult() {
return LibVosk.vosk_recognizer_final_result(this.getPointer());
}
public void reset() {
LibVosk.vosk_recognizer_reset(this.getPointer());
}
@Override
public void close() {
LibVosk.vosk_recognizer_free(this.getPointer());
}
}
@@ -1,17 +0,0 @@
package org.vosk;
import com.sun.jna.PointerType;
public class SpeakerModel extends PointerType implements AutoCloseable {
public SpeakerModel() {
}
public SpeakerModel(String path) {
super(LibVosk.vosk_spk_model_new(path));
}
@Override
public void close() {
LibVosk.vosk_spk_model_free(this.getPointer());
}
}
@@ -1,257 +0,0 @@
// Copyright 2019 Alpha Cephei Inc.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
package org.vosk.android;
import android.media.AudioFormat;
import android.media.AudioRecord;
import android.media.MediaRecorder.AudioSource;
import android.os.Handler;
import android.os.Looper;
import org.vosk.Recognizer;
import java.io.IOException;
/**
* Service that records audio in a thread, passes it to a recognizer and emits
* recognition results. Recognition events are passed to a client using
* {@link RecognitionListener}
*/
public class SpeechService {
private final Recognizer recognizer;
private final int sampleRate;
private final static float BUFFER_SIZE_SECONDS = 0.2f;
private final int bufferSize;
private final AudioRecord recorder;
private RecognizerThread recognizerThread;
private final Handler mainHandler = new Handler(Looper.getMainLooper());
/**
* Creates speech service. Service holds the AudioRecord object, so you
* need to call {@link #shutdown()} in order to properly finalize it.
*
* @throws IOException thrown if audio recorder can not be created for some reason.
*/
public SpeechService(Recognizer recognizer, float sampleRate) throws IOException {
this.recognizer = recognizer;
this.sampleRate = (int) sampleRate;
bufferSize = Math.round(this.sampleRate * BUFFER_SIZE_SECONDS);
recorder = new AudioRecord(
AudioSource.VOICE_RECOGNITION, this.sampleRate,
AudioFormat.CHANNEL_IN_MONO,
AudioFormat.ENCODING_PCM_16BIT, bufferSize * 2);
if (recorder.getState() == AudioRecord.STATE_UNINITIALIZED) {
recorder.release();
throw new IOException(
"Failed to initialize recorder. Microphone might be already in use.");
}
}
/**
* Starts recognition. Does nothing if recognition is active.
*
* @return true if recognition was actually started
*/
public boolean startListening(RecognitionListener listener) {
if (null != recognizerThread)
return false;
recognizerThread = new RecognizerThread(listener);
recognizerThread.start();
return true;
}
/**
* Starts recognition. After specified timeout listening stops and the
* endOfSpeech signals about that. Does nothing if recognition is active.
* <p>
* timeout - timeout in milliseconds to listen.
*
* @return true if recognition was actually started
*/
public boolean startListening(RecognitionListener listener, int timeout) {
if (null != recognizerThread)
return false;
recognizerThread = new RecognizerThread(listener, timeout);
recognizerThread.start();
return true;
}
private boolean stopRecognizerThread() {
if (null == recognizerThread)
return false;
try {
recognizerThread.interrupt();
recognizerThread.join();
} catch (InterruptedException e) {
// Restore the interrupted status.
Thread.currentThread().interrupt();
}
recognizerThread = null;
return true;
}
/**
* Stops recognition. Listener should receive final result if there is
* any. Does nothing if recognition is not active.
*
* @return true if recognition was actually stopped
*/
public boolean stop() {
return stopRecognizerThread();
}
/**
* Cancel recognition. Do not post any new events, simply cancel processing.
* Does nothing if recognition is not active.
*
* @return true if recognition was actually stopped
*/
public boolean cancel() {
if (recognizerThread != null) {
recognizerThread.setPause(true);
}
return stopRecognizerThread();
}
/**
* Shutdown the recognizer and release the recorder
*/
public void shutdown() {
recorder.release();
}
public void setPause(boolean paused) {
if (recognizerThread != null) {
recognizerThread.setPause(paused);
}
}
/**
* Resets recognizer in a thread, starts recognition over again
*/
public void reset() {
if (recognizerThread != null) {
recognizerThread.reset();
}
}
private final class RecognizerThread extends Thread {
private int remainingSamples;
private final int timeoutSamples;
private final static int NO_TIMEOUT = -1;
private volatile boolean paused = false;
private volatile boolean reset = false;
RecognitionListener listener;
public RecognizerThread(RecognitionListener listener, int timeout) {
this.listener = listener;
if (timeout != NO_TIMEOUT)
this.timeoutSamples = timeout * sampleRate / 1000;
else
this.timeoutSamples = NO_TIMEOUT;
this.remainingSamples = this.timeoutSamples;
}
public RecognizerThread(RecognitionListener listener) {
this(listener, NO_TIMEOUT);
}
/**
* When we are paused, don't process audio by the recognizer and don't emit
* any listener results
*
* @param paused the status of pause
*/
public void setPause(boolean paused) {
this.paused = paused;
}
/**
* Set reset state to signal reset of the recognizer and start over
*/
public void reset() {
this.reset = true;
}
@Override
public void run() {
recorder.startRecording();
if (recorder.getRecordingState() == AudioRecord.RECORDSTATE_STOPPED) {
recorder.stop();
IOException ioe = new IOException(
"Failed to start recording. Microphone might be already in use.");
mainHandler.post(() -> listener.onError(ioe));
}
short[] buffer = new short[bufferSize];
while (!interrupted()
&& ((timeoutSamples == NO_TIMEOUT) || (remainingSamples > 0))) {
int nread = recorder.read(buffer, 0, buffer.length);
if (paused) {
continue;
}
if (reset) {
recognizer.reset();
reset = false;
}
if (nread < 0)
throw new RuntimeException("error reading audio buffer");
if (recognizer.acceptWaveForm(buffer, nread)) {
final String result = recognizer.getResult();
mainHandler.post(() -> listener.onResult(result));
} else {
final String partialResult = recognizer.getPartialResult();
mainHandler.post(() -> listener.onPartialResult(partialResult));
}
if (timeoutSamples != NO_TIMEOUT) {
remainingSamples = remainingSamples - nread;
}
}
recorder.stop();
if (!paused) {
// If we met timeout signal that speech ended
if (timeoutSamples != NO_TIMEOUT && remainingSamples <= 0) {
mainHandler.post(() -> listener.onTimeout());
} else {
final String finalResult = recognizer.getFinalResult();
mainHandler.post(() -> listener.onFinalResult(finalResult));
}
}
}
}
}
@@ -1,165 +0,0 @@
// Copyright 2019 Alpha Cephei Inc.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
package org.vosk.android;
import android.os.Handler;
import android.os.Looper;
import org.vosk.Recognizer;
import java.io.IOException;
import java.io.InputStream;
/**
* Service that recognizes stream audio in a thread, passes it to a recognizer and emits
* recognition results. Recognition events are passed to a client using
* {@link RecognitionListener}
*/
public class SpeechStreamService {
private final Recognizer recognizer;
private final InputStream inputStream;
private final int sampleRate;
private final static float BUFFER_SIZE_SECONDS = 0.2f;
private final int bufferSize;
private Thread recognizerThread;
private final Handler mainHandler = new Handler(Looper.getMainLooper());
/**
* Creates speech service.
**/
public SpeechStreamService(Recognizer recognizer, InputStream inputStream, float sampleRate) {
this.recognizer = recognizer;
this.sampleRate = (int) sampleRate;
this.inputStream = inputStream;
bufferSize = Math.round(this.sampleRate * BUFFER_SIZE_SECONDS * 2);
}
/**
* Starts recognition. Does nothing if recognition is active.
*
* @return true if recognition was actually started
*/
public boolean start(RecognitionListener listener) {
if (null != recognizerThread)
return false;
recognizerThread = new RecognizerThread(listener);
recognizerThread.start();
return true;
}
/**
* Starts recognition. After specified timeout listening stops and the
* endOfSpeech signals about that. Does nothing if recognition is active.
* <p>
* timeout - timeout in milliseconds to listen.
*
* @return true if recognition was actually started
*/
public boolean start(RecognitionListener listener, int timeout) {
if (null != recognizerThread)
return false;
recognizerThread = new RecognizerThread(listener, timeout);
recognizerThread.start();
return true;
}
/**
* Stops recognition. All listeners should receive final result if there is
* any. Does nothing if recognition is not active.
*
* @return true if recognition was actually stopped
*/
public boolean stop() {
if (null == recognizerThread)
return false;
try {
recognizerThread.interrupt();
recognizerThread.join();
} catch (InterruptedException e) {
// Restore the interrupted status.
Thread.currentThread().interrupt();
}
recognizerThread = null;
return true;
}
private final class RecognizerThread extends Thread {
private int remainingSamples;
private final int timeoutSamples;
private final static int NO_TIMEOUT = -1;
RecognitionListener listener;
public RecognizerThread(RecognitionListener listener, int timeout) {
this.listener = listener;
if (timeout != NO_TIMEOUT)
this.timeoutSamples = timeout * sampleRate / 1000;
else
this.timeoutSamples = NO_TIMEOUT;
this.remainingSamples = this.timeoutSamples;
}
public RecognizerThread(RecognitionListener listener) {
this(listener, NO_TIMEOUT);
}
@Override
public void run() {
byte[] buffer = new byte[bufferSize];
while (!interrupted()
&& ((timeoutSamples == NO_TIMEOUT) || (remainingSamples > 0))) {
try {
int nread = inputStream.read(buffer, 0, buffer.length);
if (nread < 0) {
break;
} else {
boolean isSilence = recognizer.acceptWaveForm(buffer, nread);
if (isSilence) {
final String result = recognizer.getResult();
mainHandler.post(() -> listener.onResult(result));
} else {
final String partialResult = recognizer.getPartialResult();
mainHandler.post(() -> listener.onPartialResult(partialResult));
}
}
if (timeoutSamples != NO_TIMEOUT) {
remainingSamples = remainingSamples - nread;
}
} catch (IOException e) {
mainHandler.post(() -> listener.onError(e));
}
}
// If we met timeout signal that speech ended
if (timeoutSamples != NO_TIMEOUT && remainingSamples <= 0) {
mainHandler.post(() -> listener.onTimeout());
} else {
final String finalResult = recognizer.getFinalResult();
mainHandler.post(() -> listener.onFinalResult(finalResult));
}
}
}
}
@@ -1,150 +0,0 @@
// Copyright 2019 Alpha Cephei Inc.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
package org.vosk.android;
import android.content.Context;
import android.content.res.AssetManager;
import android.os.Environment;
import android.os.Handler;
import android.os.Looper;
import android.util.Log;
import org.vosk.Model;
import java.io.BufferedReader;
import java.io.File;
import java.io.FileInputStream;
import java.io.FileNotFoundException;
import java.io.FileOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.io.OutputStream;
import java.util.concurrent.Executor;
import java.util.concurrent.Executors;
/**
* Provides utility methods to sync model files to external storage to allow
* C++ code access them. Relies on file named "uuid" to track updates.
*/
public class StorageService {
protected static final String TAG = StorageService.class.getSimpleName();
public interface Callback<R> {
void onComplete(R result);
}
public static void unpack(Context context, String sourcePath, final String targetPath, final Callback<Model> completeCallback, final Callback<IOException> errorCallback) {
Executor executor = Executors.newSingleThreadExecutor(); // change according to your requirements
Handler handler = new Handler(Looper.getMainLooper());
executor.execute(() -> {
try {
final String outputPath = sync(context, sourcePath, targetPath);
Model model = new Model(outputPath);
handler.post(() -> completeCallback.onComplete(model));
} catch (final IOException e) {
handler.post(() -> errorCallback.onComplete(e));
}
});
}
public static String sync(Context context, String sourcePath, String targetPath) throws IOException {
AssetManager assetManager = context.getAssets();
File externalFilesDir = context.getExternalFilesDir(null);
if (externalFilesDir == null) {
throw new IOException("cannot get external files dir, "
+ "external storage state is " + Environment.getExternalStorageState());
}
File targetDir = new File(externalFilesDir, targetPath);
String resultPath = new File(targetDir, sourcePath).getAbsolutePath();
String sourceUUID = readLine(assetManager.open(sourcePath + "/uuid"));
try {
String targetUUID = readLine(new FileInputStream(new File(targetDir, sourcePath + "/uuid")));
if (targetUUID.equals(sourceUUID)) return resultPath;
} catch (FileNotFoundException e) {
// ignore
}
deleteContents(targetDir);
copyAssets(assetManager, sourcePath, targetDir);
// Copy uuid
copyFile(assetManager, sourcePath + "/uuid", targetDir);
return resultPath;
}
private static String readLine(InputStream is) throws IOException {
return new BufferedReader(new InputStreamReader(is)).readLine();
}
private static boolean deleteContents(File dir) {
File[] files = dir.listFiles();
boolean success = true;
if (files != null) {
for (File file : files) {
if (file.isDirectory()) {
success &= deleteContents(file);
}
if (!file.delete()) {
success = false;
}
}
}
return success;
}
private static void copyAssets(AssetManager assetManager, String path, File outPath) throws IOException {
String[] assets = assetManager.list(path);
if (assets == null) {
return;
}
if (assets.length == 0) {
if (!path.endsWith("uuid"))
copyFile(assetManager, path, outPath);
} else {
File dir = new File(outPath, path);
if (!dir.exists()) {
Log.v(TAG, "Making directory " + dir.getAbsolutePath());
if (!dir.mkdirs()) {
Log.v(TAG, "Failed to create directory " + dir.getAbsolutePath());
}
}
for (String asset : assets) {
copyAssets(assetManager, path + "/" + asset, outPath);
}
}
}
private static void copyFile(AssetManager assetManager, String fileName, File outPath) throws IOException {
InputStream in;
Log.v(TAG, "Copy " + fileName + " to " + outPath);
in = assetManager.open(fileName);
OutputStream out = new FileOutputStream(outPath + "/" + fileName);
byte[] buffer = new byte[4000];
int read;
while ((read = in.read(buffer)) != -1) {
out.write(buffer, 0, read);
}
in.close();
out.close();
}
}
-47
View File
@@ -1,47 +0,0 @@
def archiveName = "vosk-model-en"
def pomName = "Vosk English Model"
def pomDescription = "Small English model for Android"
android {
compileSdkVersion 29
defaultConfig {
minSdkVersion 21
targetSdkVersion 29
versionCode 6
versionName = version
archivesBaseName = archiveName
}
buildFeatures {
buildConfig = false
}
sourceSets {
main {
assets.srcDirs += "$buildDir/generated/assets"
}
}
}
tasks.register('genUUID') {
def uuid = UUID.randomUUID().toString()
def odir = file("$buildDir/generated/assets/model-en-us")
def ofile = file("$odir/uuid")
doLast {
mkdir odir
ofile.text = uuid
}
}
preBuild.dependsOn(genUUID)
publishing {
publications {
aar(MavenPublication) {
artifactId = archiveName
artifact("$buildDir/outputs/aar/$archiveName-release.aar")
pom {
name = pomName
description = pomDescription
}
}
}
}
@@ -1,3 +0,0 @@
<manifest xmlns:android="http://schemas.android.com/apk/res/android"
package="org.vosk.model.en">
</manifest>
@@ -1,7 +0,0 @@
US English model for mobile Vosk applications
Copyright 2020 Alpha Cephei Inc
Accuracy: 10.38 (tedlium test) 9.85 (librispeech test-clean)
Speed: 0.11xRT (desktop)
Latency: 0.15s (right context)
-1
View File
@@ -1 +0,0 @@
include ':lib', ':model-en'
@@ -1,3 +1,3 @@
<?xml version="1.0" encoding="utf-8"?>
<manifest xmlns:android="http://schemas.android.com/apk/res/android" package="org.vosk">
<manifest xmlns:android="http://schemas.android.com/apk/res/android" package="org.kaldi">
</manifest>
+262
View File
@@ -0,0 +1,262 @@
// Copyright 2019 Alpha Cephei Inc.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
package org.kaldi;
import java.io.BufferedReader;
import java.io.File;
import java.io.FileInputStream;
import java.io.FileOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.io.OutputStream;
import java.io.PrintWriter;
import java.io.Reader;
import java.util.ArrayDeque;
import java.util.ArrayList;
import java.util.Collection;
import java.util.Collections;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.Queue;
import android.content.Context;
import android.content.res.AssetManager;
import android.os.Environment;
import android.util.Log;
/**
* Provides utility methods to keep asset files to external storage to allow
* further JNI code access assets from a filesystem.
*
* There must be special file {@value #ASSET_LIST_NAME} among the application
* assets containing relative paths of assets to synchronize. If the
* corresponding path does not exist on the external storage it is copied. If
* the path exists checksums are compared and the asset is copied only if there
* is a mismatch. Checksum is stored in a separate asset with the name that
* consists of the original name and a suffix that depends on the checksum
* algorithm (e.g. MD5). Checksum files are copied along with the corresponding
* asset files.
*
* @author Alexander Solovets
*/
public class Assets {
protected static final String TAG = Assets.class.getSimpleName();
public static final String ASSET_LIST_NAME = "assets.lst";
public static final String SYNC_DIR = "sync";
public static final String HASH_EXT = ".md5";
private final AssetManager assetManager;
private final File externalDir;
/**
* Creates new instance for asset synchronization
*
* @param context
* application context
*
* @throws IOException
* if the directory does not exist
*
* @see android.content.Context#getExternalFilesDir
* @see android.os.Environment#getExternalStorageState
*/
public Assets(Context context) throws IOException {
File appDir = context.getExternalFilesDir(null);
if (null == appDir)
throw new IOException("cannot get external files dir, "
+ "external storage state is " + Environment.getExternalStorageState());
externalDir = new File(appDir, SYNC_DIR);
assetManager = context.getAssets();
}
/**
* Creates new instance with specified destination for assets
*
* @param context
* application context to retrieve the assets
* @param path
* path to sync the files
*/
public Assets(Context context, String dest) {
externalDir = new File(dest);
assetManager = context.getAssets();
}
/**
* Returns destination path on external storage where assets are copied.
*
* @return path to application directory or null if it does not exists
*/
public File getExternalDir() {
return externalDir;
}
/**
* Returns the map of asset paths to the files checksums.
*
* @return path to the root of resources directory on external storage
* @throws IOException
* if an I/O error occurs or "assets.lst" is missing
*/
public Map<String, String> getItems() throws IOException {
Map<String, String> items = new HashMap<String, String>();
for (String path : readLines(openAsset(ASSET_LIST_NAME))) {
Reader reader = new InputStreamReader(openAsset(path + HASH_EXT));
items.put(path, new BufferedReader(reader).readLine());
}
return items;
}
/**
* Returns path to hash mappings for the previously copied files. This
* method can be used to find out assets which must be updated.
*/
public Map<String, String> getExternalItems() {
try {
Map<String, String> items = new HashMap<String, String>();
File assetFile = new File(externalDir, ASSET_LIST_NAME);
for (String line : readLines(new FileInputStream(assetFile))) {
String[] fields = line.split(" ");
items.put(fields[0], fields[1]);
}
return items;
} catch (IOException e) {
return Collections.emptyMap();
}
}
/**
* In case you want to create more smart sync implementation, this method
* returns the list of items which must be synchronized.
*/
public Collection<String> getItemsToCopy(String path) throws IOException {
Collection<String> items = new ArrayList<String>();
Queue<String> queue = new ArrayDeque<String>();
queue.offer(path);
while (!queue.isEmpty()) {
path = queue.poll();
String[] list = assetManager.list(path);
for (String nested : list)
queue.offer(nested);
if (list.length == 0)
items.add(path);
}
return items;
}
private List<String> readLines(InputStream source) throws IOException {
List<String> lines = new ArrayList<String>();
BufferedReader br = new BufferedReader(new InputStreamReader(source));
String line;
while (null != (line = br.readLine()))
lines.add(line);
return lines;
}
private InputStream openAsset(String asset) throws IOException {
return assetManager.open(new File(SYNC_DIR, asset).getPath());
}
/**
* Saves the list of synchronized items. The list is stored as a two-column
* space-separated list of items in a text file. The file is located at the
* root of synchronization directory in the external storage.
*
* @param items
* the items
* @throws IOException
* if an I/O error occurs
*/
public void updateItemList(Map<String, String> items) throws IOException {
File assetListFile = new File(externalDir, ASSET_LIST_NAME);
PrintWriter pw = new PrintWriter(new FileOutputStream(assetListFile));
for (Map.Entry<String, String> entry : items.entrySet())
pw.format("%s %s\n", entry.getKey(), entry.getValue());
pw.close();
}
/**
* Copies raw asset resource to external storage of the device.
*
* @param path
* path of the asset to copy
* @throws IOException
* if an I/O error occurs
*/
public File copy(String asset) throws IOException {
InputStream source = openAsset(asset);
File destinationFile = new File(externalDir, asset);
destinationFile.getParentFile().mkdirs();
OutputStream destination = new FileOutputStream(destinationFile);
byte[] buffer = new byte[1024];
int nread;
while ((nread = source.read(buffer)) != -1) {
if (nread == 0) {
nread = source.read();
if (nread < 0)
break;
destination.write(nread);
continue;
}
destination.write(buffer, 0, nread);
}
destination.close();
return destinationFile;
}
/**
* Performs the sync of assets in the application and on the external
* storage
*
* @return The folder on external storage with data
* @throws IOException
*/
public File syncAssets() throws IOException {
Collection<String> newItems = new ArrayList<String>();
Collection<String> unusedItems = new ArrayList<String>();
Map<String, String> items = getItems();
Map<String, String> externalItems = getExternalItems();
for (String path : items.keySet()) {
if (!items.get(path).equals(externalItems.get(path))
|| !(new File(externalDir, path).exists()))
newItems.add(path);
}
unusedItems.addAll(externalItems.keySet());
unusedItems.removeAll(items.keySet());
for (String path : newItems) {
File file = copy(path);
}
for (String path : unusedItems) {
File file = new File(externalDir, path);
file.delete();
}
updateItemList(items);
return externalDir;
}
}
@@ -12,35 +12,28 @@
// See the License for the specific language governing permissions and
// limitations under the License.
package org.vosk.android;
package org.kaldi;
/**
* Interface to receive recognition results
*/
/** Interface to receive recognition results */
public interface RecognitionListener {
/**
* Called when partial recognition result is available.
*/
void onPartialResult(String hypothesis);
public void onPartialResult(String hypothesis);
/**
* Called after silence occured.
* Called after the recognition is ended.
*/
void onResult(String hypothesis);
/**
* Called after stream end.
*/
void onFinalResult(String hypothesis);
public void onResult(String hypothesis);
/**
* Called when an error occurs.
*/
void onError(Exception exception);
public void onError(Exception exception);
/**
* Called after timeout expired
*/
void onTimeout();
public void onTimeout();
}
@@ -0,0 +1,302 @@
// Copyright 2019 Alpha Cephei Inc.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
package org.kaldi;
import static java.lang.String.format;
import java.io.File;
import java.io.IOException;
import java.util.Collection;
import java.util.HashSet;
import android.media.AudioFormat;
import android.media.AudioRecord;
import android.media.MediaRecorder.AudioSource;
import android.os.Handler;
import android.os.Looper;
import android.util.Log;
/**
* Main class to access recognizer functions. After configuration this class
* starts a listener thread which records the data and recognizes it using
* VOSK engine. Recognition events are passed to a client using
* {@link RecognitionListener}
*
*/
public class SpeechRecognizer {
protected static final String TAG = SpeechRecognizer.class.getSimpleName();
private final KaldiRecognizer recognizer;
private final int sampleRate;
private final static float BUFFER_SIZE_SECONDS = 0.4f;
private int bufferSize;
private final AudioRecord recorder;
private Thread recognizerThread;
private final Handler mainHandler = new Handler(Looper.getMainLooper());
private final Collection<RecognitionListener> listeners = new HashSet<RecognitionListener>();
/**
* Creates speech recognizer. Recognizer holds the AudioRecord object, so you
* need to call {@link release} in order to properly finalize it.
*
* @throws IOException thrown if audio recorder can not be created for some reason.
*/
public SpeechRecognizer(Model model) throws IOException {
recognizer = new KaldiRecognizer(model, 16000.0f);
sampleRate = 16000;
bufferSize = Math.round(sampleRate * BUFFER_SIZE_SECONDS);
recorder = new AudioRecord(
AudioSource.VOICE_RECOGNITION, sampleRate,
AudioFormat.CHANNEL_IN_MONO,
AudioFormat.ENCODING_PCM_16BIT, bufferSize * 2);
if (recorder.getState() == AudioRecord.STATE_UNINITIALIZED) {
recorder.release();
throw new IOException(
"Failed to initialize recorder. Microphone might be already in use.");
}
}
public SpeechRecognizer(Model model, SpkModel spkModel) throws IOException {
recognizer = new KaldiRecognizer(model, spkModel, 16000.0f);
sampleRate = 16000;
bufferSize = Math.round(sampleRate * BUFFER_SIZE_SECONDS);
recorder = new AudioRecord(
AudioSource.VOICE_RECOGNITION, sampleRate,
AudioFormat.CHANNEL_IN_MONO,
AudioFormat.ENCODING_PCM_16BIT, bufferSize * 2);
if (recorder.getState() == AudioRecord.STATE_UNINITIALIZED) {
recorder.release();
throw new IOException(
"Failed to initialize recorder. Microphone might be already in use.");
}
}
/**
* Adds listener.
*/
public void addListener(RecognitionListener listener) {
synchronized (listeners) {
listeners.add(listener);
}
}
/**
* Removes listener.
*/
public void removeListener(RecognitionListener listener) {
synchronized (listeners) {
listeners.remove(listener);
}
}
/**
* Starts recognition. Does nothing if recognition is active.
*
* @return true if recognition was actually started
*/
public boolean startListening() {
if (null != recognizerThread)
return false;
recognizerThread = new RecognizerThread();
recognizerThread.start();
return true;
}
/**
* Starts recognition. After specified timeout listening stops and the
* endOfSpeech signals about that. Does nothing if recognition is active.
*
* @timeout - timeout in milliseconds to listen.
*
* @return true if recognition was actually started
*/
public boolean startListening(int timeout) {
if (null != recognizerThread)
return false;
recognizerThread = new RecognizerThread(timeout);
recognizerThread.start();
return true;
}
private boolean stopRecognizerThread() {
if (null == recognizerThread)
return false;
try {
recognizerThread.interrupt();
recognizerThread.join();
} catch (InterruptedException e) {
// Restore the interrupted status.
Thread.currentThread().interrupt();
}
recognizerThread = null;
return true;
}
/**
* Stops recognition. All listeners should receive final result if there is
* any. Does nothing if recognition is not active.
*
* @return true if recognition was actually stopped
*/
public boolean stop() {
boolean result = stopRecognizerThread();
if (result) {
mainHandler.post(new ResultEvent(recognizer.Result(), true));
}
return result;
}
/**
* Cancels recognition. Listeners do not receive final result. Does nothing
* if recognition is not active.
*
* @return true if recognition was actually canceled
*/
public boolean cancel() {
boolean result = stopRecognizerThread();
recognizer.Result(); // Reset recognizer state
return result;
}
/**
* Shutdown the recognizer and release the recorder
*/
public void shutdown() {
recorder.release();
}
private final class RecognizerThread extends Thread {
private int remainingSamples;
private int timeoutSamples;
private final static int NO_TIMEOUT = -1;
public RecognizerThread(int timeout) {
if (timeout != NO_TIMEOUT)
this.timeoutSamples = timeout * sampleRate / 1000;
else
this.timeoutSamples = NO_TIMEOUT;
this.remainingSamples = this.timeoutSamples;
}
public RecognizerThread() {
this(NO_TIMEOUT);
}
@Override
public void run() {
recorder.startRecording();
if (recorder.getRecordingState() == AudioRecord.RECORDSTATE_STOPPED) {
recorder.stop();
IOException ioe = new IOException(
"Failed to start recording. Microphone might be already in use.");
mainHandler.post(new OnErrorEvent(ioe));
return;
}
short[] buffer = new short[bufferSize];
while (!interrupted()
&& ((timeoutSamples == NO_TIMEOUT) || (remainingSamples > 0))) {
int nread = recorder.read(buffer, 0, buffer.length);
if (nread < 0) {
throw new RuntimeException("error reading audio buffer");
} else {
boolean isFinal = recognizer.AcceptWaveform(buffer, nread);
if (isFinal) {
mainHandler.post(new ResultEvent(recognizer.Result(), true));
} else {
mainHandler.post(new ResultEvent(recognizer.PartialResult(), false));
}
}
if (timeoutSamples != NO_TIMEOUT) {
remainingSamples = remainingSamples - nread;
}
}
recorder.stop();
// Remove all pending notifications.
mainHandler.removeCallbacksAndMessages(null);
// If we met timeout signal that speech ended
if (timeoutSamples != NO_TIMEOUT && remainingSamples <= 0) {
mainHandler.post(new TimeoutEvent());
}
}
}
private abstract class RecognitionEvent implements Runnable {
public void run() {
RecognitionListener[] emptyArray = new RecognitionListener[0];
for (RecognitionListener listener : listeners.toArray(emptyArray))
execute(listener);
}
protected abstract void execute(RecognitionListener listener);
}
private class ResultEvent extends RecognitionEvent {
protected final String hypothesis;
private final boolean finalResult;
ResultEvent(String hypothesis, boolean finalResult) {
this.hypothesis = hypothesis;
this.finalResult = finalResult;
}
@Override
protected void execute(RecognitionListener listener) {
if (finalResult)
listener.onResult(hypothesis);
else
listener.onPartialResult(hypothesis);
}
}
private class OnErrorEvent extends RecognitionEvent {
private final Exception exception;
OnErrorEvent(Exception exception) {
this.exception = exception;
}
@Override
protected void execute(RecognitionListener listener) {
listener.onError(exception);
}
}
private class TimeoutEvent extends RecognitionEvent {
@Override
protected void execute(RecognitionListener listener) {
listener.onTimeout();
}
}
}
-16
View File
@@ -1,16 +0,0 @@
CFLAGS=-I../src
LDFLAGS=-L../src -lvosk -ldl -lpthread -Wl,-rpath=../src
all: test_vosk test_vosk_speaker
test_vosk: test_vosk.o
g++ $^ -o $@ $(LDFLAGS)
test_vosk_speaker: test_vosk_speaker.o
g++ $^ -o $@ $(LDFLAGS)
%.o: %.c
g++ $(CFLAGS) -c -o $@ $<
clean:
rm -f *.o *.a test_vosk test_vosk_speaker
-29
View File
@@ -1,29 +0,0 @@
#include <vosk_api.h>
#include <stdio.h>
int main() {
FILE *wavin;
char buf[3200];
int nread, final;
VoskModel *model = vosk_model_new("model");
VoskRecognizer *recognizer = vosk_recognizer_new(model, 16000.0);
wavin = fopen("test.wav", "rb");
fseek(wavin, 44, SEEK_SET);
while (!feof(wavin)) {
nread = fread(buf, 1, sizeof(buf), wavin);
final = vosk_recognizer_accept_waveform(recognizer, buf, nread);
if (final) {
printf("%s\n", vosk_recognizer_result(recognizer));
} else {
printf("%s\n", vosk_recognizer_partial_result(recognizer));
}
}
printf("%s\n", vosk_recognizer_final_result(recognizer));
vosk_recognizer_free(recognizer);
vosk_model_free(model);
fclose(wavin);
return 0;
}
-30
View File
@@ -1,30 +0,0 @@
#include <vosk_api.h>
#include <stdio.h>
int main() {
FILE *wavin;
char buf[3200];
int nread, final;
VoskModel *model = vosk_model_new("model");
VoskSpkModel *spk_model = vosk_spk_model_new("spk-model");
VoskRecognizer *recognizer = vosk_recognizer_new_spk(model, 16000.0, spk_model);
wavin = fopen("test.wav", "rb");
fseek(wavin, 44, SEEK_SET);
while (!feof(wavin)) {
nread = fread(buf, 1, sizeof(buf), wavin);
final = vosk_recognizer_accept_waveform(recognizer, buf, nread);
if (final) {
printf("%s\n", vosk_recognizer_result(recognizer));
} else {
printf("%s\n", vosk_recognizer_partial_result(recognizer));
}
}
printf("%s\n", vosk_recognizer_final_result(recognizer));
vosk_recognizer_free(recognizer);
vosk_spk_model_free(spk_model);
vosk_model_free(model);
return 0;
}
+55
View File
@@ -0,0 +1,55 @@
KALDI_ROOT ?= $(HOME)/kaldi
CFLAGS := -std=c++11 -g -O2 -DPIC -fPIC -Wno-unused-function
CPPFLAGS := -I$(KALDI_ROOT)/src -I$(KALDI_ROOT)/tools/openfst/include -I../src -DFST_NO_DYNAMIC_LINKING
KALDI_LIBS = \
${KALDI_ROOT}/src/online2/kaldi-online2.a \
${KALDI_ROOT}/src/decoder/kaldi-decoder.a \
${KALDI_ROOT}/src/ivector/kaldi-ivector.a \
${KALDI_ROOT}/src/gmm/kaldi-gmm.a \
${KALDI_ROOT}/src/nnet3/kaldi-nnet3.a \
${KALDI_ROOT}/src/tree/kaldi-tree.a \
${KALDI_ROOT}/src/feat/kaldi-feat.a \
${KALDI_ROOT}/src/lat/kaldi-lat.a \
${KALDI_ROOT}/src/lm/kaldi-lm.a \
${KALDI_ROOT}/src/hmm/kaldi-hmm.a \
${KALDI_ROOT}/src/transform/kaldi-transform.a \
${KALDI_ROOT}/src/cudamatrix/kaldi-cudamatrix.a \
${KALDI_ROOT}/src/matrix/kaldi-matrix.a \
${KALDI_ROOT}/src/fstext/kaldi-fstext.a \
${KALDI_ROOT}/src/util/kaldi-util.a \
${KALDI_ROOT}/src/base/kaldi-base.a \
${KALDI_ROOT}/tools/openfst/lib/libfst.a \
${KALDI_ROOT}/tools/openfst/lib/libfstngram.a \
${KALDI_ROOT}/tools/OpenBLAS/libopenblas.a \
-lgfortran -lstdc++
all: test.exe
test.exe: libkaldiwrap.so test.cs
mcs test.cs gen/*.cs
VOSK_SOURCES = \
vosk_wrap.c \
../src/kaldi_recognizer.cc \
../src/kaldi_recognizer.h \
../src/model.cc \
../src/model.h \
../src/spk_model.cc \
../src/spk_model.h \
../src/vosk_api.cc \
../src/vosk_api.h
libkaldiwrap.so: $(VOSK_SOURCES)
$(CXX) -fpermissive $(CFLAGS) $(CPPFLAGS) -shared -o $@ $(VOSK_SOURCES) $(KALDI_LIBS)
vosk_wrap.c: ../src/vosk.i
mkdir -p gen
swig -csharp -DSWIG_CSHARP_NO_EXCEPTION_HELPER -dllimport "libkaldiwrap" \
-namespace "Kaldi" -outdir gen -o vosk_wrap.c ../src/vosk.i
run: test.exe
mono test.exe
clean:
$(RM) *.so vosk_wrap.c *.o gen/*.cs test.exe
-12
View File
@@ -1,12 +0,0 @@
This is a nuget-based wrapper for libvosk library
See demo folder for example how to use the library. You can simply run
"dotnet run" to run the demo. Make sure you unpacked the model and the
test file.
See the nuget folder for the sources of the wrapper. Run build.sh to
build nuget package.
Note we only support win64 and linux64 for now. No support for win32
since it is a little bit painful to load the libraries depending on
architecture. In theory we can add OSX some time or even Android.
-17
View File
@@ -1,17 +0,0 @@
<Project Sdk="Microsoft.NET.Sdk">
<PropertyGroup>
<OutputType>Exe</OutputType>
<TargetFramework>net5.0</TargetFramework>
<RootNamespace>VoskDemo</RootNamespace>
</PropertyGroup>
<PropertyGroup>
<RestoreSources>$(RestoreSources);../nuget</RestoreSources>
</PropertyGroup>
<ItemGroup>
<PackageReference Include="Vosk" Version="0.3.30" />
</ItemGroup>
</Project>
-30
View File
@@ -1,30 +0,0 @@
<?xml version="1.0"?>
<package>
<metadata>
<id>Vosk</id>
<version>0.3.30</version>
<authors>Alpha Cephei Inc</authors>
<owners>Alpha Cephei Inc</owners>
<license type="expression">Apache-2.0</license>
<projectUrl>https://alphacephei.com/vosk/</projectUrl>
<requireLicenseAcceptance>false</requireLicenseAcceptance>
<description>Vosk is an offline open source speech recognition toolkit. It enables speech recognition models for 16 languages and dialects - English, Indian English, German, French, Spanish, Portuguese, Chinese, Russian, Turkish, Vietnamese, Italian, Dutch, Catalan, Arabic, Greek, Farsi.
Vosk models are small (50 Mb) but provide continuous large vocabulary transcription, zero-latency response with streaming API, reconfigurable vocabulary and speaker identification.
Speech recognition bindings implemented for various programming languages like Python, Java, Node.JS, C#, C++ and others.
Vosk supplies speech recognition for chatbots, smart home appliances, virtual assistants. It can also create subtitles for movies, transcription for lectures and interviews.
Vosk scales from small devices like Raspberry Pi or Android smartphone to big clusters.</description>
<releaseNotes>See for details https://github.com/alphacep/vosk-api/releases</releaseNotes>
<copyright>Copyright 2020 Alpha Cephei Inc</copyright>
<tags>speech recognition voice stt asr speech-to-text ai offline privacy</tags>
<dependencies>
<group targetFramework=".NETStandard2.0"/>
</dependencies>
</metadata>
<files>
<file src="**" exclude="src/*.cs;build.sh;**/.keep-me;*.nupkg" />
</files>
</package>
-2
View File
@@ -1,2 +0,0 @@
mcs -out:lib/netstandard2.0/Vosk.dll -target:library src/*.cs
nuget pack
-11
View File
@@ -1,11 +0,0 @@
<Project xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
<ItemGroup>
<NativeLibs Include="$(MSBuildThisFileDirectory)\lib\linux-x64\*.so" Condition="'$([MSBuild]::IsOsPlatform(Linux))'" />
<NativeLibs Include="$(MSBuildThisFileDirectory)\lib\win-x64\*.dll" Condition="'$([MSBuild]::IsOsPlatform(Windows))'" />
<NativeLibs Include="$(MSBuildThisFileDirectory)\lib\osx-x64\*.dylib" Condition="'$([MSBuild]::IsOsPlatform(OSX))'" />
<None Include="@(NativeLibs)">
<Link>%(FileName)%(Extension)</Link>
<CopyToOutputDirectory>PreserveNewest</CopyToOutputDirectory>
</None>
</ItemGroup>
</Project>
-41
View File
@@ -1,41 +0,0 @@
namespace Vosk {
public class Model : global::System.IDisposable {
private global::System.Runtime.InteropServices.HandleRef handle;
internal Model(global::System.IntPtr cPtr) {
handle = new global::System.Runtime.InteropServices.HandleRef(this, cPtr);
}
internal static global::System.Runtime.InteropServices.HandleRef getCPtr(Model obj) {
return (obj == null) ? new global::System.Runtime.InteropServices.HandleRef(null, global::System.IntPtr.Zero) : obj.handle;
}
~Model() {
Dispose(false);
}
public void Dispose() {
Dispose(true);
global::System.GC.SuppressFinalize(this);
}
protected virtual void Dispose(bool disposing) {
lock(this) {
if (handle.Handle != global::System.IntPtr.Zero) {
VoskPINVOKE.delete_Model(handle);
handle = new global::System.Runtime.InteropServices.HandleRef(null, global::System.IntPtr.Zero);
}
}
}
public Model(string model_path) : this(VoskPINVOKE.new_Model(model_path)) {
}
public int vosk_model_find_word(string word) {
return VoskPINVOKE.Model_vosk_model_find_word(handle, word);
}
}
}
-37
View File
@@ -1,37 +0,0 @@
namespace Vosk {
public class SpkModel : global::System.IDisposable {
private global::System.Runtime.InteropServices.HandleRef handle;
internal SpkModel(global::System.IntPtr cPtr) {
handle = new global::System.Runtime.InteropServices.HandleRef(this, cPtr);
}
internal static global::System.Runtime.InteropServices.HandleRef getCPtr(SpkModel obj) {
return (obj == null) ? new global::System.Runtime.InteropServices.HandleRef(null, global::System.IntPtr.Zero) : obj.handle;
}
~SpkModel() {
Dispose(false);
}
public void Dispose() {
Dispose(true);
global::System.GC.SuppressFinalize(this);
}
protected virtual void Dispose(bool disposing) {
lock(this) {
if (handle.Handle != global::System.IntPtr.Zero) {
VoskPINVOKE.delete_SpkModel(handle);
handle = new global::System.Runtime.InteropServices.HandleRef(null, global::System.IntPtr.Zero);
}
}
}
public SpkModel(string model_path) : this(VoskPINVOKE.new_SpkModel(model_path)) {
}
}
}
-17
View File
@@ -1,17 +0,0 @@
namespace Vosk {
public class Vosk {
public static void SetLogLevel(int level) {
VoskPINVOKE.SetLogLevel(level);
}
public static void GpuInit() {
VoskPINVOKE.GpuInit();
}
public static void GpuThreadInit() {
VoskPINVOKE.GpuThreadInit();
}
}
}
-75
View File
@@ -1,75 +0,0 @@
namespace Vosk {
class VoskPINVOKE {
static VoskPINVOKE() {
}
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_model_new")]
public static extern global::System.IntPtr new_Model(string jarg1);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_model_free")]
public static extern void delete_Model(global::System.Runtime.InteropServices.HandleRef jarg1);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_model_find_word")]
public static extern int Model_vosk_model_find_word(global::System.Runtime.InteropServices.HandleRef jarg1, string jarg2);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_spk_model_new")]
public static extern global::System.IntPtr new_SpkModel(string jarg1);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_spk_model_free")]
public static extern void delete_SpkModel(global::System.Runtime.InteropServices.HandleRef jarg1);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_new")]
public static extern global::System.IntPtr new_VoskRecognizer(global::System.Runtime.InteropServices.HandleRef jarg1, float jarg2);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_new_spk")]
public static extern global::System.IntPtr new_VoskRecognizerSpk(global::System.Runtime.InteropServices.HandleRef jarg1, float jarg2, global::System.Runtime.InteropServices.HandleRef jarg3);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_new_grm")]
public static extern global::System.IntPtr new_VoskRecognizerGrm(global::System.Runtime.InteropServices.HandleRef jarg1, float jarg2, string jarg3);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_free")]
public static extern void delete_VoskRecognizer(global::System.Runtime.InteropServices.HandleRef jarg1);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_set_max_alternatives")]
public static extern void VoskRecognizer_SetMaxAlternatives(global::System.Runtime.InteropServices.HandleRef jarg1, int jarg2);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_set_words")]
public static extern void VoskRecognizer_SetWords(global::System.Runtime.InteropServices.HandleRef jarg1, int jarg2);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_set_spk_model")]
public static extern void VoskRecognizer_SetSpkModel(global::System.Runtime.InteropServices.HandleRef jarg1, global::System.Runtime.InteropServices.HandleRef jarg2);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_accept_waveform")]
public static extern bool VoskRecognizer_AcceptWaveform(global::System.Runtime.InteropServices.HandleRef jarg1, [global::System.Runtime.InteropServices.In, global::System.Runtime.InteropServices.MarshalAs(global::System.Runtime.InteropServices.UnmanagedType.LPArray)]byte[] jarg2, int jarg3);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_accept_waveform_s")]
public static extern bool VoskRecognizer_AcceptWaveformShort(global::System.Runtime.InteropServices.HandleRef jarg1, [global::System.Runtime.InteropServices.In, global::System.Runtime.InteropServices.MarshalAs(global::System.Runtime.InteropServices.UnmanagedType.LPArray)]short[] jarg2, int jarg3);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_accept_waveform_f")]
public static extern bool VoskRecognizer_AcceptWaveformFloat(global::System.Runtime.InteropServices.HandleRef jarg1, [global::System.Runtime.InteropServices.In, global::System.Runtime.InteropServices.MarshalAs(global::System.Runtime.InteropServices.UnmanagedType.LPArray)]float[] jarg2, int jarg3);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_result")]
public static extern global::System.IntPtr VoskRecognizer_Result(global::System.Runtime.InteropServices.HandleRef jarg1);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_partial_result")]
public static extern global::System.IntPtr VoskRecognizer_PartialResult(global::System.Runtime.InteropServices.HandleRef jarg1);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_final_result")]
public static extern global::System.IntPtr VoskRecognizer_FinalResult(global::System.Runtime.InteropServices.HandleRef jarg1);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_recognizer_reset")]
public static extern void VoskRecognizer_Reset(global::System.Runtime.InteropServices.HandleRef jarg1);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_set_log_level")]
public static extern void SetLogLevel(int jarg1);
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_gpu_init")]
public static extern void GpuInit();
[global::System.Runtime.InteropServices.DllImport("libvosk", EntryPoint="vosk_gpu_thread_init")]
public static extern void GpuThreadInit();
}
}
-92
View File
@@ -1,92 +0,0 @@
namespace Vosk {
public class VoskRecognizer : System.IDisposable {
private System.Runtime.InteropServices.HandleRef handle;
internal VoskRecognizer(System.IntPtr cPtr) {
handle = new System.Runtime.InteropServices.HandleRef(this, cPtr);
}
internal static System.Runtime.InteropServices.HandleRef getCPtr(VoskRecognizer obj) {
return (obj == null) ? new System.Runtime.InteropServices.HandleRef(null, System.IntPtr.Zero) : obj.handle;
}
~VoskRecognizer() {
Dispose(false);
}
public void Dispose() {
Dispose(true);
System.GC.SuppressFinalize(this);
}
protected virtual void Dispose(bool disposing) {
lock(this) {
if (handle.Handle != System.IntPtr.Zero) {
VoskPINVOKE.delete_VoskRecognizer(handle);
handle = new System.Runtime.InteropServices.HandleRef(null, System.IntPtr.Zero);
}
}
}
public VoskRecognizer(Model model, float sample_rate) : this(VoskPINVOKE.new_VoskRecognizer(Model.getCPtr(model), sample_rate)) {
}
public VoskRecognizer(Model model, float sample_rate, SpkModel spk_model) : this(VoskPINVOKE.new_VoskRecognizerSpk(Model.getCPtr(model), sample_rate, SpkModel.getCPtr(spk_model))) {
}
public VoskRecognizer(Model model, float sample_rate, string grammar) : this(VoskPINVOKE.new_VoskRecognizerGrm(Model.getCPtr(model), sample_rate, grammar)) {
}
public void SetMaxAlternatives(int max_alternatives) {
VoskPINVOKE.VoskRecognizer_SetMaxAlternatives(handle, max_alternatives);
}
public void SetWords(bool words) {
VoskPINVOKE.VoskRecognizer_SetWords(handle, words ? 1 : 0);
}
public void SetSpkModel(SpkModel spk_model) {
VoskPINVOKE.VoskRecognizer_SetSpkModel(handle, SpkModel.getCPtr(spk_model));
}
public bool AcceptWaveform(byte[] data, int len) {
return VoskPINVOKE.VoskRecognizer_AcceptWaveform(handle, data, len);
}
public bool AcceptWaveform(short[] sdata, int len) {
return VoskPINVOKE.VoskRecognizer_AcceptWaveformShort(handle, sdata, len);
}
public bool AcceptWaveform(float[] fdata, int len) {
return VoskPINVOKE.VoskRecognizer_AcceptWaveformFloat(handle, fdata, len);
}
private static string PtrToStringUTF8(System.IntPtr ptr) {
int len = 0;
while (System.Runtime.InteropServices.Marshal.ReadByte(ptr, len) != 0)
len++;
byte[] array = new byte[len];
System.Runtime.InteropServices.Marshal.Copy(ptr, array, 0, len);
return System.Text.Encoding.UTF8.GetString(array);
}
public string Result() {
return PtrToStringUTF8(VoskPINVOKE.VoskRecognizer_Result(handle));
}
public string PartialResult() {
return PtrToStringUTF8(VoskPINVOKE.VoskRecognizer_PartialResult(handle));
}
public string FinalResult() {
return PtrToStringUTF8(VoskPINVOKE.VoskRecognizer_FinalResult(handle));
}
public void Reset() {
VoskPINVOKE.VoskRecognizer_Reset(handle);
}
}
}
+11 -44
View File
@@ -1,15 +1,16 @@
using System;
using System.IO;
using Vosk;
using Kaldi;
public class VoskDemo
public class Test
{
public static void DemoBytes(Model model)
public static void Main()
{
// Demo byte buffer
VoskRecognizer rec = new VoskRecognizer(model, 16000.0f);
rec.SetMaxAlternatives(0);
rec.SetWords(true);
Vosk.SetLogLevel(0);
Model model = new Model("model");
KaldiRecognizer rec = new KaldiRecognizer(model, 16000.0f);
using(Stream source = File.OpenRead("test.wav")) {
byte[] buffer = new byte[4096];
int bytesRead;
@@ -22,12 +23,9 @@ public class VoskDemo
}
}
Console.WriteLine(rec.FinalResult());
}
public static void DemoFloats(Model model)
{
// Demo float array
VoskRecognizer rec = new VoskRecognizer(model, 16000.0f);
rec = new KaldiRecognizer(model, 16000.0f);
using(Stream source = File.OpenRead("test.wav")) {
byte[] buffer = new byte[4096];
int bytesRead;
@@ -38,6 +36,7 @@ public class VoskDemo
}
if (rec.AcceptWaveform(fbuffer, fbuffer.Length)) {
Console.WriteLine(rec.Result());
GC.Collect();
} else {
Console.WriteLine(rec.PartialResult());
}
@@ -45,36 +44,4 @@ public class VoskDemo
}
Console.WriteLine(rec.FinalResult());
}
public static void DemoSpeaker(Model model)
{
// Output speakers
SpkModel spkModel = new SpkModel("model-spk");
VoskRecognizer rec = new VoskRecognizer(model, 16000.0f);
rec.SetSpkModel(spkModel);
using(Stream source = File.OpenRead("test.wav")) {
byte[] buffer = new byte[4096];
int bytesRead;
while((bytesRead = source.Read(buffer, 0, buffer.Length)) > 0) {
if (rec.AcceptWaveform(buffer, bytesRead)) {
Console.WriteLine(rec.Result());
} else {
Console.WriteLine(rec.PartialResult());
}
}
}
Console.WriteLine(rec.FinalResult());
}
public static void Main()
{
// You can set to -1 to disable logging messages
Vosk.Vosk.SetLogLevel(0);
Model model = new Model("model");
DemoBytes(model);
DemoFloats(model);
DemoSpeaker(model);
}
}
+1
View File
@@ -0,0 +1 @@
See https://alphacephei.com/vosk/accuracy.html
+1
View File
@@ -0,0 +1 @@
See https://alphacephei.com/vosk/adaptation.html
+1
View File
@@ -0,0 +1 @@
See https://alphacephei.com/vosk/models.html
-36
View File
@@ -1,36 +0,0 @@
package main
import (
"flag"
"os"
".."
)
func main() {
var filename string
flag.StringVar(&filename, "f", "", "file to transcribe")
flag.Parse()
model, err := vosk.NewModel("model")
rec, err := vosk.NewRecognizer(model)
file, err := os.Open(filename)
if err != nil {
panic(err)
}
defer file.Close()
fileinfo, err := file.Stat()
if err != nil {
panic(err)
}
filesize := fileinfo.Size()
buffer := make([]byte, filesize)
_, err = file.Read(buffer)
if err != nil {
panic(err)
}
println(vosk.VoskFinalResult(rec, buffer))
}
-3
View File
@@ -1,3 +0,0 @@
module github.com/alphacep/vosk-api
go 1.16
-62
View File
@@ -1,62 +0,0 @@
package vosk
// #cgo CPPFLAGS: -I ${SRCDIR}/../src
// #cgo LDFLAGS: -L ${SRCDIR}/../src -lvosk -ldl -lpthread
// #include <stdlib.h>
// #include <vosk_api.h>
import "C"
// VoskModel contains a reference to the C VoskModel
type VoskModel struct {
model *C.struct_VoskModel
}
// VoskSpkModel contains a reference to the C VoskSpkModel
type VoskSpkModel struct {
spkModel *C.struct_VoskSpkModel
}
// VoskRecognizer contains a reference to the C VoskRecognizer
type VoskRecognizer struct {
rec *C.struct_VoskRecognizer
}
func VoskFinalResult(recognizer *VoskRecognizer, buffer []byte) string {
cbuf := C.CBytes(buffer)
defer C.free(cbuf)
_ = C.vosk_recognizer_accept_waveform(recognizer.rec, (*C.char)(cbuf), C.int(len(buffer)))
result := C.GoString(C.vosk_recognizer_final_result(recognizer.rec))
return result
}
// NewModel creates a new VoskModel instance
func NewModel(modelPath string) (*VoskModel, error) {
var internal *C.struct_VoskModel
internal = C.vosk_model_new(C.CString(modelPath))
model := &VoskModel{model: internal}
return model, nil
}
// NewRecognizer creates a new VoskRecognizer instance
func NewRecognizer(model *VoskModel) (*VoskRecognizer, error) {
var internal *C.struct_VoskRecognizer
internal = C.vosk_recognizer_new(model.model, 16000.0)
rec := &VoskRecognizer{rec: internal}
return rec, nil
}
func freeModel(model *VoskModel) {
C.vosk_model_free(model.model)
}
func freeRecognizer(recognizer *VoskRecognizer) {
C.vosk_recognizer_free(recognizer.rec)
}
// NewSpkModel creates a new VoskSpkModel instance
func NewSpkModel(spkModelPath string) (*VoskSpkModel, error) {
var internal *C.struct_VoskSpkModel
internal = C.vosk_spk_model_new(C.CString(spkModelPath))
spkModel := &VoskSpkModel{spkModel: internal}
return spkModel, nil
}
+93 -14
View File
@@ -16,12 +16,56 @@
9237523C240C642000DD6076 /* libkaldiwrap.a in Frameworks */ = {isa = PBXBuildFile; fileRef = 9237523A240C642000DD6076 /* libkaldiwrap.a */; };
92375244240C6DAF00DD6076 /* Accelerate.framework in Frameworks */ = {isa = PBXBuildFile; fileRef = 92375243240C6DAF00DD6076 /* Accelerate.framework */; };
92375246240C6DC900DD6076 /* libstdc++.tbd in Frameworks */ = {isa = PBXBuildFile; fileRef = 92375245240C6DC900DD6076 /* libstdc++.tbd */; };
92375266240C6EFE00DD6076 /* disambig_tid.int in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375248240C6E3D00DD6076 /* disambig_tid.int */; };
92375267240C6EFE00DD6076 /* final.mdl in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375249240C6E3D00DD6076 /* final.mdl */; };
92375268240C6EFE00DD6076 /* Gr.fst in CopyFiles */ = {isa = PBXBuildFile; fileRef = 9237524A240C6E3D00DD6076 /* Gr.fst */; };
92375269240C6EFE00DD6076 /* HCLr.fst in CopyFiles */ = {isa = PBXBuildFile; fileRef = 9237524B240C6E3D00DD6076 /* HCLr.fst */; };
9237526A240C6EFE00DD6076 /* mfcc.conf in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375253240C6E3D00DD6076 /* mfcc.conf */; };
9237526B240C6EFE00DD6076 /* word_boundary.int in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375254240C6E3D00DD6076 /* word_boundary.int */; };
9237526C240C6EFE00DD6076 /* words.txt in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375255240C6E3D00DD6076 /* words.txt */; };
9237526E240C6F1500DD6076 /* final.dubm in CopyFiles */ = {isa = PBXBuildFile; fileRef = 9237524D240C6E3D00DD6076 /* final.dubm */; };
9237526F240C6F1500DD6076 /* final.ie in CopyFiles */ = {isa = PBXBuildFile; fileRef = 9237524E240C6E3D00DD6076 /* final.ie */; };
92375270240C6F1500DD6076 /* final.mat in CopyFiles */ = {isa = PBXBuildFile; fileRef = 9237524F240C6E3D00DD6076 /* final.mat */; };
92375271240C6F1500DD6076 /* global_cmvn.stats in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375250240C6E3D00DD6076 /* global_cmvn.stats */; };
92375272240C6F1500DD6076 /* online_cmvn.conf in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375251240C6E3D00DD6076 /* online_cmvn.conf */; };
92375273240C6F1500DD6076 /* splice.conf in CopyFiles */ = {isa = PBXBuildFile; fileRef = 92375252240C6E3D00DD6076 /* splice.conf */; };
92375274240C6F1E00DD6076 /* 10001-90210-01803.wav in Resources */ = {isa = PBXBuildFile; fileRef = 92375256240C6E3D00DD6076 /* 10001-90210-01803.wav */; };
92BACED125BE125A00B5CC93 /* vosk-model-small-en-us-0.15 in Resources */ = {isa = PBXBuildFile; fileRef = 928CC50C25BE124400490481 /* vosk-model-small-en-us-0.15 */; };
92D6B8D325BDFEAC007FF08D /* VoskModel.swift in Sources */ = {isa = PBXBuildFile; fileRef = 92D6B8D225BDFEAC007FF08D /* VoskModel.swift */; };
92D86BD6253F823F0040D53F /* vosk-model-spk-0.4 in Resources */ = {isa = PBXBuildFile; fileRef = 92D86BD4253F823F0040D53F /* vosk-model-spk-0.4 */; };
/* End PBXBuildFile section */
/* Begin PBXCopyFilesBuildPhase section */
92375265240C6ECF00DD6076 /* CopyFiles */ = {
isa = PBXCopyFilesBuildPhase;
buildActionMask = 2147483647;
dstPath = "model-en";
dstSubfolderSpec = 7;
files = (
92375266240C6EFE00DD6076 /* disambig_tid.int in CopyFiles */,
92375267240C6EFE00DD6076 /* final.mdl in CopyFiles */,
92375268240C6EFE00DD6076 /* Gr.fst in CopyFiles */,
92375269240C6EFE00DD6076 /* HCLr.fst in CopyFiles */,
9237526A240C6EFE00DD6076 /* mfcc.conf in CopyFiles */,
9237526B240C6EFE00DD6076 /* word_boundary.int in CopyFiles */,
9237526C240C6EFE00DD6076 /* words.txt in CopyFiles */,
);
runOnlyForDeploymentPostprocessing = 0;
};
9237526D240C6F0400DD6076 /* CopyFiles */ = {
isa = PBXCopyFilesBuildPhase;
buildActionMask = 2147483647;
dstPath = "model-en/ivector";
dstSubfolderSpec = 7;
files = (
9237526E240C6F1500DD6076 /* final.dubm in CopyFiles */,
9237526F240C6F1500DD6076 /* final.ie in CopyFiles */,
92375270240C6F1500DD6076 /* final.mat in CopyFiles */,
92375271240C6F1500DD6076 /* global_cmvn.stats in CopyFiles */,
92375272240C6F1500DD6076 /* online_cmvn.conf in CopyFiles */,
92375273240C6F1500DD6076 /* splice.conf in CopyFiles */,
);
runOnlyForDeploymentPostprocessing = 0;
};
/* End PBXCopyFilesBuildPhase section */
/* Begin PBXFileReference section */
9237521E240C550B00DD6076 /* VoskApiTest.app */ = {isa = PBXFileReference; explicitFileType = wrapper.application; includeInIndex = 0; path = VoskApiTest.app; sourceTree = BUILT_PRODUCTS_DIR; };
92375221240C550B00DD6076 /* AppDelegate.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = AppDelegate.swift; sourceTree = "<group>"; };
@@ -34,12 +78,22 @@
9237523A240C642000DD6076 /* libkaldiwrap.a */ = {isa = PBXFileReference; lastKnownFileType = archive.ar; path = libkaldiwrap.a; sourceTree = "<group>"; };
92375243240C6DAF00DD6076 /* Accelerate.framework */ = {isa = PBXFileReference; lastKnownFileType = wrapper.framework; name = Accelerate.framework; path = System/Library/Frameworks/Accelerate.framework; sourceTree = SDKROOT; };
92375245240C6DC900DD6076 /* libstdc++.tbd */ = {isa = PBXFileReference; lastKnownFileType = "sourcecode.text-based-dylib-definition"; name = "libstdc++.tbd"; path = "usr/lib/libstdc++.tbd"; sourceTree = SDKROOT; };
92375248240C6E3D00DD6076 /* disambig_tid.int */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = text; path = disambig_tid.int; sourceTree = "<group>"; };
92375249240C6E3D00DD6076 /* final.mdl */ = {isa = PBXFileReference; lastKnownFileType = file; path = final.mdl; sourceTree = "<group>"; };
9237524A240C6E3D00DD6076 /* Gr.fst */ = {isa = PBXFileReference; lastKnownFileType = file; path = Gr.fst; sourceTree = "<group>"; };
9237524B240C6E3D00DD6076 /* HCLr.fst */ = {isa = PBXFileReference; lastKnownFileType = file; path = HCLr.fst; sourceTree = "<group>"; };
9237524D240C6E3D00DD6076 /* final.dubm */ = {isa = PBXFileReference; lastKnownFileType = file; path = final.dubm; sourceTree = "<group>"; };
9237524E240C6E3D00DD6076 /* final.ie */ = {isa = PBXFileReference; lastKnownFileType = file; path = final.ie; sourceTree = "<group>"; };
9237524F240C6E3D00DD6076 /* final.mat */ = {isa = PBXFileReference; lastKnownFileType = file; path = final.mat; sourceTree = "<group>"; };
92375250240C6E3D00DD6076 /* global_cmvn.stats */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = text; path = global_cmvn.stats; sourceTree = "<group>"; };
92375251240C6E3D00DD6076 /* online_cmvn.conf */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = text; path = online_cmvn.conf; sourceTree = "<group>"; };
92375252240C6E3D00DD6076 /* splice.conf */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = text; path = splice.conf; sourceTree = "<group>"; };
92375253240C6E3D00DD6076 /* mfcc.conf */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = text; path = mfcc.conf; sourceTree = "<group>"; };
92375254240C6E3D00DD6076 /* word_boundary.int */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = text; path = word_boundary.int; sourceTree = "<group>"; };
92375255240C6E3D00DD6076 /* words.txt */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = text; path = words.txt; sourceTree = "<group>"; };
92375256240C6E3D00DD6076 /* 10001-90210-01803.wav */ = {isa = PBXFileReference; lastKnownFileType = audio.wav; path = "10001-90210-01803.wav"; sourceTree = "<group>"; };
928CC50C25BE124400490481 /* vosk-model-small-en-us-0.15 */ = {isa = PBXFileReference; lastKnownFileType = folder; name = "vosk-model-small-en-us-0.15"; path = "/Users/shmyrev/Documents/IOS/VoskApiTest/VoskApiTest/Vosk/vosk-model-small-en-us-0.15"; sourceTree = "<absolute>"; };
92AA22AD244CDD1200DA464B /* vosk_api.h */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = sourcecode.c.h; path = vosk_api.h; sourceTree = "<group>"; };
92AA22AE244CDD5200DA464B /* bridging.h */ = {isa = PBXFileReference; fileEncoding = 4; lastKnownFileType = sourcecode.c.h; path = bridging.h; sourceTree = "<group>"; };
92D6B8D225BDFEAC007FF08D /* VoskModel.swift */ = {isa = PBXFileReference; lastKnownFileType = sourcecode.swift; path = VoskModel.swift; sourceTree = "<group>"; };
92D86BD4253F823F0040D53F /* vosk-model-spk-0.4 */ = {isa = PBXFileReference; lastKnownFileType = folder; path = "vosk-model-spk-0.4"; sourceTree = "<group>"; };
/* End PBXFileReference section */
/* Begin PBXFrameworksBuildPhase section */
@@ -85,7 +139,6 @@
9237522A240C550B00DD6076 /* LaunchScreen.storyboard */,
9237522D240C550B00DD6076 /* Info.plist */,
92375233240C558900DD6076 /* Vosk.swift */,
92D6B8D225BDFEAC007FF08D /* VoskModel.swift */,
);
path = VoskApiTest;
sourceTree = "<group>";
@@ -93,8 +146,7 @@
92375239240C642000DD6076 /* Vosk */ = {
isa = PBXGroup;
children = (
928CC50C25BE124400490481 /* vosk-model-small-en-us-0.15 */,
92D86BD4253F823F0040D53F /* vosk-model-spk-0.4 */,
92375247240C6E3D00DD6076 /* model-android */,
92375256240C6E3D00DD6076 /* 10001-90210-01803.wav */,
92AA22AD244CDD1200DA464B /* vosk_api.h */,
9237523A240C642000DD6076 /* libkaldiwrap.a */,
@@ -112,6 +164,34 @@
name = Frameworks;
sourceTree = "<group>";
};
92375247240C6E3D00DD6076 /* model-android */ = {
isa = PBXGroup;
children = (
92375248240C6E3D00DD6076 /* disambig_tid.int */,
92375249240C6E3D00DD6076 /* final.mdl */,
9237524A240C6E3D00DD6076 /* Gr.fst */,
9237524B240C6E3D00DD6076 /* HCLr.fst */,
9237524C240C6E3D00DD6076 /* ivector */,
92375253240C6E3D00DD6076 /* mfcc.conf */,
92375254240C6E3D00DD6076 /* word_boundary.int */,
92375255240C6E3D00DD6076 /* words.txt */,
);
path = "model-android";
sourceTree = "<group>";
};
9237524C240C6E3D00DD6076 /* ivector */ = {
isa = PBXGroup;
children = (
9237524D240C6E3D00DD6076 /* final.dubm */,
9237524E240C6E3D00DD6076 /* final.ie */,
9237524F240C6E3D00DD6076 /* final.mat */,
92375250240C6E3D00DD6076 /* global_cmvn.stats */,
92375251240C6E3D00DD6076 /* online_cmvn.conf */,
92375252240C6E3D00DD6076 /* splice.conf */,
);
path = ivector;
sourceTree = "<group>";
};
/* End PBXGroup section */
/* Begin PBXNativeTarget section */
@@ -122,6 +202,8 @@
9237521A240C550B00DD6076 /* Sources */,
9237521B240C550B00DD6076 /* Frameworks */,
9237521C240C550B00DD6076 /* Resources */,
92375265240C6ECF00DD6076 /* CopyFiles */,
9237526D240C6F0400DD6076 /* CopyFiles */,
);
buildRules = (
);
@@ -172,11 +254,9 @@
isa = PBXResourcesBuildPhase;
buildActionMask = 2147483647;
files = (
92BACED125BE125A00B5CC93 /* vosk-model-small-en-us-0.15 in Resources */,
92375274240C6F1E00DD6076 /* 10001-90210-01803.wav in Resources */,
9237522C240C550B00DD6076 /* LaunchScreen.storyboard in Resources */,
92375229240C550B00DD6076 /* Assets.xcassets in Resources */,
92D86BD6253F823F0040D53F /* vosk-model-spk-0.4 in Resources */,
92375227240C550B00DD6076 /* Main.storyboard in Resources */,
);
runOnlyForDeploymentPostprocessing = 0;
@@ -190,7 +270,6 @@
files = (
92375224240C550B00DD6076 /* ViewController.swift in Sources */,
92375222240C550B00DD6076 /* AppDelegate.swift in Sources */,
92D6B8D325BDFEAC007FF08D /* VoskModel.swift in Sources */,
92375234240C558900DD6076 /* Vosk.swift in Sources */,
);
runOnlyForDeploymentPostprocessing = 0;
@@ -333,7 +412,7 @@
buildSettings = {
ASSETCATALOG_COMPILER_APPICON_NAME = AppIcon;
CLANG_ENABLE_MODULES = YES;
ENABLE_BITCODE = YES;
ENABLE_BITCODE = NO;
INFOPLIST_FILE = VoskApiTest/Info.plist;
LD_RUNPATH_SEARCH_PATHS = "$(inherited) @executable_path/Frameworks";
LIBRARY_SEARCH_PATHS = (
@@ -355,7 +434,7 @@
buildSettings = {
ASSETCATALOG_COMPILER_APPICON_NAME = AppIcon;
CLANG_ENABLE_MODULES = YES;
ENABLE_BITCODE = YES;
ENABLE_BITCODE = NO;
INFOPLIST_FILE = VoskApiTest/Info.plist;
LD_RUNPATH_SEARCH_PATHS = "$(inherited) @executable_path/Frameworks";
LIBRARY_SEARCH_PATHS = (
@@ -2,9 +2,6 @@
<Workspace
version = "1.0">
<FileRef
location = "group:/Users/shmyrev/Documents/IOS/VoskApiTest/VoskApiTest/Vosk/vosk-model-small-en-us-0.15">
</FileRef>
<FileRef
location = "self:">
location = "self:VoskApiTest.xcodeproj">
</FileRef>
</Workspace>
+13 -42
View File
@@ -1,6 +1,6 @@
<?xml version="1.0" encoding="UTF-8"?>
<document type="com.apple.InterfaceBuilder3.CocoaTouch.Storyboard.XIB" version="3.0" toolsVersion="13771" targetRuntime="iOS.CocoaTouch" propertyAccessControl="none" useAutolayout="YES" useTraitCollections="YES" colorMatched="YES" initialViewController="bdW-KL-Y8Z">
<device id="retina5_5" orientation="portrait">
<document type="com.apple.InterfaceBuilder3.CocoaTouch.Storyboard.XIB" version="3.0" toolsVersion="13771" targetRuntime="iOS.CocoaTouch" propertyAccessControl="none" useAutolayout="YES" useTraitCollections="YES" colorMatched="YES" initialViewController="BYZ-38-t0r">
<device id="retina4_7" orientation="portrait">
<adaptation id="fullscreen"/>
</device>
<dependencies>
@@ -10,52 +10,23 @@
</dependencies>
<scenes>
<!--View Controller-->
<scene sceneID="nEc-89-Iqu">
<scene sceneID="tne-QT-ifu">
<objects>
<viewController id="bdW-KL-Y8Z" customClass="ViewController" customModule="VoskApiTest" customModuleProvider="target" sceneMemberID="viewController">
<layoutGuides>
<viewControllerLayoutGuide type="top" id="Hyr-Dz-4mU"/>
<viewControllerLayoutGuide type="bottom" id="w4A-5X-uBu"/>
</layoutGuides>
<view key="view" contentMode="scaleToFill" id="m5v-US-bvR">
<rect key="frame" x="0.0" y="0.0" width="414" height="736"/>
<autoresizingMask key="autoresizingMask" widthSizable="YES" heightSizable="YES"/>
<subviews>
<button opaque="NO" contentMode="scaleToFill" fixedFrame="YES" contentHorizontalAlignment="center" contentVerticalAlignment="center" buttonType="roundedRect" lineBreakMode="middleTruncation" translatesAutoresizingMaskIntoConstraints="NO" id="IaT-no-U3i" userLabel="Microphone">
<rect key="frame" x="124" y="34" width="157" height="30"/>
<autoresizingMask key="autoresizingMask" flexibleMaxX="YES" flexibleMaxY="YES"/>
<state key="normal" title="Recognize Microphone"/>
<connections>
<action selector="runRecognizeMicrohpone:" destination="bdW-KL-Y8Z" eventType="touchUpInside" id="hGB-lz-N2B"/>
</connections>
</button>
<button opaque="NO" contentMode="scaleToFill" fixedFrame="YES" contentHorizontalAlignment="center" contentVerticalAlignment="center" buttonType="roundedRect" lineBreakMode="middleTruncation" translatesAutoresizingMaskIntoConstraints="NO" id="GC5-nT-FQR" userLabel="File">
<rect key="frame" x="90" y="84" width="221" height="41"/>
<autoresizingMask key="autoresizingMask" flexibleMaxX="YES" flexibleMaxY="YES"/>
<state key="normal" title="Recognize File"/>
<connections>
<action selector="runRecognizeFile:" destination="bdW-KL-Y8Z" eventType="touchUpInside" id="xp5-Yi-rnN"/>
</connections>
</button>
<textView clipsSubviews="YES" multipleTouchEnabled="YES" contentMode="scaleToFill" fixedFrame="YES" text="Results here" textAlignment="natural" translatesAutoresizingMaskIntoConstraints="NO" id="w4X-cu-USq">
<rect key="frame" x="11" y="112" width="383" height="569"/>
<autoresizingMask key="autoresizingMask" flexibleMaxX="YES" flexibleMaxY="YES"/>
<color key="backgroundColor" white="1" alpha="1" colorSpace="calibratedWhite"/>
<fontDescription key="fontDescription" type="system" pointSize="14"/>
<textInputTraits key="textInputTraits" autocapitalizationType="sentences"/>
</textView>
</subviews>
<viewController id="BYZ-38-t0r" customClass="ViewController" customModule="VoskApiTest" customModuleProvider="target" sceneMemberID="viewController">
<textView key="view" clipsSubviews="YES" multipleTouchEnabled="YES" contentMode="scaleToFill" editable="NO" textAlignment="natural" id="CtX-mx-X98">
<rect key="frame" x="0.0" y="0.0" width="375" height="667"/>
<autoresizingMask key="autoresizingMask" flexibleMaxX="YES" flexibleMaxY="YES"/>
<color key="backgroundColor" white="1" alpha="1" colorSpace="calibratedWhite"/>
</view>
<fontDescription key="fontDescription" type="system" pointSize="14"/>
<textInputTraits key="textInputTraits" autocapitalizationType="sentences"/>
</textView>
<connections>
<outlet property="mainText" destination="w4X-cu-USq" id="rZS-nz-Wql"/>
<outlet property="recognizeFile" destination="GC5-nT-FQR" id="dRe-tc-IA0"/>
<outlet property="recognizeMicrophone" destination="IaT-no-U3i" id="IuM-aa-pAP"/>
<outlet property="mainText" destination="CtX-mx-X98" id="oJy-5J-NKp"/>
</connections>
</viewController>
<placeholder placeholderIdentifier="IBFirstResponder" id="nWA-4D-pA6" userLabel="First Responder" sceneMemberID="firstResponder"/>
<placeholder placeholderIdentifier="IBFirstResponder" id="dkx-z0-nzr" sceneMemberID="firstResponder"/>
</objects>
<point key="canvasLocation" x="-17.39130434782609" y="-285.32608695652175"/>
<point key="canvasLocation" x="32.799999999999997" y="32.833583208395808"/>
</scene>
</scenes>
</document>
+16 -99
View File
@@ -3,113 +3,30 @@
// VoskApiTest
//
// Created by Niсkolay Shmyrev on 01.03.20.
// Copyright © 2020-2021 Alpha Cephei. All rights reserved.
// Copyright © 2020 Alpha Cephei. All rights reserved.
//
import UIKit
import AVFoundation
enum WorkMode {
case stopped
case microphone
case file
}
class ViewController: UIViewController {
var mode: WorkMode!
@IBOutlet weak var recognizeFile: UIButton!
@IBOutlet weak var mainText: UITextView!
@IBOutlet weak var recognizeMicrophone: UIButton!
var audioEngine : AVAudioEngine!
var processingQueue: DispatchQueue!
var model : VoskModel!
func setMode(mode: WorkMode) {
switch mode {
case .stopped:
self.recognizeFile.isEnabled = true
self.recognizeMicrophone.isEnabled = true
self.recognizeMicrophone.setTitle("Recognize Microphone",for: .normal)
case .microphone:
self.recognizeFile.isEnabled = false
self.recognizeMicrophone.isEnabled = true
self.recognizeMicrophone.setTitle("Stop Microphone",for: .normal)
self.mainText.text = ""
case .file:
self.recognizeFile.isEnabled = false
self.recognizeMicrophone.isEnabled = false
self.mainText.text = "Processing file..."
}
self.mode = mode
}
func startAudioEngine() {
do {
// Create a new audio engine.
audioEngine = AVAudioEngine()
let inputNode = audioEngine.inputNode
let formatInput = inputNode.inputFormat(forBus: 0)
let formatPcm = AVAudioFormat.init(commonFormat: AVAudioCommonFormat.pcmFormatInt16, sampleRate: formatInput.sampleRate, channels: 1, interleaved: true)
let recognizer = Vosk(model: model, sampleRate: Float(formatInput.sampleRate))
inputNode.installTap(onBus: 0,
bufferSize: UInt32(formatInput.sampleRate / 10),
format: formatPcm) { buffer, time in
self.processingQueue.async {
let res = recognizer.recognizeData(buffer: buffer)
DispatchQueue.main.async {
self.mainText.text = res + "\n" + self.mainText.text
}
}
}
// Start the stream of audio data.
audioEngine.prepare()
try audioEngine.start()
} catch {
print("Unable to start AVAudioEngine: \(error.localizedDescription)")
}
}
func stopAudioEngine() {
audioEngine.stop()
}
@IBAction func runRecognizeMicrohpone(_ sender: Any) {
if (mode == .stopped) {
setMode(mode: .microphone)
startAudioEngine()
} else {
stopAudioEngine()
setMode(mode: .stopped)
}
}
@IBAction func runRecognizeFile(_ sender: Any) {
setMode(mode: .file)
processingQueue.async {
let recognizer = Vosk(model: self.model, sampleRate: 16000.0)
let res = recognizer.recognizeFile()
DispatchQueue.main.async {
self.mainText.text = res
self.setMode(mode: .stopped)
}
}
}
@IBOutlet var mainText: UITextView!
override func viewDidLoad() {
super.viewDidLoad()
setMode(mode: .stopped)
processingQueue = DispatchQueue(label: "recognizerQueue")
model = VoskModel()
}
DispatchQueue.global(qos: .userInitiated).async {
DispatchQueue.main.async {
self.mainText.text = "Processing file..."
}
let vosk = Vosk()
let res = vosk.recognizeFile()
DispatchQueue.main.async {
self.mainText.text = res
}
}
}
override func didReceiveMemoryWarning() {
super.didReceiveMemoryWarning()
}
+16 -33
View File
@@ -3,52 +3,35 @@
// VoskApiTest
//
// Created by Niсkolay Shmyrev on 01.03.20.
// Copyright © 2020-2021 Alpha Cephei. All rights reserved.
// Copyright © 2020 Alpha Cephei. All rights reserved.
//
import Foundation
import AVFoundation
public final class Vosk {
var recognizer : OpaquePointer!
init(model: VoskModel, sampleRate: Float) {
recognizer = vosk_recognizer_new_spk(model.model, model.spkModel, sampleRate)
}
deinit {
vosk_recognizer_free(recognizer);
}
func recognizeFile() -> String {
var sres = ""
if let resourcePath = Bundle.main.resourcePath {
let modelPath = resourcePath + "/model-en"
let model = vosk_model_new(modelPath);
let recognizer = vosk_recognizer_new(model, 16000.0)
let audioFile = URL(fileURLWithPath: resourcePath + "/10001-90210-01803.wav")
if let data = try? Data(contentsOf: audioFile) {
let _ = data.withUnsafeBytes {
vosk_recognizer_accept_waveform(recognizer, $0, Int32(data.count))
}
let res = vosk_recognizer_final_result(recognizer);
sres = String(validatingUTF8: res!)!;
print(sres);
let _ = data.withUnsafeBytes {
vosk_recognizer_accept_waveform(recognizer, $0, Int32(data.count))
}
let res = vosk_recognizer_final_result(recognizer);
sres = String(validatingUTF8: res!)!;
print(sres);
}
vosk_recognizer_free(recognizer)
vosk_model_free(model)
}
return sres
}
func recognizeData(buffer : AVAudioPCMBuffer) -> String {
let dataLen = Int(buffer.frameLength * 2)
let channels = UnsafeBufferPointer(start: buffer.int16ChannelData, count: 1)
let endOfSpeech = channels[0].withMemoryRebound(to: Int8.self, capacity: dataLen) {
vosk_recognizer_accept_waveform(recognizer, $0, Int32(dataLen))
}
let res = endOfSpeech == 1 ?vosk_recognizer_result(recognizer) :vosk_recognizer_partial_result(recognizer)
return String(validatingUTF8: res!)!;
}
}
+3 -166
View File
@@ -12,200 +12,37 @@
// See the License for the specific language governing permissions and
// limitations under the License.
/* This header contains the C API for Vosk speech recognition system */
#ifndef VOSK_API_H
#define VOSK_API_H
#ifndef _VOSK_API_H_
#define _VOSK_API_H_
#ifdef __cplusplus
extern "C" {
#endif
/** Model stores all the data required for recognition
* it contains static data and can be shared across processing
* threads. */
typedef struct VoskModel VoskModel;
/** Speaker model is the same as model but contains the data
* for speaker identification. */
typedef struct VoskSpkModel VoskSpkModel;
/** Recognizer object is the main object which processes data.
* Each recognizer usually runs in own thread and takes audio as input.
* Once audio is processed recognizer returns JSON object as a string
* which represent decoded information - words, confidences, times, n-best lists,
* speaker information and so on */
typedef struct VoskRecognizer VoskRecognizer;
/** Loads model data from the file and returns the model object
*
* @param model_path: the path of the model on the filesystem
@ @returns model object */
VoskModel *vosk_model_new(const char *model_path);
/** Releases the model memory
*
* The model object is reference-counted so if some recognizer
* depends on this model, model might still stay alive. When
* last recognizer is released, model will be released too. */
void vosk_model_free(VoskModel *model);
/** Loads speaker model data from the file and returns the model object
*
* @param model_path: the path of the model on the filesystem
* @returns model object */
VoskSpkModel *vosk_spk_model_new(const char *model_path);
/** Releases the model memory
*
* The model object is reference-counted so if some recognizer
* depends on this model, model might still stay alive. When
* last recognizer is released, model will be released too. */
void vosk_spk_model_free(VoskSpkModel *model);
/** Creates the recognizer object
*
* The recognizers process the speech and return text using shared model data
* @param sample_rate The sample rate of the audio you going to feed into the recognizer
* @returns recognizer object */
VoskRecognizer *vosk_recognizer_new(VoskModel *model, float sample_rate);
/** Creates the recognizer object with speaker recognition
*
* With the speaker recognition mode the recognizer not just recognize
* text but also return speaker vectors one can use for speaker identification
*
* @param spk_model speaker model for speaker identification
* @param sample_rate The sample rate of the audio you going to feed into the recognizer
* @returns recognizer object */
VoskRecognizer *vosk_recognizer_new_spk(VoskModel *model, VoskSpkModel *spk_model, float sample_rate);
/** Creates the recognizer object with the phrase list
*
* Sometimes when you want to improve recognition accuracy and when you don't need
* to recognize large vocabulary you can specify a list of phrases to recognize. This
* will improve recognizer speed and accuracy but might return [unk] if user said
* something different.
*
* Only recognizers with lookahead models support this type of quick configuration.
* Precompiled HCLG graph models are not supported.
*
* @param sample_rate The sample rate of the audio you going to feed into the recognizer
* @param grammar The string with the list of phrases to recognize as JSON array of strings,
* for example "["one two three four five", "[unk]"]".
*
* @returns recognizer object */
VoskRecognizer *vosk_recognizer_new_grm(VoskModel *model, float sample_rate, const char *grammar);
/** Accept voice data
*
* accept and process new chunk of voice data
*
* @param data - audio data in PCM 16-bit mono format
* @param length - length of the audio data
* @returns true if silence is occured and you can retrieve a new utterance with result method */
int vosk_recognizer_accept_waveform(VoskRecognizer *recognizer, const char *data, int length);
/** Same as above but the version with the short data for language bindings where you have
* audio as array of shorts */
int vosk_recognizer_accept_waveform_s(VoskRecognizer *recognizer, const short *data, int length);
/** Same as above but the version with the float data for language bindings where you have
* audio as array of floats */
int vosk_recognizer_accept_waveform_f(VoskRecognizer *recognizer, const float *data, int length);
/** Returns speech recognition result
*
* @returns the result in JSON format which contains decoded line, decoded
* words, times in seconds and confidences. You can parse this result
* with any json parser
*
* <pre>
* {
* "result" : [{
* "conf" : 1.000000,
* "end" : 1.110000,
* "start" : 0.870000,
* "word" : "what"
* }, {
* "conf" : 1.000000,
* "end" : 1.530000,
* "start" : 1.110000,
* "word" : "zero"
* }, {
* "conf" : 1.000000,
* "end" : 1.950000,
* "start" : 1.530000,
* "word" : "zero"
* }, {
* "conf" : 1.000000,
* "end" : 2.340000,
* "start" : 1.950000,
* "word" : "zero"
* }, {
* "conf" : 1.000000,
* "end" : 2.610000,
* "start" : 2.340000,
* "word" : "one"
* }],
* "text" : "what zero zero zero one"
* }
* </pre>
*/
const char *vosk_recognizer_result(VoskRecognizer *recognizer);
/** Returns partial speech recognition
*
* @returns partial speech recognition text which is not yet finalized.
* result may change as recognizer process more data.
*
* <pre>
* {
* "partial" : "cyril one eight zero"
* }
* </pre>
*/
const char *vosk_recognizer_partial_result(VoskRecognizer *recognizer);
/** Returns speech recognition result. Same as result, but doesn't wait for silence
* You usually call it in the end of the stream to get final bits of audio. It
* flushes the feature pipeline, so all remaining audio chunks got processed.
*
* @returns speech result in JSON format.
*/
const char *vosk_recognizer_final_result(VoskRecognizer *recognizer);
/** Releases recognizer object
*
* Underlying model is also unreferenced and if needed released */
void vosk_recognizer_free(VoskRecognizer *recognizer);
/** Set log level for Kaldi messages
*
* @param log_level the level
* 0 - default value to print info and error messages but no debug
* less than 0 - don't print info messages
* greather than 0 - more verbose mode
*/
void vosk_set_log_level(int log_level);
#ifdef __cplusplus
}
#endif
#endif /* VOSK_API_H */
#endif /* _VOSK_API_H_ */
-36
View File
@@ -1,36 +0,0 @@
//
// Vosk.swift
// VoskApiTest
//
// Created by Niсkolay Shmyrev on 01.03.20.
// Copyright © 2020-2021 Alpha Cephei. All rights reserved.
//
import Foundation
public final class VoskModel {
var model : OpaquePointer!
var spkModel : OpaquePointer!
init() {
// Set to -1 to disable logs
vosk_set_log_level(0);
if let resourcePath = Bundle.main.resourcePath {
let modelPath = resourcePath + "/vosk-model-small-en-us-0.15"
let spkModelPath = resourcePath + "/vosk-model-spk-0.4"
model = vosk_model_new(modelPath)
spkModel = vosk_spk_model_new(spkModelPath)
}
}
deinit {
vosk_model_free(model)
vosk_spk_model_free(spkModel)
}
}
+65
View File
@@ -0,0 +1,65 @@
KALDI_ROOT ?= $(HOME)/kaldi
CFLAGS := -g -O2 -DPIC -fPIC -Wno-unused-function
CPPFLAGS := -I$(JAVA_HOME)/include -I$(JAVA_HOME)/include/linux -I$(KALDI_ROOT)/src -I$(KALDI_ROOT)/tools/openfst/include -I../src
KALDI_LIBS = \
${KALDI_ROOT}/src/online2/kaldi-online2.a \
${KALDI_ROOT}/src/decoder/kaldi-decoder.a \
${KALDI_ROOT}/src/ivector/kaldi-ivector.a \
${KALDI_ROOT}/src/gmm/kaldi-gmm.a \
${KALDI_ROOT}/src/nnet3/kaldi-nnet3.a \
${KALDI_ROOT}/src/tree/kaldi-tree.a \
${KALDI_ROOT}/src/feat/kaldi-feat.a \
${KALDI_ROOT}/src/lat/kaldi-lat.a \
${KALDI_ROOT}/src/lm/kaldi-lm.a \
${KALDI_ROOT}/src/hmm/kaldi-hmm.a \
${KALDI_ROOT}/src/transform/kaldi-transform.a \
${KALDI_ROOT}/src/cudamatrix/kaldi-cudamatrix.a \
${KALDI_ROOT}/src/matrix/kaldi-matrix.a \
${KALDI_ROOT}/src/fstext/kaldi-fstext.a \
${KALDI_ROOT}/src/util/kaldi-util.a \
${KALDI_ROOT}/src/base/kaldi-base.a \
${KALDI_ROOT}/tools/openfst/lib/libfst.a \
${KALDI_ROOT}/tools/openfst/lib/libfstngram.a \
${KALDI_ROOT}/tools/OpenBLAS/libopenblas.a \
-lgfortran
all: libvosk_jni.so
VOSK_SOURCES = \
vosk_wrap.cc \
../src/kaldi_recognizer.cc \
../src/kaldi_recognizer.h \
../src/model.cc \
../src/model.h \
../src/spk_model.cc \
../src/spk_model.h \
../src/vosk_api.cc \
../src/vosk_api.h
libvosk_jni.so: $(VOSK_SOURCES)
$(CXX) -shared -o $@ $(CPPFLAGS) $(CFLAGS) $(VOSK_SOURCES) $(KALDI_LIBS)
vosk_wrap.cc: ../src/vosk.i
mkdir -p org/kaldi
swig -c++ -I../src \
-java -package org.kaldi \
-outdir org/kaldi -o $@ $<
clean:
$(RM) *.so *_wrap.cc *_wrap.o test/*.class
$(RM) -r org model-en
model:
wget https://alphacephei.com/kaldi/models/vosk-model-small-en-us-0.3.zip
unzip vosk-model-small-en-us-0.3.zip && rm vosk-model-small-en-us-0.3.zip
mv vosk-model-small-en-us-0.3 model
model-spk:
wget https://alphacephei.com/kaldi/models/vosk-model-spk-0.3.zip
unzip vosk-model-spk-0.3.zip && rm vosk-model-spk-0.3.zip
mv vosk-model-spk-0.3 model-spk
run: model model-spk
javac test/*.java org/kaldi/*.java
java -Djava.library.path=. -cp . test.DecoderTest
+16 -4
View File
@@ -1,7 +1,19 @@
Java bindings for Vosk API using jnr-ffi
Java API sample
See demo project for details, build it with Gradle.
Doesn't work on Windows or Mac yet, help to prepare the packaged jars is welcome.
Download model and unpack as "model" folder in the demo project.
For now to try it:
Make sure you are using recent JDK and Gradle.
On Linux you can do
1. Build recent kaldi
1. `git clone https://github.com/alphacep/vosk-api`
1. `cd vosk-api/java`
1. `export KALDI_ROOT=<KALDI_ROOT>`
1. `export JAVA_HOME=<JAVA_HOME>`
1. `make`
1. `make run`
For details of the code you can check:
https://github.com/alphacep/vosk-api/blob/master/java/test/DecoderTest.java
-19
View File
@@ -1,19 +0,0 @@
plugins {
id 'application'
}
application {
mainClass = 'org.vosk.demo.DecoderDemo'
}
repositories {
mavenCentral()
maven {
url 'https://alphacephei.com/maven/'
}
}
dependencies {
implementation group: 'net.java.dev.jna', name: 'jna', version: '5.7.0'
implementation group: 'com.alphacephei', name: 'vosk', version: '0.3.30+'
}
@@ -1,38 +0,0 @@
package org.vosk.demo;
import java.io.FileInputStream;
import java.io.BufferedInputStream;
import java.io.IOException;
import java.io.InputStream;
import javax.sound.sampled.AudioSystem;
import javax.sound.sampled.UnsupportedAudioFileException;
import org.vosk.LogLevel;
import org.vosk.Recognizer;
import org.vosk.LibVosk;
import org.vosk.Model;
public class DecoderDemo {
public static void main(String[] argv) throws IOException, UnsupportedAudioFileException {
LibVosk.setLogLevel(LogLevel.DEBUG);
try (Model model = new Model("model");
InputStream ais = AudioSystem.getAudioInputStream(new BufferedInputStream(new FileInputStream("../../python/example/test.wav")));
Recognizer recognizer = new Recognizer(model, 16000)) {
int nbytes;
byte[] b = new byte[4096];
while ((nbytes = ais.read(b)) >= 0) {
if (recognizer.acceptWaveForm(b, nbytes)) {
System.out.println(recognizer.getResult());
} else {
System.out.println(recognizer.getPartialResult());
}
}
System.out.println(recognizer.getFinalResult());
}
}
}
-58
View File
@@ -1,58 +0,0 @@
plugins {
id 'java-library'
id 'maven-publish'
}
archivesBaseName = 'vosk'
group = 'com.alphacephei'
version = '0.3.30'
repositories {
mavenCentral()
}
dependencies {
implementation group: 'net.java.dev.jna', name: 'jna', version: '5.7.0'
testImplementation 'junit:junit:4.13'
}
publishing {
publications {
mavenJava(MavenPublication) {
artifactId = 'vosk'
from components.java
pom {
name = 'Vosk'
description = 'Speech recognition library'
url = 'http://www.alphacephei.com.com/vosk/'
licenses {
license {
name = 'The Apache License, Version 2.0'
url = 'http://www.apache.org/licenses/LICENSE-2.0.txt'
}
}
developers {
developer {
id = 'alphacephei'
name = 'Alpha Cephei Inc'
email = 'contact@alphacephei.com'
}
}
scm {
connection = 'scm:git:git://github.com/alphacep/vosk-api.git'
url = 'https://github.com/alphacep/vosk-api/'
}
}
}
}
repositories {
maven {
url = "repo"
}
}
}
test {
dependsOn cleanTest
testLogging.showStandardStreams = true
}
@@ -1,86 +0,0 @@
package org.vosk;
import com.sun.jna.Native;
import com.sun.jna.Library;
import com.sun.jna.Platform;
import com.sun.jna.Pointer;
import java.io.File;
import java.io.InputStream;
import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.StandardCopyOption;
public class LibVosk {
private static void unpackDll(File targetDir, String lib) throws IOException {
InputStream source = LibVosk.class.getResourceAsStream("/win32-x86-64/" + lib + ".dll");
Files.copy(source, new File(targetDir, lib + ".dll").toPath(), StandardCopyOption.REPLACE_EXISTING);
}
static {
if (Platform.isWindows()) {
// We have to unpack dependencies
try {
// To get a tmp folder we unpack small library and mark it for deletion
File tmpFile = Native.extractFromResourcePath("/win32-x86-64/empty");
File tmpDir = tmpFile.getParentFile();
new File(tmpDir, tmpFile.getName() + ".x").createNewFile();
// Now unpack dependencies
unpackDll(tmpDir, "libwinpthread-1");
unpackDll(tmpDir, "libgcc_s_seh-1");
unpackDll(tmpDir, "libstdc++-6");
} catch (IOException e) {
// Nothing for now, it will fail on next step
} finally {
Native.register(LibVosk.class, "libvosk");
}
} else {
Native.register(LibVosk.class, "vosk");
}
}
public static native void vosk_set_log_level(int level);
public static native Pointer vosk_model_new(String path);
public static native void vosk_model_free(Pointer model);
public static native Pointer vosk_spk_model_new(String path);
public static native void vosk_spk_model_free(Pointer model);
public static native Pointer vosk_recognizer_new(Model model, float sample_rate);
public static native Pointer vosk_recognizer_new_spk(Pointer model, float sample_rate, Pointer spk_model);
public static native Pointer vosk_recognizer_new_grm(Pointer model, float sample_rate, String grammar);
public static native void vosk_recognizer_set_max_alternatives(Pointer recognizer, int max_alternatives);
public static native void vosk_recognizer_set_words(Pointer recognizer, boolean words);
public static native void vosk_recognizer_set_spk_model(Pointer recognizer, Pointer spk_model);
public static native boolean vosk_recognizer_accept_waveform(Pointer recognizer, byte[] data, int len);
public static native boolean vosk_recognizer_accept_waveform_s(Pointer recognizer, short[] data, int len);
public static native boolean vosk_recognizer_accept_waveform_f(Pointer recognizer, float[] data, int len);
public static native String vosk_recognizer_result(Pointer recognizer);
public static native String vosk_recognizer_final_result(Pointer recognizer);
public static native String vosk_recognizer_partial_result(Pointer recognizer);
public static native void vosk_recognizer_reset(Pointer recognizer);
public static native void vosk_recognizer_free(Pointer recognizer);
public static void setLogLevel(LogLevel loglevel) {
vosk_set_log_level(loglevel.getValue());
}
}
@@ -1,17 +0,0 @@
package org.vosk;
public enum LogLevel {
WARNINGS(-1), // Print warning and errors
INFO(0), // Print info, along with warning and error messages, but no debug
DEBUG(1); // Print debug info
private final int value;
LogLevel(int value) {
this.value = value;
}
public int getValue() {
return this.value;
}
}
@@ -1,17 +0,0 @@
package org.vosk;
import com.sun.jna.PointerType;
public class Model extends PointerType implements AutoCloseable {
public Model() {
}
public Model(String path) {
super(LibVosk.vosk_model_new(path));
}
@Override
public void close() {
LibVosk.vosk_model_free(this.getPointer());
}
}
@@ -1,62 +0,0 @@
package org.vosk;
import com.sun.jna.PointerType;
public class Recognizer extends PointerType implements AutoCloseable {
public Recognizer(Model model, float sampleRate) {
super(LibVosk.vosk_recognizer_new(model, sampleRate));
}
public Recognizer(Model model, float sampleRate, SpeakerModel spkModel) {
super(LibVosk.vosk_recognizer_new_spk(model.getPointer(), sampleRate, spkModel.getPointer()));
}
public Recognizer(Model model, float sampleRate, String grammar) {
super(LibVosk.vosk_recognizer_new_grm(model.getPointer(), sampleRate, grammar));
}
public void setMaxAlternatives(int maxAlternatives) {
LibVosk.vosk_recognizer_set_max_alternatives(this.getPointer(), maxAlternatives);
}
public void setWords(boolean words) {
LibVosk.vosk_recognizer_set_words(this.getPointer(), words);
}
public void setSpeakerModel(SpeakerModel spkModel) {
LibVosk.vosk_recognizer_set_spk_model(this.getPointer(), spkModel.getPointer());
}
public boolean acceptWaveForm(byte[] data, int len) {
return LibVosk.vosk_recognizer_accept_waveform(this.getPointer(), data, len);
}
public boolean acceptWaveForm(short[] data, int len) {
return LibVosk.vosk_recognizer_accept_waveform_s(this.getPointer(), data, len);
}
public boolean acceptWaveForm(float[] data, int len) {
return LibVosk.vosk_recognizer_accept_waveform_f(this.getPointer(), data, len);
}
public String getResult() {
return LibVosk.vosk_recognizer_result(this.getPointer());
}
public String getPartialResult() {
return LibVosk.vosk_recognizer_partial_result(this.getPointer());
}
public String getFinalResult() {
return LibVosk.vosk_recognizer_final_result(this.getPointer());
}
public void reset() {
LibVosk.vosk_recognizer_reset(this.getPointer());
}
@Override
public void close() {
LibVosk.vosk_recognizer_free(this.getPointer());
}
}
@@ -1,17 +0,0 @@
package org.vosk;
import com.sun.jna.PointerType;
public class SpeakerModel extends PointerType implements AutoCloseable {
public SpeakerModel() {
}
public SpeakerModel(String path) {
super(LibVosk.vosk_spk_model_new(path));
}
@Override
public void close() {
LibVosk.vosk_spk_model_free(this.getPointer());
}
}
@@ -1 +0,0 @@
Mainly for work around JNA API
@@ -1,96 +0,0 @@
package org.vosk.test;
import java.io.FileInputStream;
import java.io.BufferedInputStream;
import java.io.IOException;
import java.io.InputStream;
import java.nio.ByteBuffer;
import java.nio.ByteOrder;
import org.junit.Test;
import org.junit.Assert;
import javax.sound.sampled.AudioSystem;
import javax.sound.sampled.UnsupportedAudioFileException;
import org.vosk.LogLevel;
import org.vosk.Recognizer;
import org.vosk.LibVosk;
import org.vosk.Model;
public class DecoderTest {
@Test
public void decoderTest() throws IOException, UnsupportedAudioFileException {
LibVosk.setLogLevel(LogLevel.DEBUG);
try (Model model = new Model("model");
InputStream ais = AudioSystem.getAudioInputStream(new BufferedInputStream(new FileInputStream("../../python/example/test.wav")));
Recognizer recognizer = new Recognizer(model, 16000)) {
recognizer.setMaxAlternatives(10);
recognizer.setWords(true);
int nbytes;
byte[] b = new byte[4096];
while ((nbytes = ais.read(b)) >= 0) {
if (recognizer.acceptWaveForm(b, nbytes)) {
System.out.println(recognizer.getResult());
} else {
System.out.println(recognizer.getPartialResult());
}
}
System.out.println(recognizer.getFinalResult());
}
Assert.assertTrue(true);
}
@Test
public void decoderTestShort() throws IOException, UnsupportedAudioFileException {
LibVosk.setLogLevel(LogLevel.DEBUG);
try (Model model = new Model("model");
InputStream ais = AudioSystem.getAudioInputStream(new BufferedInputStream(new FileInputStream("../../python/example/test.wav")));
Recognizer recognizer = new Recognizer(model, 16000)) {
int nbytes;
byte[] b = new byte[4096];
short[] s = new short[2048];
while ((nbytes = ais.read(b)) >= 0) {
ByteBuffer.wrap(b).order(ByteOrder.LITTLE_ENDIAN).asShortBuffer().get(s);
if (recognizer.acceptWaveForm(s, nbytes / 2)) {
System.out.println(recognizer.getResult());
} else {
System.out.println(recognizer.getPartialResult());
}
}
System.out.println(recognizer.getFinalResult());
}
Assert.assertTrue(true);
}
@Test
public void decoderTestGrammar() throws IOException, UnsupportedAudioFileException {
LibVosk.setLogLevel(LogLevel.DEBUG);
try (Model model = new Model("model");
InputStream ais = AudioSystem.getAudioInputStream(new BufferedInputStream(new FileInputStream("../../python/example/test.wav")));
Recognizer recognizer = new Recognizer(model, 16000, "[\"one two three four five six seven eight nine zero oh\"]")) {
int nbytes;
byte[] b = new byte[4096];
while ((nbytes = ais.read(b)) >= 0) {
if (recognizer.acceptWaveForm(b, nbytes)) {
System.out.println(recognizer.getResult());
} else {
System.out.println(recognizer.getPartialResult());
}
}
System.out.println(recognizer.getFinalResult());
}
Assert.assertTrue(true);
}
}
+40
View File
@@ -0,0 +1,40 @@
package test;
import java.io.File;
import java.io.FileInputStream;
import java.io.FileOutputStream;
import java.io.DataOutputStream;
import java.io.IOException;
import java.net.URL;
import java.nio.*;
import org.kaldi.KaldiRecognizer;
import org.kaldi.Model;
import org.kaldi.SpkModel;
import org.kaldi.Vosk;
public class DecoderTest {
static {
System.loadLibrary("vosk_jni");
}
public static void main(String args[]) throws IOException {
Vosk.SetLogLevel(-10);
FileInputStream ais = new FileInputStream(new File("../python/example/test.wav"));
Model model = new Model("model");
SpkModel spkModel = new SpkModel("model-spk");
KaldiRecognizer rec = new KaldiRecognizer(model, spkModel, 16000.0f);
int nbytes;
byte[] b = new byte[4096];
while ((nbytes = ais.read(b)) >= 0) {
if (rec.AcceptWaveform(b)) {
System.out.println(rec.Result());
} else {
System.out.println(rec.PartialResult());
}
}
System.out.println(rec.FinalResult());
}
}
+3 -3
View File
@@ -1,3 +1,3 @@
demo/model
demo/model-spk
demo/test.wav
build
node_modules
vosk_wrap.cc
+14 -27
View File
@@ -1,33 +1,20 @@
This is an FFI-NAPI wrapper for the Vosk library.
Installation requires vosk-api checkout, it doesn't yet work with `npm
install vosk`. We have to figure out how to properly distribute native
modules for Vosk.
## Usage
The build tested with node-0.10.15, node-0.12 is not yet supported by swig.
It mostly follows Vosk interface, some methods are not yet fully implemented.
Still, you need swig of newest version 4.0.1
To use it you need to compile libvosk library, see Python module build
instructions for details. You can find prebuilt library inside python
wheel.
Build like this
## About
```
npm install --kaldi_root=/home/suser/kaldi
```
Vosk is an offline open source speech recognition toolkit. It enables
speech recognition models for 17 languages and dialects - English, Indian
English, German, French, Spanish, Portuguese, Chinese, Russian, Turkish,
Vietnamese, Italian, Dutch, Catalan, Arabic, Greek, Farsi, Filipino.
Then test with
Vosk models are small (50 Mb) but provide continuous large vocabulary
transcription, zero-latency response with streaming API, reconfigurable
vocabulary and speaker identification.
Vosk supplies speech recognition for chatbots, smart home appliances,
virtual assistants. It can also create subtitles for movies,
transcription for lectures and interviews.
Vosk scales from small devices like Raspberry Pi or Android smartphone to
big clusters.
# Documentation
For installation instructions, examples and documentation visit [Vosk
Website](https://alphacephei.com/vosk). See also our project on
[Github](https://github.com/alphacep/vosk-api).
```
cd example
node test.js
```
+70
View File
@@ -0,0 +1,70 @@
{
'targets': [
{
'target_name': 'vosk',
'sources': [
'../src/kaldi_recognizer.cc',
'../src/model.cc',
'../src/spk_model.cc',
'../src/vosk_api.cc',
'vosk_wrap.cc',
],
'cflags': [
'-std=c++11',
'-DFST_NO_DYNAMIC_LINKING',
'-Wno-deprecated-declarations',
'-Wno-sign-compare',
'-Wno-unused-local-typedefs',
'-Wno-ignored-quaifiers',
'-Wno-extra',
],
'cflags_cc!' : [
'-fno-rtti',
'-fno-exceptions',
],
'actions': [
{
'action_name': 'swig',
'inputs': [
'../src/vosk.i',
],
'outputs': [
'vosk_wrap.cc',
],
'action': ['swig', '-c++', '-javascript', '-o', 'vosk_wrap.cc', '-v8', '-DV8_MAJOR_VERSION=10', '../src/vosk.i']
},
],
'include_dirs': [
'<@(kaldi_root)/src',
'<@(kaldi_root)/tools/openfst/include',
'../src',
],
'link_settings': {
'libraries': [
'<@(kaldi_root)/src/online2/kaldi-online2.a',
'<@(kaldi_root)/src/decoder/kaldi-decoder.a',
'<@(kaldi_root)/src/ivector/kaldi-ivector.a',
'<@(kaldi_root)/src/gmm/kaldi-gmm.a',
'<@(kaldi_root)/src/nnet3/kaldi-nnet3.a',
'<@(kaldi_root)/src/tree/kaldi-tree.a',
'<@(kaldi_root)/src/feat/kaldi-feat.a',
'<@(kaldi_root)/src/lat/kaldi-lat.a',
'<@(kaldi_root)/src/lm/kaldi-lm.a',
'<@(kaldi_root)/src/hmm/kaldi-hmm.a',
'<@(kaldi_root)/src/transform/kaldi-transform.a',
'<@(kaldi_root)/src/cudamatrix/kaldi-cudamatrix.a',
'<@(kaldi_root)/src/matrix/kaldi-matrix.a',
'<@(kaldi_root)/src/fstext/kaldi-fstext.a',
'<@(kaldi_root)/src/util/kaldi-util.a',
'<@(kaldi_root)/src/base/kaldi-base.a',
'<@(kaldi_root)/tools/openfst/lib/libfst.a',
'<@(kaldi_root)/tools/openfst/lib/libfstngram.a',
'<@(kaldi_root)/tools/OpenBLAS/libopenblas.a',
],
'library_dirs': [
'/usr/lib',
],
},
}
]
}
-33
View File
@@ -1,33 +0,0 @@
var vosk = require('..')
const fs = require("fs");
const { spawn } = require("child_process");
MODEL_PATH = "model"
FILE_NAME = "test.wav"
SAMPLE_RATE = 16000
BUFFER_SIZE = 4000
if (!fs.existsSync(MODEL_PATH)) {
console.log("Please download the model from https://alphacephei.com/vosk/models and unpack as " + MODEL_PATH + " in the current folder.")
process.exit()
}
if (process.argv.length > 2)
FILE_NAME = process.argv[2]
vosk.setLogLevel(0);
const model = new vosk.Model(MODEL_PATH);
const rec = new vosk.Recognizer({model: model, sampleRate: SAMPLE_RATE});
const ffmpeg_run = spawn('ffmpeg', ['-loglevel', 'quiet', '-i', FILE_NAME,
'-ar', String(SAMPLE_RATE) , '-ac', '1',
'-f', 's16le', '-bufsize', String(BUFFER_SIZE) , '-']);
ffmpeg_run.stdout.on('data', (stdout) => {
if (rec.acceptWaveform(stdout))
console.log(rec.result());
else
console.log(rec.partialResult());
console.log(rec.finalResult());
});
-39
View File
@@ -1,39 +0,0 @@
var vosk = require('..')
const fs = require("fs");
var mic = require("mic");
MODEL_PATH = "model"
SAMPLE_RATE = 16000
if (!fs.existsSync(MODEL_PATH)) {
console.log("Please download the model from https://alphacephei.com/vosk/models and unpack as " + MODEL_PATH + " in the current folder.")
process.exit()
}
vosk.setLogLevel(0);
const model = new vosk.Model(MODEL_PATH);
const rec = new vosk.Recognizer({model: model, sampleRate: SAMPLE_RATE});
var micInstance = mic({
rate: String(SAMPLE_RATE),
channels: '1',
debug: false
});
var micInputStream = micInstance.getAudioStream();
micInstance.start();
micInputStream.on('data', data => {
if (rec.acceptWaveform(data))
console.log(rec.result());
else
console.log(rec.partialResult());
});
process.on('SIGINT', function() {
console.log(rec.finalResult());
console.log("\nDone");
rec.free();
model.free();
});
-45
View File
@@ -1,45 +0,0 @@
var vosk = require('..')
const fs = require("fs");
const { Readable } = require("stream");
const wav = require("wav");
MODEL_PATH = "model"
FILE_NAME = "test.wav"
if (!fs.existsSync(MODEL_PATH)) {
console.log("Please download the model from https://alphacephei.com/vosk/models and unpack as " + MODEL_PATH + " in the current folder.")
process.exit()
}
if (process.argv.length > 2)
FILE_NAME = process.argv[2]
vosk.setLogLevel(0);
const model = new vosk.Model(MODEL_PATH);
const wfReader = new wav.Reader();
const wfReadable = new Readable().wrap(wfReader);
wfReader.on('format', async ({ audioFormat, sampleRate, channels }) => {
if (audioFormat != 1 || channels != 1) {
console.error("Audio file must be WAV format mono PCM.");
process.exit(1);
}
const rec = new vosk.Recognizer({model: model, sampleRate: sampleRate});
rec.setMaxAlternatives(10);
rec.setWords(true);
for await (const data of wfReadable) {
const end_of_speech = rec.acceptWaveform(data);
if (end_of_speech) {
console.log(JSON.stringify(rec.result(), null, 4));
}
}
console.log(JSON.stringify(rec.finalResult(rec), null, 4));
rec.free();
});
fs.createReadStream(FILE_NAME, {'highWaterMark': 4096}).pipe(wfReader).on('finish',
function (err) {
model.free();
});
-46
View File
@@ -1,46 +0,0 @@
var vosk = require('..')
const async = require("async");
const fs = require("fs");
const { Readable } = require("stream");
const wav = require("wav");
MODEL_PATH = "model"
if (!fs.existsSync(MODEL_PATH)) {
console.log("Please download the model from https://alphacephei.com/vosk/models and unpack as " + MODEL_PATH + " in the current folder.")
process.exit()
}
// Process file 4 times in parallel with a single model
files = Array(10).fill("test.wav")
const model = new vosk.Model(MODEL_PATH)
async.filter(files, function(filePath, callback) {
const wfReader = new wav.Reader();
const wfReadable = new Readable().wrap(wfReader);
wfReader.on('format', async ({ audioFormat, sampleRate, channels }) => {
const rec = new vosk.Recognizer({model: model, sampleRate: sampleRate});
if (audioFormat != 1 || channels != 1) {
console.error("Audio file must be WAV format mono PCM.");
process.exit(1);
}
for await (const data of wfReadable) {
const end_of_speech = await rec.acceptWaveformAsync(data);
if (end_of_speech) {
console.log(rec.result());
}
}
console.log(rec.finalResult(rec));
rec.free();
// Signal we are done without errors
callback(null, true);
});
fs.createReadStream(filePath, {'highWaterMark': 4096}).pipe(wfReader);
}, function(err, results) {
model.free();
console.log("Done!!!!!");
});
-53
View File
@@ -1,53 +0,0 @@
const vosk = require('..');
const fs = require("fs");
const { Readable } = require("stream");
const wav = require("wav");
MODEL_PATH = "model"
SPEAKER_MODEL_PATH = "model-spk"
FILE_NAME = "test.wav"
if (!fs.existsSync(MODEL_PATH)) {
console.log("Please download the model from https://alphacephei.com/vosk/models and unpack as " + MODEL_PATH + " in the current folder.")
process.exit()
}
if (!fs.existsSync(SPEAKER_MODEL_PATH)) {
console.log("Please download the speaker model from https://alphacephei.com/vosk/models and unpack as " + SPEAKER_MODEL_PATH + " in the current folder.")
process.exit()
}
if (process.argv.length > 2)
FILE_NAME = process.argv[2]
const model = new vosk.Model(MODEL_PATH);
const speakerModel = new vosk.SpeakerModel(SPEAKER_MODEL_PATH);
const wfReader = new wav.Reader();
const wfReadable = new Readable().wrap(wfReader);
wfReader.on('format', async ({ audioFormat, sampleRate, channels }) => {
if (audioFormat != 1 || channels != 1) {
console.error('Audio file must be WAV format mono PCM.');
process.exit(1);
}
// const rec = new vosk.Recognizer({ model: model,
// speakerModel: speakerModel,
// sampleRate: sampleRate });
const rec = new vosk.Recognizer({model: model, sampleRate: sampleRate});
rec.setSpkModel(speakerModel);
for await (const data of wfReadable) {
const end_of_speech = rec.acceptWaveform(data);
if (end_of_speech) {
console.log(rec.finalResult());
}
}
console.log(rec.finalResult());
rec.free();
});
fs.createReadStream(FILE_NAME, { highWaterMark: 4096 }).pipe(wfReader).on('finish', function (err) {
model.free();
speakerModel.free();
});
-84
View File
@@ -1,84 +0,0 @@
var vosk = require('..')
const fs = require("fs");
const { spawn } = require("child_process");
const { stringifySync } = require('subtitle')
MODEL_PATH = "model"
FILE_NAME = "test.wav"
SAMPLE_RATE = 16000
BUFFER_SIZE = 4000
if (!fs.existsSync(MODEL_PATH)) {
console.log("Please download the model from https://alphacephei.com/vosk/models and unpack as " + MODEL_PATH + " in the current folder.")
process.exit()
}
if (process.argv.length > 2)
FILE_NAME = process.argv[2]
vosk.setLogLevel(-1);
const model = new vosk.Model(MODEL_PATH);
const rec = new vosk.Recognizer({model: model, sampleRate: SAMPLE_RATE});
rec.setWords(true);
const ffmpeg_run = spawn('ffmpeg', ['-loglevel', 'quiet', '-i', FILE_NAME,
'-ar', String(SAMPLE_RATE) , '-ac', '1',
'-f', 's16le', '-bufsize', String(BUFFER_SIZE), '-']);
WORDS_PER_LINE = 7
const subs = []
const results = []
ffmpeg_run.stdout.on('data', (stdout) => {
if (rec.acceptWaveform(stdout))
results.push(rec.result());
results.push(rec.finalResult());
});
ffmpeg_run.on('exit', code => {
rec.free();
model.free();
results.forEach(element =>{
if (!element.hasOwnProperty('result'))
return;
const words = element.result;
if (words.length == 1) {
subs.push({
type: 'cue',
data: {
start: words[0].start,
end: words[0].end,
text: words[0].word
}
});
return;
}
var start_index = 0;
var text = words[0].word + " ";
for (let i = 1; i < words.length; i++) {
text += words[i].word + " ";
if (i % WORDS_PER_LINE == 0) {
subs.push({
type: 'cue',
data: {
start: words[start_index].start,
end: words[i].end,
text: text.slice(0, text.length-1)
}
});
start_index = i;
text = "";
}
}
if (start_index != words.length - 1)
subs.push({
type: 'cue',
data: {
start: words[start_index].start,
end: words[words.length-1].end,
text: text
}
});
});
console.log(stringifySync(subs, {format: "SRT"}));
});
+90
View File
@@ -0,0 +1,90 @@
const wav = require('wav')
const fs = require('fs')
const {Readable} = require('stream')
const {Model, KaldiRecognizer, SpkModel} = require('..')
try {
fs.accessSync('model', fs.constants.R_OK)
} catch (err) {
console.error("Please download the model from https://github.com/alphacep/kaldi-android-demo/releases and unpack as 'model' in the current folder.")
process.exit(1)
}
try {
fs.accessSync('model-spk', fs.constants.R_OK)
} catch (err) {
console.error("Please download the speaker model from https://github.com/alphacep/kaldi-android-demo/releases and unpack as 'model-spk' in the current folder.")
process.exit(1)
}
const wfStream = fs.createReadStream('test.wav', { highWaterMark: 4096 })
const wfReader = new wav.Reader()
const model = new Model('model')
const spkModel = new SpkModel('model-spk')
const spk_sig = [4.658117, 1.277387, 3.346158, -1.473036, -2.15727,
2.461757, 3.76756, -1.241252, 2.333765, 0.642588, -2.848165, 1.229534,
3.907015, 1.726496, -1.188692, 1.16322, -0.668811, -0.623309, 4.628018,
0.407197, 0.089955, 0.920438, 1.47237, -0.311365, -0.437051, -0.531738,
-1.591781, 3.095415, 0.439524, -0.274787, 4.03165, 2.665864, 4.815553,
1.581063, 1.078242, 5.017717, -0.089395, -3.123428, 5.34038, 0.456982,
2.465727, 2.131833, 4.056272, 1.178392, -2.075712, -1.568503, 0.847139,
0.409214, 1.84727, 0.986758, 4.222116, 2.235512, 1.369377, 4.283126,
2.278125, -1.467577, -0.999971, 3.070041, 1.462214, 0.423204, 2.143578,
0.567174, -2.294655, 1.864723, 4.307356, 2.610872, -1.238721, 0.551861,
2.861954, 0.59613, -0.715396, -1.395357, 2.706177, -2.004444, 2.055255,
0.458283, 1.231968, 3.48234, 2.993858, 0.402819, 0.940885, 0.360162,
-2.173674, -2.504609, 0.329541, 3.653913, 3.638025, -1.406409, 2.14059,
1.662765, -0.991323, 0.770921, 0.010094, 3.775469, 1.847511, 2.074432,
-1.928593, 0.807414, 2.964505, 0.128597, 1.297962, 2.645227, 0.136405,
-2.543087, 0.932246, 2.405783, -2.122267, 3.044013, 0.486728, 4.395338,
0.474267, 0.781297, 1.694144, -0.831078, -0.462362, -0.964715, 3.187863,
6.008708, 1.725954, 3.667886, -1.467623, 3.370667, 2.72555, -0.796541,
2.416543, 0.675401, -0.737634, -1.709676]
function dotp(x, y) {
function dotp_sum(a, b) {
return a + b
}
function dotp_times(a, i) {
return x[i] * y[i]
}
return x.map(dotp_times).reduce(dotp_sum, 0)
}
function cosineSimilarity(A, B) {
var similarity =
dotp(A, B) / (Math.sqrt(dotp(A, A)) * Math.sqrt(dotp(B, B)))
return similarity
}
function cosine_dist(x, y) {
return 1 - cosineSimilarity(x, y)
}
wfReader.on('format', async ({ audioFormat, sampleRate, channels }) => {
if (audioFormat != 1 || channels != 1) {
console.error('Audio file must be WAV format mono PCM.')
process.exit(1)
}
const rec = new KaldiRecognizer(model, spkModel, sampleRate)
for await (const data of new Readable().wrap(wfReader)) {
const endOfSpeech = await rec.AcceptWaveform(data)
if (endOfSpeech) {
res = await JSON.parse(rec.Result());
console.log(res)
console.log('X-vector:', JSON.stringify(res['spk']))
console.log('Speaker distance:', cosine_dist(spk_sig, res['spk']))
} else {
console.log(await rec.PartialResult())
}
}
res = await JSON.parse(rec.FinalResult());
console.log(res)
console.log('X-vector:', JSON.stringify(res['spk']))
console.log('Speaker distance:', cosine_dist(spk_sig, res['spk']))
})
wfStream.pipe(wfReader)
+38
View File
@@ -0,0 +1,38 @@
#!/usr/bin/env node
const fs = require("fs");
const { Readable } = require("stream");
const wav = require("wav");
const { Model, KaldiRecognizer } = require("..");
try {
fs.accessSync("model", fs.constants.R_OK);
} catch(err) {
console.error("Please download the model from https://github.com/alphacep/kaldi-android-demo/releases and unpack as 'model' in the current folder.");
process.exit(1);
}
const wfStream = fs.createReadStream("test.wav", {'highWaterMark': 4096});
const wfReader = new wav.Reader();
const model = new Model("model");
wfReader.on('format', async ({ audioFormat, sampleRate, channels }) => {
if (audioFormat != 1 || channels != 1) {
console.error("Audio file must be WAV format mono PCM.");
process.exit(1);
}
const rec = new KaldiRecognizer(model, sampleRate);
for await (const data of new Readable().wrap(wfReader)) {
const result = await rec.AcceptWaveform(data);
if (result != 0) {
console.log(await rec.Result());
} else {
console.log(await rec.PartialResult());
}
}
console.log(await rec.FinalResult());
});
wfStream.pipe(wfReader);
+2 -426
View File
@@ -1,427 +1,3 @@
// @ts-check
'use strict'
const voskNativeModule = require('./build/Release/vosk.node');
/**
* @module vosk
*/
const os = require('os');
const path = require('path');
/** @type {import('ffi-napi')} */
const ffi = require('ffi-napi');
/** @type {import('ref-napi')} */
const ref = require('ref-napi');
const vosk_model = ref.types.void;
const vosk_model_ptr = ref.refType(vosk_model);
const vosk_spk_model = ref.types.void;
const vosk_spk_model_ptr = ref.refType(vosk_spk_model);
const vosk_recognizer = ref.types.void;
const vosk_recognizer_ptr = ref.refType(vosk_recognizer);
/**
* @typedef {Object} WordResult
* @property {number} conf The confidence rate in the detection. 0 For unlikely, and 1 for totally accurate.
* @property {number} start The start of the timeframe when the word is pronounced in seconds
* @property {number} end The end of the timeframe when the word is pronounced in seconds
* @property {string} word The word detected
*/
/**
* @typedef {Object} RecognitionResults
* @property {WordResult[]} result Details about the words that have been detected
* @property {string} text The complete sentence that have been detected
*/
/**
* @typedef {Object} SpeakerResults
* @property {number[]} spk A floating vector representing speaker identity. It is usually about 128 numbers which uniquely represent speaker voice.
* @property {number} spk_frames The number of frames used to extract speaker vector. The more frames you have the more reliable is speaker vector.
*/
/**
* @typedef {Object} BaseRecognizerParam
* @property {Model} model The language model to be used
* @property {number} sampleRate The sample rate. Most models are trained at 16kHz
*/
/**
* @typedef {Object} GrammarRecognizerParam
* @property {string[]} grammar The list of sentences to be recognized.
*/
/**
* @typedef {Object} SpeakerRecognizerParam
* @property {SpeakerModel} speakerModel The SpeakerModel that will enable speaker identification
*/
/**
* @template {SpeakerRecognizerParam | GrammarRecognizerParam} T
* @typedef {T extends SpeakerRecognizerParam ? SpeakerResults & RecognitionResults : RecognitionResults} Result
*/
/**
* @typedef {Object} PartialResults
* @property {string} partial The partial sentence that have been detected until now
*/
/** @typedef {string[]} Grammar The list of strings to be recognized */
let soname;
if (os.platform() == 'win32') {
// Update path to load dependent dlls
let currentPath = process.env.Path;
let dllDirectory = path.resolve(path.join(__dirname, "lib", "win-x86_64"));
process.env.Path = currentPath + path.delimiter + dllDirectory;
soname = path.join(__dirname, "lib", "win-x86_64", "libvosk.dll")
} else if (os.platform() == 'darwin') {
soname = path.join(__dirname, "lib", "osx-x86_64", "libvosk.dylib")
} else {
soname = path.join(__dirname, "lib", "linux-x86_64", "libvosk.so")
}
const libvosk = ffi.Library(soname, {
'vosk_set_log_level': ['void', ['int']],
'vosk_model_new': [vosk_model_ptr, ['string']],
'vosk_model_free': ['void', [vosk_model_ptr]],
'vosk_spk_model_new': [vosk_spk_model_ptr, ['string']],
'vosk_spk_model_free': ['void', [vosk_spk_model_ptr]],
'vosk_recognizer_new': [vosk_recognizer_ptr, [vosk_model_ptr, 'float']],
'vosk_recognizer_new_spk': [vosk_recognizer_ptr, [vosk_model_ptr, 'float', vosk_spk_model_ptr]],
'vosk_recognizer_new_grm': [vosk_recognizer_ptr, [vosk_model_ptr, 'float', 'string']],
'vosk_recognizer_free': ['void', [vosk_recognizer_ptr]],
'vosk_recognizer_set_max_alternatives': ['void', [vosk_recognizer_ptr, 'int']],
'vosk_recognizer_set_words': ['void', [vosk_recognizer_ptr, 'bool']],
'vosk_recognizer_set_spk_model': ['void', [vosk_recognizer_ptr, vosk_spk_model_ptr]],
'vosk_recognizer_accept_waveform': ['bool', [vosk_recognizer_ptr, 'pointer', 'int']],
'vosk_recognizer_result': ['string', [vosk_recognizer_ptr]],
'vosk_recognizer_final_result': ['string', [vosk_recognizer_ptr]],
'vosk_recognizer_partial_result': ['string', [vosk_recognizer_ptr]],
'vosk_recognizer_reset': ['void', [vosk_recognizer_ptr]],
});
/**
* Set log level for Kaldi messages
* @param {number} level The higher, the more verbose. 0 for infos and errors. Less than 0 for silence.
*/
function setLogLevel(level) {
libvosk.vosk_set_log_level(level);
}
/**
* Build a Model from a model file.
* @see models [models](https://alphacephei.com/vosk/models)
*/
class Model {
/**
* Build a Model to be used with the voice recognition. Each language should have it's own Model
* for the speech recognition to work.
* @param {string} modelPath The abstract pathname to the model
* @see models [models](https://alphacephei.com/vosk/models)
*/
constructor(modelPath) {
/**
* Store the handle.
* For internal use only
* @type {unknown}
*/
this.handle = libvosk.vosk_model_new(modelPath);
}
/**
* Releases the model memory
*
* The model object is reference-counted so if some recognizer
* depends on this model, model might still stay alive. When
* last recognizer is released, model will be released too.
*/
free() {
libvosk.vosk_model_free(this.handle);
}
}
/**
* Build a Speaker Model from a speaker model file.
* The Speaker Model enables speaker identification.
* @see models [models](https://alphacephei.com/vosk/models)
*/
class SpeakerModel {
/**
* Loads speaker model data from the file and returns the model object
*
* @param {string} modelPath the path of the model on the filesystem
* @see models [models](https://alphacephei.com/vosk/models)
*/
constructor(modelPath) {
/**
* Store the handle.
* For internal use only
* @type {unknown}
*/
this.handle = libvosk.vosk_spk_model_new(modelPath);
}
/**
* Releases the model memory
*
* The model object is reference-counted so if some recognizer
* depends on this model, model might still stay alive. When
* last recognizer is released, model will be released too.
*/
free() {
libvosk.vosk_spk_model_free(this.handle);
}
}
/**
* Helper to narrow down type while using `hasOwnProperty`.
* @see hasOwnProperty [typescript issue](https://fettblog.eu/typescript-hasownproperty/)
* @template {Object} Obj
* @template {PropertyKey} Key
* @param {Obj} obj
* @param {Key} prop
* @returns {obj is Obj & Record<Key, unknown>}
*/
function hasOwnProperty(obj, prop) {
return obj.hasOwnProperty(prop)
}
/**
* @template T
* @template U
* @typedef {{ [P in Exclude<keyof T, keyof U>]?: never }} Without
*/
/**
* @template T
* @template U
* @typedef {(T | U) extends object ? (Without<T, U> & U) | (Without<U, T> & T) : T | U} XOR
*/
/**
* Create a Recognizer that will be able to transform audio streams into text using a Model.
* @template {XOR<SpeakerRecognizerParam, Partial<GrammarRecognizerParam>>} T extra parameter
* @see Model
*/
class Recognizer {
/**
* Create a Recognizer that will handle speech to text recognition.
* @constructor
* @param {T & BaseRecognizerParam} param The Recognizer parameters
*
* Sometimes when you want to improve recognition accuracy and when you don't need
* to recognize large vocabulary you can specify a list of phrases to recognize. This
* will improve recognizer speed and accuracy but might return [unk] if user said
* something different.
*
* Only recognizers with lookahead models support this type of quick configuration.
* Precompiled HCLG graph models are not supported.
*/
constructor(param) {
const { model, sampleRate } = param
// Prevent the user to receive unpredictable results
if (hasOwnProperty(param, 'speakerModel') && hasOwnProperty(param, 'grammar')) {
throw new Error('grammar and speakerModel cannot be used together for now.')
}
/**
* Store the handle.
* For internal use only
* @type {unknown}
*/
this.handle = hasOwnProperty(param, 'speakerModel')
? libvosk.vosk_recognizer_new_spk(model.handle, sampleRate, param.speakerModel.handle)
: hasOwnProperty(param, 'grammar')
? libvosk.vosk_recognizer_new_grm(model.handle, sampleRate, JSON.stringify(param.grammar))
: libvosk.vosk_recognizer_new(model.handle, sampleRate);
}
/**
* Releases the model memory
*
* The model object is reference-counted so if some recognizer
* depends on this model, model might still stay alive. When
* last recognizer is released, model will be released too.
*/
free() {
libvosk.vosk_recognizer_free(this.handle);
}
/** Configures recognizer to output n-best results
*
* <pre>
* {
* "alternatives": [
* { "text": "one two three four five", "confidence": 0.97 },
* { "text": "one two three for five", "confidence": 0.03 },
* ]
* }
* </pre>
*
* @param max_alternatives - maximum alternatives to return from recognition results
*/
setMaxAlternatives(max_alternatives) {
libvosk.vosk_recognizer_set_max_alternatives(this.handle, max_alternatives);
}
/** Configures recognizer to output words with times
*
* <pre>
* "result" : [{
* "conf" : 1.000000,
* "end" : 1.110000,
* "start" : 0.870000,
* "word" : "what"
* }, {
* "conf" : 1.000000,
* "end" : 1.530000,
* "start" : 1.110000,
* "word" : "zero"
* }, {
* "conf" : 1.000000,
* "end" : 1.950000,
* "start" : 1.530000,
* "word" : "zero"
* }, {
* "conf" : 1.000000,
* "end" : 2.340000,
* "start" : 1.950000,
* "word" : "zero"
* }, {
* "conf" : 1.000000,
* "end" : 2.610000,
* "start" : 2.340000,
* "word" : "one"
* }],
* </pre>
*
* @param words - boolean value
*/
setWords(words) {
libvosk.vosk_recognizer_set_words(this.handle, words);
}
/** Adds speaker recognition model to already created recognizer. Helps to initialize
* speaker recognition for grammar-based recognizer.
*
* @param spk_model Speaker recognition model
*/
setSpkModel(spk_model) {
libvosk.vosk_recognizer_set_spk_model(this.handle, spk_model.handle);
}
/**
* Accept voice data
*
* accept and process new chunk of voice data
*
* @param {Buffer} data audio data in PCM 16-bit mono format
* @returns true if silence is occured and you can retrieve a new utterance with result method
*/
acceptWaveform(data) {
return libvosk.vosk_recognizer_accept_waveform(this.handle, data, data.length);
};
/**
* Accept voice data
*
* accept and process new chunk of voice data
*
* @param {Buffer} data audio data in PCM 16-bit mono format
* @returns true if silence is occured and you can retrieve a new utterance with result method
*/
acceptWaveformAsync(data) {
return new Promise((resolve, reject) => {
libvosk.vosk_recognizer_accept_waveform.async(this.handle, data, data.length, function(err, result) {
if (err) {
reject(err);
} else {
resolve(result);
}
});
});
};
/** Returns speech recognition result in a string
*
* @returns the result in JSON format which contains decoded line, decoded
* words, times in seconds and confidences. You can parse this result
* with any json parser
* <pre>
* {
* "result" : [{
* "conf" : 1.000000,
* "end" : 1.110000,
* "start" : 0.870000,
* "word" : "what"
* }, {
* "conf" : 1.000000,
* "end" : 1.530000,
* "start" : 1.110000,
* "word" : "zero"
* }, {
* "conf" : 1.000000,
* "end" : 1.950000,
* "start" : 1.530000,
* "word" : "zero"
* }, {
* "conf" : 1.000000,
* "end" : 2.340000,
* "start" : 1.950000,
* "word" : "zero"
* }, {
* "conf" : 1.000000,
* "end" : 2.610000,
* "start" : 2.340000,
* "word" : "one"
* }],
* "text" : "what zero zero zero one"
* }
* </pre>
*/
resultString() {
return libvosk.vosk_recognizer_result(this.handle);
};
/**
* Returns speech recognition results
* @returns {Result<T>} The results
*/
result() {
return JSON.parse(libvosk.vosk_recognizer_result(this.handle));
};
/**
* speech recognition text which is not yet finalized.
* result may change as recognizer process more data.
*
* @returns {PartialResults} The partial results
*/
partialResult() {
return JSON.parse(libvosk.vosk_recognizer_partial_result(this.handle));
};
/**
* Returns speech recognition result. Same as result, but doesn't wait for silence
* You usually call it in the end of the stream to get final bits of audio. It
* flushes the feature pipeline, so all remaining audio chunks got processed.
*
* @returns {Result<T>} speech result.
*/
finalResult() {
return JSON.parse(libvosk.vosk_recognizer_final_result(this.handle));
};
/**
*
* Resets current results so the recognition can continue from scratch
*/
reset() {
libvosk.vosk_recognizer_reset(this.handle);
}
}
exports.setLogLevel = setLogLevel
exports.Model = Model
exports.SpeakerModel = SpeakerModel
exports.Recognizer = Recognizer
module.exports = voskNativeModule;
+4 -7
View File
@@ -1,7 +1,7 @@
{
"name": "vosk",
"version": "0.3.30",
"description": "Node binding for continuous offline voice recoginition with Vosk library.",
"version": "0.3.8",
"description": "Node binding for continuous voice recoginition through pocketsphinx.",
"repository": {
"type": "git",
"url": "git://github.com/alphacep/vosk-api.git"
@@ -15,13 +15,10 @@
"author": "Alpha Cephei Inc.",
"license": "Apache-2.0",
"engines": {
"node": ">= 12.x.x"
"node": ">= 10.x.x"
},
"dependencies": {
"async": "^3.2.0",
"ffi-napi": "^4.0.3",
"mic": "^2.1.2",
"ref-napi": ">=2.0.0",
"node-gyp": "^5.1.1",
"wav": "^1.0.2"
}
}
+2 -22
View File
@@ -1,23 +1,3 @@
This is a Python module for Vosk.
Python module for vosk-api
Vosk is an offline open source speech recognition toolkit. It enables
speech recognition models for 17 languages and dialects - English, Indian
English, German, French, Spanish, Portuguese, Chinese, Russian, Turkish,
Vietnamese, Italian, Dutch, Catalan, Arabic, Greek, Farsi, Filipino.
Vosk models are small (50 Mb) but provide continuous large vocabulary
transcription, zero-latency response with streaming API, reconfigurable
vocabulary and speaker identification.
Vosk supplies speech recognition for chatbots, smart home appliances,
virtual assistants. It can also create subtitles for movies,
transcription for lectures and interviews.
Vosk scales from small devices like Raspberry Pi or Android smartphone to
big clusters.
# Documentation
For installation instructions, examples and documentation visit [Vosk
Website](https://alphacephei.com/vosk). See also our project on
[Github](https://github.com/alphacep/vosk-api).
See for details https://github.com/alphacep/vosk-api
-34
View File
@@ -1,34 +0,0 @@
#!/usr/bin/env python3
from vosk import Model, KaldiRecognizer, SetLogLevel
import sys
import os
import wave
import json
SetLogLevel(0)
if not os.path.exists("model"):
print ("Please download the model from https://alphacephei.com/vosk/models and unpack as 'model' in the current folder.")
exit (1)
wf = wave.open(sys.argv[1], "rb")
if wf.getnchannels() != 1 or wf.getsampwidth() != 2 or wf.getcomptype() != "NONE":
print ("Audio file must be WAV format mono PCM.")
exit (1)
model = Model("model")
rec = KaldiRecognizer(model, wf.getframerate())
rec.SetMaxAlternatives(10)
rec.SetWords(True)
while True:
data = wf.readframes(4000)
if len(data) == 0:
break
if rec.AcceptWaveform(data):
print(json.loads(rec.Result()))
else:
print(json.loads(rec.PartialResult()))
print(json.loads(rec.FinalResult()))
+1 -1
View File
@@ -9,7 +9,7 @@ import subprocess
SetLogLevel(0)
if not os.path.exists("model"):
print ("Please download the model from https://alphacephei.com/vosk/models and unpack as 'model' in the current folder.")
print ("Please download the model from https://github.com/alphacep/vosk-api/blob/master/doc/models.md and unpack as 'model' in the current folder.")
exit (1)
sample_rate=16000
+18 -79
View File
@@ -1,89 +1,28 @@
#!/usr/bin/env python3
import argparse
from vosk import Model, KaldiRecognizer
import os
import queue
import sounddevice as sd
import vosk
import sys
q = queue.Queue()
if not os.path.exists("model"):
print ("Please download the model from https://github.com/alphacep/vosk-api/blob/master/doc/models.md and unpack as 'model' in the current folder.")
exit (1)
def int_or_str(text):
"""Helper function for argument parsing."""
try:
return int(text)
except ValueError:
return text
import pyaudio
def callback(indata, frames, time, status):
"""This is called (from a separate thread) for each audio block."""
if status:
print(status, file=sys.stderr)
q.put(bytes(indata))
model = Model("model")
rec = KaldiRecognizer(model, 16000)
parser = argparse.ArgumentParser(add_help=False)
parser.add_argument(
'-l', '--list-devices', action='store_true',
help='show list of audio devices and exit')
args, remaining = parser.parse_known_args()
if args.list_devices:
print(sd.query_devices())
parser.exit(0)
parser = argparse.ArgumentParser(
description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter,
parents=[parser])
parser.add_argument(
'-f', '--filename', type=str, metavar='FILENAME',
help='audio file to store recording to')
parser.add_argument(
'-m', '--model', type=str, metavar='MODEL_PATH',
help='Path to the model')
parser.add_argument(
'-d', '--device', type=int_or_str,
help='input device (numeric ID or substring)')
parser.add_argument(
'-r', '--samplerate', type=int, help='sampling rate')
args = parser.parse_args(remaining)
p = pyaudio.PyAudio()
stream = p.open(format=pyaudio.paInt16, channels=1, rate=16000, input=True, frames_per_buffer=8000)
stream.start_stream()
try:
if args.model is None:
args.model = "model"
if not os.path.exists(args.model):
print ("Please download a model for your language from https://alphacephei.com/vosk/models")
print ("and unpack as 'model' in the current folder.")
parser.exit(0)
if args.samplerate is None:
device_info = sd.query_devices(args.device, 'input')
# soundfile expects an int, sounddevice provides a float:
args.samplerate = int(device_info['default_samplerate'])
model = vosk.Model(args.model)
if args.filename:
dump_fn = open(args.filename, "wb")
while True:
data = stream.read(4000)
if len(data) == 0:
break
if rec.AcceptWaveform(data):
print(rec.Result())
else:
dump_fn = None
print(rec.PartialResult())
with sd.RawInputStream(samplerate=args.samplerate, blocksize = 8000, device=args.device, dtype='int16',
channels=1, callback=callback):
print('#' * 80)
print('Press Ctrl+C to stop the recording')
print('#' * 80)
rec = vosk.KaldiRecognizer(model, args.samplerate)
while True:
data = q.get()
if rec.AcceptWaveform(data):
print(rec.Result())
else:
print(rec.PartialResult())
if dump_fn is not None:
dump_fn.write(data)
except KeyboardInterrupt:
print('\nDone')
parser.exit(0)
except Exception as e:
parser.exit(type(e).__name__ + ': ' + str(e))
print(rec.FinalResult())
-37
View File
@@ -1,37 +0,0 @@
#!/usr/bin/env python3
from vosk import Model, KaldiRecognizer, SetLogLevel
import sys
import os
import wave
import json
SetLogLevel(0)
if not os.path.exists("model"):
print ("Please download the model from https://alphacephei.com/vosk/models and unpack as 'model' in the current folder.")
exit (1)
wf = wave.open(sys.argv[1], "rb")
if wf.getnchannels() != 1 or wf.getsampwidth() != 2 or wf.getcomptype() != "NONE":
print ("Audio file must be WAV format mono PCM.")
exit (1)
model = Model("model")
rec = KaldiRecognizer(model, wf.getframerate())
while True:
data = wf.readframes(4000)
if len(data) == 0:
break
if rec.AcceptWaveform(data):
print(rec.Result())
break
else:
jres = json.loads(rec.PartialResult())
print(jres)
if jres['partial'] == "one zero zero zero":
print("We can reset recognizer here and start over")
rec.Reset();
+1 -2
View File
@@ -8,7 +8,7 @@ import wave
SetLogLevel(0)
if not os.path.exists("model"):
print ("Please download the model from https://alphacephei.com/vosk/models and unpack as 'model' in the current folder.")
print ("Please download the model from https://github.com/alphacep/vosk-api/blob/master/doc/models.md and unpack as 'model' in the current folder.")
exit (1)
wf = wave.open(sys.argv[1], "rb")
@@ -18,7 +18,6 @@ if wf.getnchannels() != 1 or wf.getsampwidth() != 2 or wf.getcomptype() != "NONE
model = Model("model")
rec = KaldiRecognizer(model, wf.getframerate())
rec.SetWords(True)
while True:
data = wf.readframes(4000)
+7 -14
View File
@@ -11,11 +11,11 @@ model_path = "model"
spk_model_path = "model-spk"
if not os.path.exists(model_path):
print ("Please download the model from https://alphacephei.com/vosk/models and unpack as {} in the current folder.".format(model_path))
print ("Please download the model from https://github.com/alphacep/vosk-api/blob/master/doc/models.md and unpack as {} in the current folder.".format(model_path))
exit (1)
if not os.path.exists(spk_model_path):
print ("Please download the speaker model from https://alphacephei.com/vosk/models and unpack as {} in the current folder.".format(spk_model_path))
print ("Please download the speaker model from https://github.com/alphacep/vosk-api/blob/master/doc/models.md and unpack as {} in the current folder.".format(spk_model_path))
exit (1)
wf = wave.open(sys.argv[1], "rb")
@@ -26,13 +26,11 @@ if wf.getnchannels() != 1 or wf.getsampwidth() != 2 or wf.getcomptype() != "NONE
# Large vocabulary free form recognition
model = Model(model_path)
spk_model = SpkModel(spk_model_path)
#rec = KaldiRecognizer(model, wf.getframerate(), spk_model)
rec = KaldiRecognizer(model, wf.getframerate())
rec.SetSpkModel(spk_model)
rec = KaldiRecognizer(model, spk_model, wf.getframerate())
# We compare speakers with cosine distance. We can keep one or several fingerprints for the speaker in a database
# to distingusih among users.
spk_sig = [-1.110417,0.09703002,1.35658,0.7798632,-0.305457,-0.339204,0.6186931,-0.4521213,0.3982236,-0.004530723,0.7651616,0.6500852,-0.6664245,0.1361499,0.1358056,-0.2887807,-0.1280468,-0.8208137,-1.620276,-0.4628615,0.7870904,-0.105754,0.9739769,-0.3258137,-0.7322628,-0.6212429,-0.5531687,-0.7796484,0.7035915,1.056094,-0.4941756,-0.6521456,-0.2238328,-0.003737517,0.2165709,1.200186,-0.7737719,0.492015,1.16058,0.6135428,-0.7183084,0.3153541,0.3458071,-1.418189,-0.9624157,0.4168292,-1.627305,0.2742135,-0.6166027,0.1962581,-0.6406527,0.4372789,-0.4296024,0.4898657,-0.9531326,-0.2945702,0.7879696,-1.517101,-0.9344181,-0.5049928,-0.005040941,-0.4637912,0.8223695,-1.079849,0.8871287,-0.9732434,-0.5548235,1.879138,-1.452064,-0.1975368,1.55047,0.5941782,-0.52897,1.368219,0.6782904,1.202505,-0.9256122,-0.9718158,-0.9570228,-0.5563112,-1.19049,-1.167985,2.606804,-2.261825,0.01340385,0.2526799,-1.125458,-1.575991,-0.363153,0.3270262,1.485984,-1.769565,1.541829,0.7293826,0.1743717,-0.4759418,1.523451,-2.487134,-1.824067,-0.626367,0.7448186,-1.425648,0.3524166,-0.9903384,3.339342,0.4563958,-0.2876643,1.521635,0.9508078,-0.1398541,0.3867955,-0.7550205,0.6568405,0.09419366,-1.583935,1.306094,-0.3501927,0.1794427,-0.3768163,0.9683866,-0.2442541,-1.696921,-1.8056,-0.6803037,-1.842043,0.3069353,0.9070363,-0.486526]
spk_sig = [4.658117, 1.277387, 3.346158, -1.473036, -2.15727, 2.461757, 3.76756, -1.241252, 2.333765, 0.642588, -2.848165, 1.229534, 3.907015, 1.726496, -1.188692, 1.16322, -0.668811, -0.623309, 4.628018, 0.407197, 0.089955, 0.920438, 1.47237, -0.311365, -0.437051, -0.531738, -1.591781, 3.095415, 0.439524, -0.274787, 4.03165, 2.665864, 4.815553, 1.581063, 1.078242, 5.017717, -0.089395, -3.123428, 5.34038, 0.456982, 2.465727, 2.131833, 4.056272, 1.178392, -2.075712, -1.568503, 0.847139, 0.409214, 1.84727, 0.986758, 4.222116, 2.235512, 1.369377, 4.283126, 2.278125, -1.467577, -0.999971, 3.070041, 1.462214, 0.423204, 2.143578, 0.567174, -2.294655, 1.864723, 4.307356, 2.610872, -1.238721, 0.551861, 2.861954, 0.59613, -0.715396, -1.395357, 2.706177, -2.004444, 2.055255, 0.458283, 1.231968, 3.48234, 2.993858, 0.402819, 0.940885, 0.360162, -2.173674, -2.504609, 0.329541, 3.653913, 3.638025, -1.406409, 2.14059, 1.662765, -0.991323, 0.770921, 0.010094, 3.775469, 1.847511, 2.074432, -1.928593, 0.807414, 2.964505, 0.128597, 1.297962, 2.645227, 0.136405, -2.543087, 0.932246, 2.405783, -2.122267, 3.044013, 0.486728, 4.395338, 0.474267, 0.781297, 1.694144, -0.831078, -0.462362, -0.964715, 3.187863, 6.008708, 1.725954, 3.667886, -1.467623, 3.370667, 2.72555, -0.796541, 2.416543, 0.675401, -0.737634, -1.709676]
def cosine_dist(x, y):
nx = np.array(x)
@@ -46,15 +44,10 @@ while True:
if rec.AcceptWaveform(data):
res = json.loads(rec.Result())
print ("Text:", res['text'])
if 'spk' in res:
print ("X-vector:", res['spk'])
print ("Speaker distance:", cosine_dist(spk_sig, res['spk']), "based on", res['spk_frames'], "frames")
print ("Note that second distance is not very reliable because utterance is too short. Utterances longer than 4 seconds give better xvector")
print ("X-vector:", res['spk'])
print ("Speaker distance:", cosine_dist(spk_sig, res['spk']))
res = json.loads(rec.FinalResult())
print ("Text:", res['text'])
if 'spk' in res:
print ("X-vector:", res['spk'])
print ("Speaker distance:", cosine_dist(spk_sig, res['spk']), "based on", res['spk_frames'], "frames")
print ("Speaker distance:", cosine_dist(spk_sig, res['spk']))
-56
View File
@@ -1,56 +0,0 @@
#!/usr/bin/env python3
from vosk import Model, KaldiRecognizer, SetLogLevel
import sys
import os
import wave
import subprocess
import srt
import json
import datetime
SetLogLevel(-1)
if not os.path.exists("model"):
print ("Please download the model from https://alphacephei.com/vosk/models and unpack as 'model' in the current folder.")
exit (1)
sample_rate=16000
model = Model("model")
rec = KaldiRecognizer(model, sample_rate)
rec.SetWords(True)
process = subprocess.Popen(['ffmpeg', '-loglevel', 'quiet', '-i',
sys.argv[1],
'-ar', str(sample_rate) , '-ac', '1', '-f', 's16le', '-'],
stdout=subprocess.PIPE)
WORDS_PER_LINE = 7
def transcribe():
results = []
subs = []
while True:
data = process.stdout.read(4000)
if len(data) == 0:
break
if rec.AcceptWaveform(data):
results.append(rec.Result())
results.append(rec.FinalResult())
for i, res in enumerate(results):
jres = json.loads(res)
if not 'result' in jres:
continue
words = jres['result']
for j in range(0, len(words), WORDS_PER_LINE):
line = words[j : j + WORDS_PER_LINE]
s = srt.Subtitle(index=len(subs),
content=" ".join([l['word'] for l in line]),
start=datetime.timedelta(seconds=line[0]['start']),
end=datetime.timedelta(seconds=line[-1]['end']))
subs.append(s)
return subs
print (srt.compose(transcribe()))
+1 -1
View File
@@ -6,7 +6,7 @@ import json
import os
if not os.path.exists("model"):
print ("Please download the model from https://alphacephei.com/vosk/models and unpack as 'model' in the current folder.")
print ("Please download the model from https://github.com/alphacep/vosk-api/blob/master/doc/models.md and unpack as 'model' in the current folder.")
exit (1)
-72
View File
@@ -1,72 +0,0 @@
#!/usr/bin/env python3
from vosk import Model, KaldiRecognizer, SetLogLevel
from webvtt import WebVTT, Caption
import sys
import os
import subprocess
import json
import textwrap
SetLogLevel(-1)
if not os.path.exists('model'):
print('Please download the model from https://alphacephei.com/vosk/models'
' and unpack as `model` in the current folder.')
exit(1)
sample_rate = 16000
model = Model('model')
rec = KaldiRecognizer(model, sample_rate)
rec.SetWords(True)
WORDS_PER_LINE = 7
def timeString(seconds):
minutes = seconds / 60
seconds = seconds % 60
hours = int(minutes / 60)
minutes = int(minutes % 60)
return '%i:%02i:%06.3f' % (hours, minutes, seconds)
def transcribe():
command = ['ffmpeg', '-nostdin', '-loglevel', 'quiet', '-i', sys.argv[1],
'-ar', str(sample_rate), '-ac', '1', '-f', 's16le', '-']
process = subprocess.Popen(command, stdout=subprocess.PIPE)
results = []
while True:
data = process.stdout.read(4000)
if len(data) == 0:
break
if rec.AcceptWaveform(data):
results.append(rec.Result())
results.append(rec.FinalResult())
vtt = WebVTT()
for i, res in enumerate(results):
words = json.loads(res).get('result')
if not words:
continue
start = timeString(words[0]['start'])
end = timeString(words[-1]['end'])
content = ' '.join([w['word'] for w in words])
caption = Caption(start, end, textwrap.fill(content))
vtt.captions.append(caption)
# save or return webvtt
if len(sys.argv) > 2:
vtt.save(sys.argv[2])
else:
print(vtt.content)
if __name__ == '__main__':
if not (1 < len(sys.argv) < 4):
print(f'Usage: {sys.argv[0]} audiofile [output file]')
exit(1)
transcribe()

Some files were not shown because too many files have changed in this diff Show More