feat(voice): add mobile linking and cancellation

Signed-off-by: John Tennant <johnmatthewtennant@gmail.com>
This commit is contained in:
John Matthew Tennant
2026-07-28 20:14:00 -04:00
committed by John Tennant
parent 388cf0ccdc
commit b1c899fa6d
27 changed files with 3073 additions and 16 deletions
Generated
-2
View File
@@ -8446,8 +8446,6 @@ dependencies = [
[[package]]
name = "sherpa-onnx-sys"
version = "1.13.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ffc951af03dc0653c0622158ca8a585a6f2bc43b7b06048cf0e5b5020005c227"
dependencies = [
"bzip2",
"tar",
+2 -1
View File
@@ -29,7 +29,7 @@ members = [
"crates/buzz-voice",
"examples/countdown-bot",
]
exclude = ["desktop/src-tauri"]
exclude = ["desktop/src-tauri", "crates/sherpa-onnx-sys"]
resolver = "2"
[workspace.package]
@@ -171,3 +171,4 @@ strip = true
# allowlist for the auth token). Revert to crates.io once #449 lands upstream.
[patch.crates-io]
aws-creds = { git = "https://github.com/tlongwell-block/rust-s3", rev = "c9fce3620dd434c1f810101d672cf384268dbb0f" }
sherpa-onnx-sys = { path = "crates/sherpa-onnx-sys" }
+4
View File
@@ -63,6 +63,10 @@ RUN apt-get update \
# locations. The normal runtime strips it below; runtime-debug retains it.
ENV CARGO_PROFILE_RELEASE_DEBUG=line-tables-only
COPY --from=planner /build/recipe.json recipe.json
# The sherpa sys patch keeps its published 1.13.4 version so Cargo can replace
# the registry crate. It is excluded from the workspace recipe because
# cargo-chef masks workspace package versions to 0.0.1.
COPY --from=planner /build/crates/sherpa-onnx-sys crates/sherpa-onnx-sys
# Cook the full workspace recipe — relay deps include workspace siblings, so
# scoping to -p buzz-relay misses transitive deps and re-builds them later.
RUN cargo chef cook --release --recipe-path recipe.json
+3
View File
@@ -15,6 +15,9 @@ RUN apt-get update \
&& apt-get install -y --no-install-recommends build-essential pkg-config libssl-dev ca-certificates \
&& rm -rf /var/lib/apt/lists/*
COPY --from=planner /build/recipe.json recipe.json
# Keep the patched sherpa sys crate at its published version; cargo-chef masks
# workspace package versions, so the patch remains outside the workspace recipe.
COPY --from=planner /build/crates/sherpa-onnx-sys crates/sherpa-onnx-sys
RUN cargo chef cook --release --recipe-path recipe.json
COPY . .
RUN cargo build --release --locked -p buzz-push-gateway --bin buzz-push-gateway \
+6 -1
View File
@@ -7,6 +7,11 @@ license.workspace = true
repository.workspace = true
description = "Reusable local voice primitives for Buzz"
[features]
default = ["static"]
static = ["sherpa-onnx/static"]
shared = ["sherpa-onnx/shared"]
[dependencies]
ort = { version = "=2.0.0-rc.12", default-features = false, features = ["api-24", "ndarray", "std"] }
ort-sys = { version = "=2.0.0-rc.12", features = ["disable-linking"] }
@@ -14,5 +19,5 @@ rand = "0.10"
sentencepiece-model = "0.1"
serde = { version = "1", features = ["derive"] }
serde_json = "1"
sherpa-onnx = "1.12"
sherpa-onnx = { version = "1.12", default-features = false }
tokenizers = { version = "0.22", default-features = false, features = ["fancy-regex"] }
+127 -3
View File
@@ -13,6 +13,7 @@
//!
//! `huddle::models` writes the complete attribution beside the cached bytes.
use std::cell::RefCell;
use std::path::{Path, PathBuf};
use std::sync::Mutex;
@@ -39,6 +40,42 @@ pub const VOICE_FILE_EXT: &str = "wav";
const TTS_NUM_THREADS: usize = 1;
thread_local! {
static ACTIVE_SYNTHESIS_ENGINES: RefCell<Vec<usize>> = const { RefCell::new(Vec::new()) };
}
struct SynthesisCallGuard {
engine_id: usize,
}
impl SynthesisCallGuard {
fn enter(engine_id: usize) -> Result<Self, String> {
ACTIVE_SYNTHESIS_ENGINES.with(|active| {
let mut active = active.borrow_mut();
if active.contains(&engine_id) {
return Err("Pocket TTS callback re-entered the active engine".to_string());
}
active.push(engine_id);
Ok(Self { engine_id })
})
}
fn is_active(engine_id: usize) -> bool {
ACTIVE_SYNTHESIS_ENGINES.with(|active| active.borrow().contains(&engine_id))
}
}
impl Drop for SynthesisCallGuard {
fn drop(&mut self) {
ACTIVE_SYNTHESIS_ENGINES.with(|active| {
let mut active = active.borrow_mut();
if let Some(index) = active.iter().rposition(|engine| *engine == self.engine_id) {
active.remove(index);
}
});
}
}
/// Loaded reference voice samples and their original sample rate.
#[derive(Debug, Clone)]
pub struct VoiceStyle {
@@ -90,6 +127,9 @@ impl PocketTts {
/// Split text into synthesis units that satisfy the bundle's exact
/// 50-token input limit.
pub fn split_text_into_chunks(&self, text: &str) -> Result<Vec<String>, String> {
if SynthesisCallGuard::is_active(self as *const Self as usize) {
return Err("Pocket TTS callback re-entered the active engine".to_string());
}
let Some(prepared) = prepare_april_prompt(text) else {
return Ok(Vec::new());
};
@@ -104,12 +144,36 @@ impl PocketTts {
/// Pocket detects language from text and this model uses one synthesis
/// step, so `_lang` and `_steps` intentionally do not affect output.
pub fn synth_chunk(
&self,
text: &str,
lang: &str,
style: &VoiceStyle,
steps: usize,
) -> Result<Vec<f32>, String> {
self.synth_chunk_with_callback(text, lang, style, steps, None::<fn(&[f32], f32) -> bool>)
}
/// Synthesize text, allowing the caller to stop generation early.
///
/// The callback receives PCM accumulated after each decoded text chunk
/// and progress in `[0, 1]`. During latent generation the callback is
/// invoked with an empty sample slice so cancellation can remain
/// responsive before PCM is available. Return `true` to continue or
/// `false` to stop and return the audio generated so far. Progress is
/// monotonic across split text chunks. Calls back into the same
/// [`PocketTts`] return an error instead of blocking on its engine lock.
pub fn synth_chunk_with_callback<F>(
&self,
text: &str,
_lang: &str,
style: &VoiceStyle,
_steps: usize,
) -> Result<Vec<f32>, String> {
mut callback: Option<F>,
) -> Result<Vec<f32>, String>
where
F: FnMut(&[f32], f32) -> bool + 'static,
{
let _call_guard = SynthesisCallGuard::enter(self as *const Self as usize)?;
let Some(prepared) = prepare_april_prompt(text) else {
return Ok(Vec::new());
};
@@ -119,15 +183,47 @@ impl PocketTts {
.map_err(|_| "Pocket TTS engine lock poisoned".to_string())?;
let chunks = engine.split_prompt(&prepared)?;
let mut samples = Vec::new();
for chunk in chunks {
let chunk_count = chunks.len();
for (index, chunk) in chunks.into_iter().enumerate() {
let prepared = prepare_april_prompt(&chunk)
.ok_or_else(|| "Pocket TTS prompt chunk became empty".to_string())?;
samples.extend(engine.synth_chunk(&prepared, style)?);
let progress_offset = index as f32 / chunk_count as f32;
let progress_scale = 1.0 / chunk_count as f32;
let (chunk_samples, cancelled) = engine.synth_chunk_with_callback(
&prepared,
style,
&mut callback,
progress_offset,
progress_scale,
)?;
samples.extend(chunk_samples);
if cancelled {
break;
}
let progress = (index + 1) as f32 / chunk_count as f32;
if !callback_allows_progress(&mut callback, &samples, progress)? {
break;
}
}
Ok(samples)
}
}
fn callback_allows_progress<F>(
callback: &mut Option<F>,
samples: &[f32],
progress: f32,
) -> Result<bool, String>
where
F: FnMut(&[f32], f32) -> bool,
{
let Some(callback) = callback.as_mut() else {
return Ok(true);
};
std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| callback(samples, progress)))
.map_err(|_| "Pocket TTS synthesis callback panicked".to_string())
}
#[cfg(test)]
mod tests {
use super::*;
@@ -147,6 +243,34 @@ mod tests {
.any(|artifact| artifact.filename == "flow_lm_main.onnx"));
}
#[test]
fn callback_can_cancel_before_pcm_is_available() {
let mut callback = Some(|samples: &[f32], progress: f32| {
assert!(samples.is_empty());
assert_eq!(progress, 0.25);
false
});
assert!(!callback_allows_progress(&mut callback, &[], 0.25).expect("callback"));
}
#[test]
fn callback_panic_is_reported_without_unwinding() {
let mut callback = Some(|_: &[f32], _: f32| -> bool {
panic!("callback failure");
});
assert_eq!(
callback_allows_progress(&mut callback, &[], 0.0).unwrap_err(),
"Pocket TTS synthesis callback panicked"
);
}
#[test]
fn active_engine_reentry_is_rejected() {
let _guard = SynthesisCallGuard::enter(42).expect("first call");
assert!(SynthesisCallGuard::enter(42).is_err());
assert!(SynthesisCallGuard::is_active(42));
}
#[test]
#[ignore = "requires BUZZ_POCKET_TEST_MODEL_DIR"]
fn production_api_emits_non_silent_april_int8_pcm() {
+32 -9
View File
@@ -304,11 +304,17 @@ impl AprilPocketTts {
.collect()
}
pub(crate) fn synth_chunk(
pub(crate) fn synth_chunk_with_callback<F>(
&mut self,
prepared: &AprilPreparedPrompt,
style: &VoiceStyle,
) -> Result<Vec<f32>, String> {
callback: &mut Option<F>,
progress_offset: f32,
progress_scale: f32,
) -> Result<(Vec<f32>, bool), String>
where
F: FnMut(&[f32], f32) -> bool,
{
let voice_embeddings = self.voice_embeddings(style)?;
let mut flow_state = self.condition_voice(&voice_embeddings)?;
let token_ids = self
@@ -321,7 +327,7 @@ impl AprilPocketTts {
.map(i64::from)
.collect::<Vec<_>>();
if token_ids.is_empty() {
return Ok(Vec::new());
return Ok((Vec::new(), false));
}
if token_ids.len() > self.bundle.max_token_per_chunk {
return Err(format!(
@@ -335,9 +341,15 @@ impl AprilPocketTts {
let text_embeddings = self.text_embeddings(token_ids)?;
self.run_flow_main_prefix(&text_embeddings, &mut flow_state)?;
let max_frames = estimate_max_frames(token_count, self.bundle.frame_rate);
let latents =
self.generate_latents(max_frames, prepared.frames_after_eos, &mut flow_state)?;
self.decode_latents(&latents)
let (latents, cancelled) = self.generate_latents(
max_frames,
prepared.frames_after_eos,
&mut flow_state,
callback,
progress_offset,
progress_scale,
)?;
Ok((self.decode_latents(&latents)?, cancelled))
}
fn prepared_token_count(&self, text: &str) -> Result<usize, String> {
@@ -497,18 +509,29 @@ impl AprilPocketTts {
replace_state_from_outputs(state, &mut outputs)
}
fn generate_latents(
fn generate_latents<F>(
&mut self,
max_frames: usize,
frames_after_eos: usize,
state: &mut [StateValue],
) -> Result<Vec<f32>, String> {
callback: &mut Option<F>,
progress_offset: f32,
progress_scale: f32,
) -> Result<(Vec<f32>, bool), String>
where
F: FnMut(&[f32], f32) -> bool,
{
let mut current = vec![f32::NAN; self.bundle.latent_dim];
let mut latents = Vec::with_capacity(max_frames * self.bundle.latent_dim);
let mut eos_step = None;
let mut rng = rand::rng();
for step in 0..max_frames {
let local_progress = step as f32 / max_frames as f32;
let progress = progress_offset + local_progress * progress_scale;
if !super::callback_allows_progress(callback, &[], progress)? {
return Ok((latents, true));
}
let sequence = Tensor::from_array((
vec![1_i64, 1, self.bundle.latent_dim as i64],
current.clone().into_boxed_slice(),
@@ -594,7 +617,7 @@ impl AprilPocketTts {
current.clone_from(&noise);
latents.extend_from_slice(&noise);
}
Ok(latents)
Ok((latents, false))
}
fn decode_latents(&mut self, latents: &[f32]) -> Result<Vec<f32>, String> {
+63
View File
@@ -0,0 +1,63 @@
# THIS FILE IS AUTOMATICALLY GENERATED BY CARGO
#
# When uploading crates to the registry Cargo will automatically
# "normalize" Cargo.toml files for maximal compatibility
# with all versions of Cargo and also rewrite `path` dependencies
# to registry (e.g., crates.io) dependencies.
#
# If you are reading this file be aware that the original Cargo.toml
# will likely look very different (and much more reasonable).
# See Cargo.toml.orig for the original contents.
[package]
edition = "2021"
name = "sherpa-onnx-sys"
version = "1.13.4"
build = "build.rs"
links = "sherpa-onnx"
include = [
"src/**",
"build.rs",
"Cargo.toml",
"README.md",
"LICENSE*",
]
autolib = false
autobins = false
autoexamples = false
autotests = false
autobenches = false
description = "Raw FFI bindings to the sherpa-onnx C API"
readme = "README.md"
keywords = [
"ffi",
"speech",
"sherpa-onnx",
"bindings",
]
categories = ["external-ffi-bindings"]
license = "Apache-2.0"
repository = "https://github.com/k2-fsa/sherpa-onnx"
[package.metadata.docs.rs]
default-features = false
features = []
[features]
default = ["static"]
shared = []
static = []
[lib]
name = "sherpa_onnx_sys"
path = "src/lib.rs"
[build-dependencies.bzip2]
version = "0.4"
[build-dependencies.tar]
version = "0.4"
[build-dependencies.ureq]
version = "2.12"
features = ["proxy-from-env"]
+34
View File
@@ -0,0 +1,34 @@
[package]
name = "sherpa-onnx-sys"
version = "1.13.4"
edition = "2021"
description = "Raw FFI bindings to the sherpa-onnx C API"
license = "Apache-2.0"
repository = "https://github.com/k2-fsa/sherpa-onnx"
readme = "README.md"
links = "sherpa-onnx"
keywords = ["ffi", "speech", "sherpa-onnx", "bindings"]
categories = ["external-ffi-bindings"]
include = [
"src/**",
"build.rs",
"Cargo.toml",
"README.md",
"LICENSE*",
]
[features]
default = ["static"]
static = []
shared = []
[build-dependencies]
bzip2 = "0.4"
tar = "0.4"
ureq = { version = "2.12", features = ["proxy-from-env"] }
[package.metadata.docs.rs]
default-features = false
features = []
+202
View File
@@ -0,0 +1,202 @@
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
1. Definitions.
"License" shall mean the terms and conditions for use, reproduction,
and distribution as defined by Sections 1 through 9 of this document.
"Licensor" shall mean the copyright owner or entity authorized by
the copyright owner that is granting the License.
"Legal Entity" shall mean the union of the acting entity and all
other entities that control, are controlled by, or are under common
control with that entity. For the purposes of this definition,
"control" means (i) the power, direct or indirect, to cause the
direction or management of such entity, whether by contract or
otherwise, or (ii) ownership of fifty percent (50%) or more of the
outstanding shares, or (iii) beneficial ownership of such entity.
"You" (or "Your") shall mean an individual or Legal Entity
exercising permissions granted by this License.
"Source" form shall mean the preferred form for making modifications,
including but not limited to software source code, documentation
source, and configuration files.
"Object" form shall mean any form resulting from mechanical
transformation or translation of a Source form, including but
not limited to compiled object code, generated documentation,
and conversions to other media types.
"Work" shall mean the work of authorship, whether in Source or
Object form, made available under the License, as indicated by a
copyright notice that is included in or attached to the work
(an example is provided in the Appendix below).
"Derivative Works" shall mean any work, whether in Source or Object
form, that is based on (or derived from) the Work and for which the
editorial revisions, annotations, elaborations, or other modifications
represent, as a whole, an original work of authorship. For the purposes
of this License, Derivative Works shall not include works that remain
separable from, or merely link (or bind by name) to the interfaces of,
the Work and Derivative Works thereof.
"Contribution" shall mean any work of authorship, including
the original version of the Work and any modifications or additions
to that Work or Derivative Works thereof, that is intentionally
submitted to Licensor for inclusion in the Work by the copyright owner
or by an individual or Legal Entity authorized to submit on behalf of
the copyright owner. For the purposes of this definition, "submitted"
means any form of electronic, verbal, or written communication sent
to the Licensor or its representatives, including but not limited to
communication on electronic mailing lists, source code control systems,
and issue tracking systems that are managed by, or on behalf of, the
Licensor for the purpose of discussing and improving the Work, but
excluding communication that is conspicuously marked or otherwise
designated in writing by the copyright owner as "Not a Contribution."
"Contributor" shall mean Licensor and any individual or Legal Entity
on behalf of whom a Contribution has been received by Licensor and
subsequently incorporated within the Work.
2. Grant of Copyright License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
copyright license to reproduce, prepare Derivative Works of,
publicly display, publicly perform, sublicense, and distribute the
Work and such Derivative Works in Source or Object form.
3. Grant of Patent License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
(except as stated in this section) patent license to make, have made,
use, offer to sell, sell, import, and otherwise transfer the Work,
where such license applies only to those patent claims licensable
by such Contributor that are necessarily infringed by their
Contribution(s) alone or by combination of their Contribution(s)
with the Work to which such Contribution(s) was submitted. If You
institute patent litigation against any entity (including a
cross-claim or counterclaim in a lawsuit) alleging that the Work
or a Contribution incorporated within the Work constitutes direct
or contributory patent infringement, then any patent licenses
granted to You under this License for that Work shall terminate
as of the date such litigation is filed.
4. Redistribution. You may reproduce and distribute copies of the
Work or Derivative Works thereof in any medium, with or without
modifications, and in Source or Object form, provided that You
meet the following conditions:
(a) You must give any other recipients of the Work or
Derivative Works a copy of this License; and
(b) You must cause any modified files to carry prominent notices
stating that You changed the files; and
(c) You must retain, in the Source form of any Derivative Works
that You distribute, all copyright, patent, trademark, and
attribution notices from the Source form of the Work,
excluding those notices that do not pertain to any part of
the Derivative Works; and
(d) If the Work includes a "NOTICE" text file as part of its
distribution, then any Derivative Works that You distribute must
include a readable copy of the attribution notices contained
within such NOTICE file, excluding those notices that do not
pertain to any part of the Derivative Works, in at least one
of the following places: within a NOTICE text file distributed
as part of the Derivative Works; within the Source form or
documentation, if provided along with the Derivative Works; or,
within a display generated by the Derivative Works, if and
wherever such third-party notices normally appear. The contents
of the NOTICE file are for informational purposes only and
do not modify the License. You may add Your own attribution
notices within Derivative Works that You distribute, alongside
or as an addendum to the NOTICE text from the Work, provided
that such additional attribution notices cannot be construed
as modifying the License.
You may add Your own copyright statement to Your modifications and
may provide additional or different license terms and conditions
for use, reproduction, or distribution of Your modifications, or
for any such Derivative Works as a whole, provided Your use,
reproduction, and distribution of the Work otherwise complies with
the conditions stated in this License.
5. Submission of Contributions. Unless You explicitly state otherwise,
any Contribution intentionally submitted for inclusion in the Work
by You to the Licensor shall be under the terms and conditions of
this License, without any additional terms or conditions.
Notwithstanding the above, nothing herein shall supersede or modify
the terms of any separate license agreement you may have executed
with Licensor regarding such Contributions.
6. Trademarks. This License does not grant permission to use the trade
names, trademarks, service marks, or product names of the Licensor,
except as required for reasonable and customary use in describing the
origin of the Work and reproducing the content of the NOTICE file.
7. Disclaimer of Warranty. Unless required by applicable law or
agreed to in writing, Licensor provides the Work (and each
Contributor provides its Contributions) on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
implied, including, without limitation, any warranties or conditions
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
PARTICULAR PURPOSE. You are solely responsible for determining the
appropriateness of using or redistributing the Work and assume any
risks associated with Your exercise of permissions under this License.
8. Limitation of Liability. In no event and under no legal theory,
whether in tort (including negligence), contract, or otherwise,
unless required by applicable law (such as deliberate and grossly
negligent acts) or agreed to in writing, shall any Contributor be
liable to You for damages, including any direct, indirect, special,
incidental, or consequential damages of any character arising as a
result of this License or out of the use or inability to use the
Work (including but not limited to damages for loss of goodwill,
work stoppage, computer failure or malfunction, or any and all
other commercial damages or losses), even if such Contributor
has been advised of the possibility of such damages.
9. Accepting Warranty or Additional Liability. While redistributing
the Work or Derivative Works thereof, You may choose to offer,
and charge a fee for, acceptance of support, warranty, indemnity,
or other liability obligations and/or rights consistent with this
License. However, in accepting such obligations, You may act only
on Your own behalf and on Your sole responsibility, not on behalf
of any other Contributor, and only if You agree to indemnify,
defend, and hold each Contributor harmless for any liability
incurred by, or claims asserted against, such Contributor by reason
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
APPENDIX: How to apply the Apache License to your work.
To apply the Apache License to your work, attach the following
boilerplate notice, with the fields enclosed by brackets "[]"
replaced with your own identifying information. (Don't include
the brackets!) The text should be enclosed in the appropriate
comment syntax for the file format. We also recommend that a
file or class name and description of purpose be included on the
same "printed page" as the copyright notice for easier
identification within third-party archives.
Copyright [yyyy] [name of copyright owner]
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
+671
View File
@@ -0,0 +1,671 @@
<div align="center">
[![Ask DeepWiki](https://deepwiki.com/badge.svg)](https://deepwiki.com/k2-fsa/sherpa-onnx)
</div>
Buzz carries the crates.io 1.13.4 FFI sources and locally patches the build
script to link caller-supplied official iOS and Android libraries through
`SHERPA_ONNX_LIB_DIR`. Desktop archive selection remains unchanged.
### Supported functions
|Speech recognition| [Speech synthesis][tts-url] | [Source separation][ss-url] |
|------------------|------------------|-------------------|
| ✔️ | ✔️ | ✔️ |
|Speaker identification| [Speaker diarization][sd-url] | Speaker verification |
|----------------------|-------------------- |------------------------|
| ✔️ | ✔️ | ✔️ |
| [Spoken Language identification][slid-url] | [Audio tagging][at-url] | [Voice activity detection][vad-url] |
|--------------------------------|---------------|--------------------------|
| ✔️ | ✔️ | ✔️ |
| [Keyword spotting][kws-url] | [Add punctuation][punct-url] | [Speech enhancement][se-url] |
|------------------|-----------------|--------------------|
| ✔️ | ✔️ | ✔️ |
### Supported platforms
|Architecture| Android | iOS | Windows | macOS | linux | HarmonyOS |
|------------|---------|---------|------------|-------|-------|-----------|
| x64 | ✔️ | | ✔️ | ✔️ | ✔️ | ✔️ |
| x86 | ✔️ | | ✔️ | | | |
| arm64 | ✔️ | ✔️ | ✔️ | ✔️ | ✔️ | ✔️ |
| arm32 | ✔️ | | | | ✔️ | ✔️ |
| riscv64 | | | | | ✔️ | |
### Supported programming languages
| 1. C++ | 2. C | 3. Python | 4. JavaScript |
|--------|-------|-----------|---------------|
| ✔️ | ✔️ | ✔️ | ✔️ |
|5. Java | 6. C# | 7. Kotlin | 8. Swift |
|--------|-------|-----------|----------|
| ✔️ | ✔️ | ✔️ | ✔️ |
| 9. Go | 10. Dart | 11. Rust | 12. Pascal |
|-------|----------|----------|------------|
| ✔️ | ✔️ | ✔️ | ✔️ |
It also supports WebAssembly.
### Supported NPUs
| [1. Rockchip NPU (RKNN)][rknpu-doc] | [2. Qualcomm NPU (QNN)][qnn-doc] | [3. Ascend NPU][ascend-doc] |
|-------------------------------------|-----------------------------------|-----------------------------|
| ✔️ | ✔️ | ✔️ |
| [4. Axera NPU][axera-npu] |
|---------------------------|
| ✔️ |
[Join our discord](https://discord.gg/fJdxzg2VbG)
## Introduction
This repository supports running the following functions **locally**
- Speech-to-text (i.e., ASR); both streaming and non-streaming are supported
- Text-to-speech (i.e., TTS)
- Speaker diarization
- Speaker identification
- Speaker verification
- Spoken language identification
- Audio tagging
- VAD (e.g., [silero-vad][silero-vad])
- Speech enhancement (e.g., [gtcrn][gtcrn], [DPDFNet](https://github.com/ceva-ip/DPDFNet))
- Keyword spotting
- Source separation (e.g., [spleeter][spleeter], [UVR][UVR])
on the following platforms and operating systems:
- x86, ``x86_64``, 32-bit ARM, 64-bit ARM (arm64, aarch64), RISC-V (riscv64), **RK NPU**, **Ascend NPU**
- Linux, macOS, Windows, openKylin
- Android, WearOS
- iOS
- HarmonyOS
- NodeJS
- WebAssembly
- [NVIDIA Jetson Orin NX][NVIDIA Jetson Orin NX] (Support running on both CPU and GPU)
- [NVIDIA Jetson Nano B01][NVIDIA Jetson Nano B01] (Support running on both CPU and GPU)
- [Raspberry Pi][Raspberry Pi]
- [RV1126][RV1126]
- [LicheePi4A][LicheePi4A]
- [VisionFive 2][VisionFive 2]
- [旭日X3派][旭日X3派]
- [爱芯派][爱芯派]
- [RK3588][RK3588]
- [SpacemiT-K1][SpacemiT-K1]
- [SpacemiT-K3][SpacemiT-K3]
- etc
with the following APIs
- C++, C, Python, Go, ``C#``
- Java, Kotlin, JavaScript
- Swift, Rust
- Dart, Object Pascal
### Links for Huggingface Spaces
<details>
<summary>You can visit the following Huggingface spaces to try sherpa-onnx without
installing anything. All you need is a browser.</summary>
| Description | URL | 中国镜像 |
|-------------------------------------------------------|-----------------------------------------|----------------------------------------|
| Speaker diarization | [Click me][hf-space-speaker-diarization]| [镜像][hf-space-speaker-diarization-cn]|
| Speech recognition | [Click me][hf-space-asr] | [镜像][hf-space-asr-cn] |
| Speech recognition with [Whisper][Whisper] | [Click me][hf-space-asr-whisper] | [镜像][hf-space-asr-whisper-cn] |
| Speech synthesis | [Click me][hf-space-tts] | [镜像][hf-space-tts-cn] |
| Generate subtitles | [Click me][hf-space-subtitle] | [镜像][hf-space-subtitle-cn] |
| Audio tagging | [Click me][hf-space-audio-tagging] | [镜像][hf-space-audio-tagging-cn] |
| Source separation | [Click me][hf-space-source-separation] | [镜像][hf-space-source-separation-cn] |
| Spoken language identification with [Whisper][Whisper]| [Click me][hf-space-slid-whisper] | [镜像][hf-space-slid-whisper-cn] |
We also have spaces built using WebAssembly. They are listed below:
| Description | Huggingface space| ModelScope space|
|------------------------------------------------------------------------------------------|------------------|-----------------|
|Voice activity detection with [silero-vad][silero-vad] | [Click me][wasm-hf-vad]|[地址][wasm-ms-vad]|
|Real-time speech recognition (Chinese + English) with Zipformer | [Click me][wasm-hf-streaming-asr-zh-en-zipformer]|[地址][wasm-hf-streaming-asr-zh-en-zipformer]|
|Real-time speech recognition (Chinese + English) with Paraformer |[Click me][wasm-hf-streaming-asr-zh-en-paraformer]| [地址][wasm-ms-streaming-asr-zh-en-paraformer]|
|Real-time speech recognition (Chinese + English + Cantonese) with [Paraformer-large][Paraformer-large]|[Click me][wasm-hf-streaming-asr-zh-en-yue-paraformer]| [地址][wasm-ms-streaming-asr-zh-en-yue-paraformer]|
|Real-time speech recognition (English) |[Click me][wasm-hf-streaming-asr-en-zipformer] |[地址][wasm-ms-streaming-asr-en-zipformer]|
|VAD + speech recognition (Chinese) with [Zipformer CTC](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-ctc/icefall/zipformer.html#sherpa-onnx-zipformer-ctc-zh-int8-2025-07-03-chinese)|[Click me][wasm-hf-vad-asr-zh-zipformer-ctc-07-03]| [地址][wasm-ms-vad-asr-zh-zipformer-ctc-07-03]|
|VAD + speech recognition (Chinese + English + Korean + Japanese + Cantonese) with [SenseVoice][SenseVoice]|[Click me][wasm-hf-vad-asr-zh-en-ko-ja-yue-sense-voice]| [地址][wasm-ms-vad-asr-zh-en-ko-ja-yue-sense-voice]|
|VAD + speech recognition (English) with [Whisper][Whisper] tiny.en|[Click me][wasm-hf-vad-asr-en-whisper-tiny-en]| [地址][wasm-ms-vad-asr-en-whisper-tiny-en]|
|VAD + speech recognition (English) with [Moonshine tiny][Moonshine tiny]|[Click me][wasm-hf-vad-asr-en-moonshine-tiny-en]| [地址][wasm-ms-vad-asr-en-moonshine-tiny-en]|
|VAD + speech recognition (English) with Zipformer trained with [GigaSpeech][GigaSpeech] |[Click me][wasm-hf-vad-asr-en-zipformer-gigaspeech]| [地址][wasm-ms-vad-asr-en-zipformer-gigaspeech]|
|VAD + speech recognition (Chinese) with Zipformer trained with [WenetSpeech][WenetSpeech] |[Click me][wasm-hf-vad-asr-zh-zipformer-wenetspeech]| [地址][wasm-ms-vad-asr-zh-zipformer-wenetspeech]|
|VAD + speech recognition (Japanese) with Zipformer trained with [ReazonSpeech][ReazonSpeech]|[Click me][wasm-hf-vad-asr-ja-zipformer-reazonspeech]| [地址][wasm-ms-vad-asr-ja-zipformer-reazonspeech]|
|VAD + speech recognition (Thai) with Zipformer trained with [GigaSpeech2][GigaSpeech2] |[Click me][wasm-hf-vad-asr-th-zipformer-gigaspeech2]| [地址][wasm-ms-vad-asr-th-zipformer-gigaspeech2]|
|VAD + speech recognition (Chinese 多种方言) with a [TeleSpeech-ASR][TeleSpeech-ASR] CTC model|[Click me][wasm-hf-vad-asr-zh-telespeech]| [地址][wasm-ms-vad-asr-zh-telespeech]|
|VAD + speech recognition (English + Chinese, 及多种中文方言) with Paraformer-large |[Click me][wasm-hf-vad-asr-zh-en-paraformer-large]| [地址][wasm-ms-vad-asr-zh-en-paraformer-large]|
|VAD + speech recognition (English + Chinese, 及多种中文方言) with Paraformer-small |[Click me][wasm-hf-vad-asr-zh-en-paraformer-small]| [地址][wasm-ms-vad-asr-zh-en-paraformer-small]|
|VAD + speech recognition (多语种及多种中文方言) with [Dolphin][Dolphin]-base |[Click me][wasm-hf-vad-asr-multi-lang-dolphin-base]| [地址][wasm-ms-vad-asr-multi-lang-dolphin-base]|
|Speech synthesis (Piper, English) |[Click me][wasm-hf-tts-piper-en]| [地址][wasm-ms-tts-piper-en]|
|Speech synthesis (Piper, German) |[Click me][wasm-hf-tts-piper-de]| [地址][wasm-ms-tts-piper-de]|
|Speech synthesis (Matcha, Chinese) |[Click me][wasm-hf-tts-matcha-zh]| [地址][wasm-ms-tts-matcha-zh]|
|Speech synthesis (Matcha, English) |[Click me][wasm-hf-tts-matcha-en]| [地址][wasm-ms-tts-matcha-en]|
|Speech synthesis (Matcha, Chinese+English) |[Click me][wasm-hf-tts-matcha-zh-en]| [地址][wasm-ms-tts-matcha-zh-en]|
|Speaker diarization |[Click me][wasm-hf-speaker-diarization]|[地址][wasm-ms-speaker-diarization]|
|Voice cloning with ZipVoice (Chinese+English) |[Click me][wasm-hf-voice-cloning-zipvoice]|[地址][wasm-ms-voice-cloning-zipvoice]|
|Voice cloning with Pocket TTS (English) |[Click me][wasm-hf-voice-cloning-pocket]|[地址][wasm-ms-voice-cloning-pocket]|
</details>
### Links for pre-built Android APKs
<details>
<summary>You can find pre-built Android APKs for this repository in the following table</summary>
| Description | URL | 中国用户 |
|----------------------------------------|------------------------------------|-----------------------------------|
| Speaker diarization | [Address][apk-speaker-diarization] | [点此][apk-speaker-diarization-cn]|
| Streaming speech recognition | [Address][apk-streaming-asr] | [点此][apk-streaming-asr-cn] |
| Simulated-streaming speech recognition | [Address][apk-simula-streaming-asr]| [点此][apk-simula-streaming-asr-cn]|
| Text-to-speech | [Address][apk-tts] | [点此][apk-tts-cn] |
| Voice activity detection (VAD) | [Address][apk-vad] | [点此][apk-vad-cn] |
| VAD + non-streaming speech recognition | [Address][apk-vad-asr] | [点此][apk-vad-asr-cn] |
| Two-pass speech recognition | [Address][apk-2pass] | [点此][apk-2pass-cn] |
| Audio tagging | [Address][apk-at] | [点此][apk-at-cn] |
| Audio tagging (WearOS) | [Address][apk-at-wearos] | [点此][apk-at-wearos-cn] |
| Speaker identification | [Address][apk-sid] | [点此][apk-sid-cn] |
| Spoken language identification | [Address][apk-slid] | [点此][apk-slid-cn] |
| Keyword spotting | [Address][apk-kws] | [点此][apk-kws-cn] |
</details>
### Links for pre-built Flutter APPs
<details>
#### Real-time speech recognition
| Description | URL | 中国用户 |
|--------------------------------|-------------------------------------|-------------------------------------|
| Streaming speech recognition | [Address][apk-flutter-streaming-asr]| [点此][apk-flutter-streaming-asr-cn]|
#### Text-to-speech
| Description | URL | 中国用户 |
|------------------------------------------|------------------------------------|------------------------------------|
| Android (arm64-v8a, armeabi-v7a, x86_64) | [Address][flutter-tts-android] | [点此][flutter-tts-android-cn] |
| Linux (x64) | [Address][flutter-tts-linux] | [点此][flutter-tts-linux-cn] |
| macOS (x64) | [Address][flutter-tts-macos-x64] | [点此][flutter-tts-macos-x64-cn] |
| macOS (arm64) | [Address][flutter-tts-macos-arm64] | [点此][flutter-tts-macos-arm64-cn] |
| Windows (x64) | [Address][flutter-tts-win-x64] | [点此][flutter-tts-win-x64-cn] |
> Note: You need to build from source for iOS.
</details>
### Links for pre-built Lazarus APPs
<details>
#### Generating subtitles
| Description | URL | 中国用户 |
|--------------------------------|----------------------------|----------------------------|
| Generate subtitles (生成字幕) | [Address][lazarus-subtitle]| [点此][lazarus-subtitle-cn]|
</details>
### Links for pre-trained models
<details>
| Description | URL |
|---------------------------------------------|---------------------------------------------------------------------------------------|
| Speech recognition (speech to text, ASR) | [Address][asr-models] |
| Text-to-speech (TTS) | [Address][tts-models] |
| VAD | [Address][vad-models] |
| Keyword spotting | [Address][kws-models] |
| Audio tagging | [Address][at-models] |
| Speaker identification (Speaker ID) | [Address][sid-models] |
| Spoken language identification (Language ID)| See multi-lingual [Whisper][Whisper] ASR models from [Speech recognition][asr-models]|
| Punctuation | [Address][punct-models] |
| Speaker segmentation | [Address][speaker-segmentation-models] |
| Speech enhancement | [Address][speech-enhancement-models] |
| Source separation | [Address][source-separation-models] |
</details>
#### Some pre-trained ASR models (Streaming)
<details>
Please see
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/index.html>
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-paraformer/index.html>
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-ctc/index.html>
for more models. The following table lists only **SOME** of them.
|Name | Supported Languages| Description|
|-----|-----|----|
|[sherpa-onnx-streaming-zipformer-bilingual-zh-en-2023-02-20][sherpa-onnx-streaming-zipformer-bilingual-zh-en-2023-02-20]| Chinese, English| See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/zipformer-transducer-models.html#csukuangfj-sherpa-onnx-streaming-zipformer-bilingual-zh-en-2023-02-20-bilingual-chinese-english)|
|[sherpa-onnx-streaming-zipformer-small-bilingual-zh-en-2023-02-16][sherpa-onnx-streaming-zipformer-small-bilingual-zh-en-2023-02-16]| Chinese, English| See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/zipformer-transducer-models.html#sherpa-onnx-streaming-zipformer-small-bilingual-zh-en-2023-02-16-bilingual-chinese-english)|
|[sherpa-onnx-streaming-zipformer-zh-14M-2023-02-23][sherpa-onnx-streaming-zipformer-zh-14M-2023-02-23]|Chinese| Suitable for Cortex A7 CPU. See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/zipformer-transducer-models.html#sherpa-onnx-streaming-zipformer-zh-14m-2023-02-23)|
|[sherpa-onnx-streaming-zipformer-en-20M-2023-02-17][sherpa-onnx-streaming-zipformer-en-20M-2023-02-17]|English|Suitable for Cortex A7 CPU. See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/zipformer-transducer-models.html#sherpa-onnx-streaming-zipformer-en-20m-2023-02-17)|
|[sherpa-onnx-streaming-zipformer-korean-2024-06-16][sherpa-onnx-streaming-zipformer-korean-2024-06-16]|Korean| See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/zipformer-transducer-models.html#sherpa-onnx-streaming-zipformer-korean-2024-06-16-korean)|
|[sherpa-onnx-streaming-zipformer-fr-2023-04-14][sherpa-onnx-streaming-zipformer-fr-2023-04-14]|French| See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/zipformer-transducer-models.html#shaojieli-sherpa-onnx-streaming-zipformer-fr-2023-04-14-french)|
</details>
#### Some pre-trained ASR models (Non-Streaming)
<details>
Please see
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/index.html>
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-paraformer/index.html>
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-ctc/index.html>
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/telespeech/index.html>
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/whisper/index.html>
for more models. The following table lists only **SOME** of them.
|Name | Supported Languages| Description|
|-----|-----|----|
|[sherpa-onnx-nemo-parakeet-tdt-0.6b-v2-int8](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/nemo-transducer-models.html#sherpa-onnx-nemo-parakeet-tdt-0-6b-v2-int8-english)| English | It is converted from <https://huggingface.co/nvidia/parakeet-tdt-0.6b-v2>|
|[Whisper tiny.en](https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-whisper-tiny.en.tar.bz2)|English| See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/whisper/tiny.en.html)|
|[Moonshine tiny][Moonshine tiny]|English|See [also](https://github.com/usefulsensors/moonshine)|
|[sherpa-onnx-zipformer-ctc-zh-int8-2025-07-03](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-ctc/icefall/zipformer.html#sherpa-onnx-zipformer-ctc-zh-int8-2025-07-03-chinese)|Chinese| A Zipformer CTC model|
|[sherpa-onnx-sense-voice-zh-en-ja-ko-yue-2024-07-17][sherpa-onnx-sense-voice-zh-en-ja-ko-yue-2024-07-17]|Chinese, Cantonese, English, Korean, Japanese| 支持多种中文方言. See [also](https://k2-fsa.github.io/sherpa/onnx/sense-voice/index.html)|
|[sherpa-onnx-paraformer-zh-2024-03-09][sherpa-onnx-paraformer-zh-2024-03-09]|Chinese, English| 也支持多种中文方言. See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-paraformer/paraformer-models.html#csukuangfj-sherpa-onnx-paraformer-zh-2024-03-09-chinese-english)|
|[sherpa-onnx-zipformer-ja-reazonspeech-2024-08-01][sherpa-onnx-zipformer-ja-reazonspeech-2024-08-01]|Japanese|See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/zipformer-transducer-models.html#sherpa-onnx-zipformer-ja-reazonspeech-2024-08-01-japanese)|
|[sherpa-onnx-nemo-transducer-giga-am-russian-2024-10-24][sherpa-onnx-nemo-transducer-giga-am-russian-2024-10-24]|Russian|See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/nemo-transducer-models.html#sherpa-onnx-nemo-transducer-giga-am-russian-2024-10-24-russian)|
|[sherpa-onnx-nemo-ctc-giga-am-russian-2024-10-24][sherpa-onnx-nemo-ctc-giga-am-russian-2024-10-24]|Russian| See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-ctc/nemo/russian.html#sherpa-onnx-nemo-ctc-giga-am-russian-2024-10-24)|
|[sherpa-onnx-zipformer-ru-2024-09-18][sherpa-onnx-zipformer-ru-2024-09-18]|Russian|See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/zipformer-transducer-models.html#sherpa-onnx-zipformer-ru-2024-09-18-russian)|
|[sherpa-onnx-zipformer-korean-2024-06-24][sherpa-onnx-zipformer-korean-2024-06-24]|Korean|See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/zipformer-transducer-models.html#sherpa-onnx-zipformer-korean-2024-06-24-korean)|
|[sherpa-onnx-zipformer-thai-2024-06-20][sherpa-onnx-zipformer-thai-2024-06-20]|Thai| See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/zipformer-transducer-models.html#sherpa-onnx-zipformer-thai-2024-06-20-thai)|
|[sherpa-onnx-telespeech-ctc-int8-zh-2024-06-04][sherpa-onnx-telespeech-ctc-int8-zh-2024-06-04]|Chinese| 支持多种方言. See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/telespeech/models.html#sherpa-onnx-telespeech-ctc-int8-zh-2024-06-04)|
</details>
### Useful links
- Documentation: https://k2-fsa.github.io/sherpa/onnx/
- Bilibili 演示视频: https://search.bilibili.com/all?keyword=%E6%96%B0%E4%B8%80%E4%BB%A3Kaldi
### How to reach us
Please see
https://k2-fsa.github.io/sherpa/social-groups.html
for 新一代 Kaldi **微信交流群** and **QQ 交流群**.
## Projects using sherpa-onnx
### [Sherpa Voice / @siteed/sherpa-onnx.rn](https://github.com/deeeed/audiolab)
> React Native wrapper and demo app for validating sherpa-onnx on iOS,
> Android, and Web, including ASR, TTS, VAD, KWS, speaker ID, diarization,
> language ID, punctuation, audio tagging, and speech enhancement.
- [NPM package](https://www.npmjs.com/package/@siteed/sherpa-onnx.rn)
- [Live demo](https://deeeed.github.io/audiolab/sherpa-voice/)
### [Speed of Sound](https://github.com/zugaldia/speedofsound)
> A voice-typing application for the Linux desktop (GTK4/Adwaita).
> It captures microphone audio, transcribes it offline using Sherpa ONNX ASR models,
> optionally polishes the text with an LLM, and types the result into the active window
> via XDG Remote Desktop Portal keyboard simulation.
### [VoxSherpa TTS](https://github.com/CodeBySonu95/VoxSherpa-TTS)
> VoxSherpa TTS is a 100% offline Android Text-to-Speech app powered by Sherpa-ONNX.
> It supports Kokoro-82M, Piper, and VITS engines with multilingual support including
> Hindi, English, British English, Japanese, Chinese and 50+ more languages.
- [Download APK (All Versions)](https://github.com/CodeBySonu95/VoxSherpa-TTS/releases)
- Android 11+ · 100% offline · No telemetry
<div align="center">
| Generate | Models | Library | Settings |
|:---:|:---:|:---:|:---:|
| <img src="https://raw.githubusercontent.com/CodeBySonu95/VoxSherpa-TTS/main/fastlane/metadata/android/en-US/images/phoneScreenshots/1.jpg" width="180"/> | <img src="https://raw.githubusercontent.com/CodeBySonu95/VoxSherpa-TTS/main/fastlane/metadata/android/en-US/images/phoneScreenshots/2.jpg" width="180"/> | <img src="https://raw.githubusercontent.com/CodeBySonu95/VoxSherpa-TTS/main/fastlane/metadata/android/en-US/images/phoneScreenshots/3.jpg" width="180"/> | <img src="https://raw.githubusercontent.com/CodeBySonu95/VoxSherpa-TTS/main/fastlane/metadata/android/en-US/images/phoneScreenshots/4.jpg" width="180"/> |
</div>
---
### [BreezeApp](https://github.com/mtkresearch/BreezeApp) from [MediaTek Research](https://github.com/mtkresearch)
> BreezeAPP is a mobile AI application developed for both Android and iOS platforms.
> Users can download it directly from the App Store and enjoy a variety of features
> offline, including speech-to-text, text-to-speech, text-based chatbot interactions,
> and image question-answering
- [Download APK for BreezeAPP](https://huggingface.co/MediaTek-Research/BreezeApp/resolve/main/BreezeApp.apk)
- [APK 中国镜像](https://hf-mirror.com/MediaTek-Research/BreezeApp/blob/main/BreezeApp.apk)
| 1 | 2 | 3 |
|---|---|---|
|![](https://github.com/user-attachments/assets/1cdbc057-b893-4de6-9e9c-f1d7dfd1d992)|![](https://github.com/user-attachments/assets/d77cd98e-b057-442f-860d-d5befd5c769b)|![](https://github.com/user-attachments/assets/57e546bf-3d39-45b9-b392-b48ca4fb3c58)|
### [Open-LLM-VTuber](https://github.com/t41372/Open-LLM-VTuber)
Talk to any LLM with hands-free voice interaction, voice interruption, and Live2D taking
face running locally across platforms
See also <https://github.com/t41372/Open-LLM-VTuber/pull/50>
### [voiceapi](https://github.com/ruzhila/voiceapi)
<details>
<summary>Streaming ASR and TTS based on FastAPI</summary>
It shows how to use the ASR and TTS Python APIs with FastAPI.
</details>
### [腾讯会议摸鱼工具 TMSpeech](https://github.com/jxlpzqc/TMSpeech)
Uses streaming ASR in C# with graphical user interface.
Video demo in Chinese: [【开源】Windows实时字幕软件(网课/开会必备)](https://www.bilibili.com/video/BV1rX4y1p7Nx)
### [lol互动助手](https://github.com/l1veIn/lol-wom-electron)
It uses the JavaScript API of sherpa-onnx along with [Electron](https://electronjs.org/)
Video demo in Chinese: [爆了!炫神教你开打字挂!真正影响胜率的英雄联盟工具!英雄联盟的最后一块拼图!和游戏中的每个人无障碍沟通!](https://www.bilibili.com/video/BV142tje9E74)
### [Sherpa-ONNX 语音识别服务器](https://github.com/hfyydd/sherpa-onnx-server)
A server based on nodejs providing Restful API for speech recognition.
### [QSmartAssistant](https://github.com/xinhecuican/QSmartAssistant)
一个模块化,全过程可离线,低占用率的对话机器人/智能音箱
It uses QT. Both [ASR](https://github.com/xinhecuican/QSmartAssistant/blob/master/doc/%E5%AE%89%E8%A3%85.md#asr)
and [TTS](https://github.com/xinhecuican/QSmartAssistant/blob/master/doc/%E5%AE%89%E8%A3%85.md#tts)
are used.
### [Flutter-EasySpeechRecognition](https://github.com/Jason-chen-coder/Flutter-EasySpeechRecognition)
It extends [./flutter-examples/streaming_asr](./flutter-examples/streaming_asr) by
downloading models inside the app to reduce the size of the app.
Note: [[Team B] Sherpa AI backend](https://github.com/umgc/spring2025/pull/82) also uses
sherpa-onnx in a Flutter APP.
### [sherpa-onnx-unity](https://github.com/xue-fei/sherpa-onnx-unity)
sherpa-onnx in Unity. See also [#1695](https://github.com/k2-fsa/sherpa-onnx/issues/1695),
[#1892](https://github.com/k2-fsa/sherpa-onnx/issues/1892), and [#1859](https://github.com/k2-fsa/sherpa-onnx/issues/1859)
### [xiaozhi-esp32-server](https://github.com/xinnan-tech/xiaozhi-esp32-server)
本项目为xiaozhi-esp32提供后端服务,帮助您快速搭建ESP32设备控制服务器
Backend service for xiaozhi-esp32, helps you quickly build an ESP32 device control server.
See also
- [ASR新增轻量级sherpa-onnx-asr](https://github.com/xinnan-tech/xiaozhi-esp32-server/issues/315)
- [feat: ASR增加sherpa-onnx模型](https://github.com/xinnan-tech/xiaozhi-esp32-server/pull/379)
### [KaithemAutomation](https://github.com/EternityForest/KaithemAutomation)
Pure Python, GUI-focused home automation/consumer grade SCADA.
It uses TTS from sherpa-onnx. See also [✨ Speak command that uses the new globally configured TTS model.](https://github.com/EternityForest/KaithemAutomation/commit/8e64d2b138725e426532f7d66bb69dd0b4f53693)
### [Open-XiaoAI KWS](https://github.com/idootop/open-xiaoai-kws)
Enable custom wake word for XiaoAi Speakers. 让小爱音箱支持自定义唤醒词。
Video demo in Chinese: [小爱同学启动~˶╹ꇴ╹˶!](https://www.bilibili.com/video/BV1YfVUz5EMj)
### [C++ WebSocket ASR Server](https://github.com/mawwalker/stt-server)
It provides a WebSocket server based on C++ for ASR using sherpa-onnx.
### [Go WebSocket Server](https://github.com/bbeyondllove/asr_server)
It provides a WebSocket server based on the Go programming language for sherpa-onnx.
### [Making robot Paimon, Ep10 "The AI Part 1"](https://www.youtube.com/watch?v=KxPKkwxGWZs)
It is a [YouTube video](https://www.youtube.com/watch?v=KxPKkwxGWZs),
showing how the author tried to use AI so he can have a conversation with Paimon.
It uses sherpa-onnx for speech-to-text and text-to-speech.
|1|
|---|
|![](https://github.com/user-attachments/assets/f6eea2d5-1807-42cb-9160-be8da2971e1f)|
### [TtsReader - Desktop application](https://github.com/ys-pro-duction/TtsReader)
A desktop text-to-speech application built using Kotlin Multiplatform.
### [MentraOS](https://github.com/Mentra-Community/MentraOS)
> Smart glasses OS, with dozens of built-in apps. Users get AI assistant, notifications,
> translation, screen mirror, captions, and more. Devs get to write 1 app that runs on
> any pair of smart glasses.
It uses sherpa-onnx for real-time speech recognition on iOS and Android devices.
See also <https://github.com/Mentra-Community/MentraOS/pull/861>
It uses Swift for iOS and Java for Android.
### [flet_sherpa_onnx](https://github.com/SamYuan1990/flet_sherpa_onnx)
Flet ASR/STT component based on sherpa-onnx.
Example [a chat box agent](https://github.com/SamYuan1990/i18n-agent-action)
### [achatbot-go](https://github.com/ai-bot-pro/achatbot-go)
a multimodal chatbot based on go with sherpa-onnx's speech lib api.
### [fcitx5-vinput](https://github.com/xifan2333/fcitx5-vinput)
Local offline voice input plugin for [Fcitx5](https://github.com/fcitx/fcitx5) (Linux input method framework).
It uses C++ with offline ASR for speech recognition, supporting push-to-talk,
command mode, and optional LLM post-processing.
Video demo in Chinese: [fcitx5-vinput](https://www.bilibili.com/video/BV1a6cUzVE6F)
### [Wake Word](https://github.com/analyticsinmotion/wake-word)
A VS Code extension for hands-free voice-activated coding. It uses sherpa-onnx for real-time
keyword spotting (KWS) to detect custom wake phrases and trigger VS Code commands by voice.
Audio capture is handled by [decibri](https://github.com/analyticsinmotion/decibri), a
cross-platform Node.js microphone streaming library with prebuilt native binaries.
- [VS Code Marketplace](https://marketplace.visualstudio.com/items?itemName=analytics-in-motion.wake-word)
- [Open VSX](https://open-vsx.org/extension/analytics-in-motion/wake-word)
- [decibri integration guides for sherpa-onnx](https://decibri.dev/docs/node/integrations/sherpa-onnx-stt.html)
### [SmartSub](https://github.com/buxuku/SmartSub)
> SmartSub is a local-first cross-platform desktop application for the complete subtitle production pipeline: audio/video transcription, subtitle translation, proofreading, and subtitle burning/muxing.
>
> It natively integrates sherpa-onnx to power three offline ASR engines — FunASR, Qwen3-ASR, and FireRedASR — delivering high-accuracy Chinese and multilingual speech recognition entirely on-device, with no file uploads required.
[silero-vad]: https://github.com/snakers4/silero-vad
[Raspberry Pi]: https://www.raspberrypi.com/
[RV1126]: https://www.rock-chips.com/uploads/pdf/2022.8.26/191/RV1126%20Brief%20Datasheet.pdf
[LicheePi4A]: https://sipeed.com/licheepi4a
[VisionFive 2]: https://www.starfivetech.com/en/site/boards
[旭日X3派]: https://developer.horizon.ai/api/v1/fileData/documents_pi/index.html
[爱芯派]: https://wiki.sipeed.com/hardware/zh/maixIII/ax-pi/axpi.html
[hf-space-speaker-diarization]: https://huggingface.co/spaces/k2-fsa/speaker-diarization
[hf-space-speaker-diarization-cn]: https://hf.qhduan.com/spaces/k2-fsa/speaker-diarization
[hf-space-asr]: https://huggingface.co/spaces/k2-fsa/automatic-speech-recognition
[hf-space-asr-cn]: https://hf.qhduan.com/spaces/k2-fsa/automatic-speech-recognition
[Whisper]: https://github.com/openai/whisper
[hf-space-asr-whisper]: https://huggingface.co/spaces/k2-fsa/automatic-speech-recognition-with-whisper
[hf-space-asr-whisper-cn]: https://hf.qhduan.com/spaces/k2-fsa/automatic-speech-recognition-with-whisper
[hf-space-tts]: https://huggingface.co/spaces/k2-fsa/text-to-speech
[hf-space-tts-cn]: https://hf.qhduan.com/spaces/k2-fsa/text-to-speech
[hf-space-subtitle]: https://huggingface.co/spaces/k2-fsa/generate-subtitles-for-videos
[hf-space-subtitle-cn]: https://hf.qhduan.com/spaces/k2-fsa/generate-subtitles-for-videos
[hf-space-audio-tagging]: https://huggingface.co/spaces/k2-fsa/audio-tagging
[hf-space-audio-tagging-cn]: https://hf.qhduan.com/spaces/k2-fsa/audio-tagging
[hf-space-source-separation]: https://huggingface.co/spaces/k2-fsa/source-separation
[hf-space-source-separation-cn]: https://hf.qhduan.com/spaces/k2-fsa/source-separation
[hf-space-slid-whisper]: https://huggingface.co/spaces/k2-fsa/spoken-language-identification
[hf-space-slid-whisper-cn]: https://hf.qhduan.com/spaces/k2-fsa/spoken-language-identification
[wasm-hf-vad]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-sherpa-onnx
[wasm-ms-vad]: https://modelscope.cn/studios/csukuangfj/web-assembly-vad-sherpa-onnx
[wasm-hf-streaming-asr-zh-en-zipformer]: https://huggingface.co/spaces/k2-fsa/web-assembly-asr-sherpa-onnx-zh-en
[wasm-ms-streaming-asr-zh-en-zipformer]: https://modelscope.cn/studios/k2-fsa/web-assembly-asr-sherpa-onnx-zh-en
[wasm-hf-streaming-asr-zh-en-paraformer]: https://huggingface.co/spaces/k2-fsa/web-assembly-asr-sherpa-onnx-zh-en-paraformer
[wasm-ms-streaming-asr-zh-en-paraformer]: https://modelscope.cn/studios/k2-fsa/web-assembly-asr-sherpa-onnx-zh-en-paraformer
[Paraformer-large]: https://www.modelscope.cn/models/damo/speech_paraformer-large_asr_nat-zh-cn-16k-common-vocab8404-pytorch/summary
[wasm-hf-streaming-asr-zh-en-yue-paraformer]: https://huggingface.co/spaces/k2-fsa/web-assembly-asr-sherpa-onnx-zh-cantonese-en-paraformer
[wasm-ms-streaming-asr-zh-en-yue-paraformer]: https://modelscope.cn/studios/k2-fsa/web-assembly-asr-sherpa-onnx-zh-cantonese-en-paraformer
[wasm-hf-streaming-asr-en-zipformer]: https://huggingface.co/spaces/k2-fsa/web-assembly-asr-sherpa-onnx-en
[wasm-ms-streaming-asr-en-zipformer]: https://modelscope.cn/studios/k2-fsa/web-assembly-asr-sherpa-onnx-en
[SenseVoice]: https://github.com/FunAudioLLM/SenseVoice
[wasm-hf-vad-asr-zh-zipformer-ctc-07-03]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-zipformer-ctc
[wasm-ms-vad-asr-zh-zipformer-ctc-07-03]: https://modelscope.cn/studios/csukuangfj/web-assembly-vad-asr-sherpa-onnx-zh-zipformer-ctc/summary
[wasm-hf-vad-asr-zh-en-ko-ja-yue-sense-voice]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-en-ja-ko-cantonese-sense-voice
[wasm-ms-vad-asr-zh-en-ko-ja-yue-sense-voice]: https://www.modelscope.cn/studios/csukuangfj/web-assembly-vad-asr-sherpa-onnx-zh-en-jp-ko-cantonese-sense-voice
[wasm-hf-vad-asr-en-whisper-tiny-en]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-en-whisper-tiny
[wasm-ms-vad-asr-en-whisper-tiny-en]: https://www.modelscope.cn/studios/csukuangfj/web-assembly-vad-asr-sherpa-onnx-en-whisper-tiny
[wasm-hf-vad-asr-en-moonshine-tiny-en]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-en-moonshine-tiny
[wasm-ms-vad-asr-en-moonshine-tiny-en]: https://www.modelscope.cn/studios/csukuangfj/web-assembly-vad-asr-sherpa-onnx-en-moonshine-tiny
[wasm-hf-vad-asr-en-zipformer-gigaspeech]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-en-zipformer-gigaspeech
[wasm-ms-vad-asr-en-zipformer-gigaspeech]: https://www.modelscope.cn/studios/k2-fsa/web-assembly-vad-asr-sherpa-onnx-en-zipformer-gigaspeech
[wasm-hf-vad-asr-zh-zipformer-wenetspeech]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-zipformer-wenetspeech
[wasm-ms-vad-asr-zh-zipformer-wenetspeech]: https://www.modelscope.cn/studios/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-zipformer-wenetspeech
[reazonspeech]: https://research.reazon.jp/_static/reazonspeech_nlp2023.pdf
[wasm-hf-vad-asr-ja-zipformer-reazonspeech]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-ja-zipformer
[wasm-ms-vad-asr-ja-zipformer-reazonspeech]: https://www.modelscope.cn/studios/csukuangfj/web-assembly-vad-asr-sherpa-onnx-ja-zipformer
[gigaspeech2]: https://github.com/speechcolab/gigaspeech2
[wasm-hf-vad-asr-th-zipformer-gigaspeech2]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-th-zipformer
[wasm-ms-vad-asr-th-zipformer-gigaspeech2]: https://www.modelscope.cn/studios/csukuangfj/web-assembly-vad-asr-sherpa-onnx-th-zipformer
[telespeech-asr]: https://github.com/tele-ai/telespeech-asr
[wasm-hf-vad-asr-zh-telespeech]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-telespeech
[wasm-ms-vad-asr-zh-telespeech]: https://www.modelscope.cn/studios/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-telespeech
[wasm-hf-vad-asr-zh-en-paraformer-large]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-en-paraformer
[wasm-ms-vad-asr-zh-en-paraformer-large]: https://www.modelscope.cn/studios/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-en-paraformer
[wasm-hf-vad-asr-zh-en-paraformer-small]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-en-paraformer-small
[wasm-ms-vad-asr-zh-en-paraformer-small]: https://www.modelscope.cn/studios/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-en-paraformer-small
[dolphin]: https://github.com/dataoceanai/dolphin
[wasm-ms-vad-asr-multi-lang-dolphin-base]: https://modelscope.cn/studios/csukuangfj/web-assembly-vad-asr-sherpa-onnx-multi-lang-dophin-ctc
[wasm-hf-vad-asr-multi-lang-dolphin-base]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-multi-lang-dophin-ctc
[wasm-hf-tts-matcha-zh-en]: https://huggingface.co/spaces/k2-fsa/web-assembly-zh-en-tts-matcha
[wasm-hf-tts-matcha-zh]: https://huggingface.co/spaces/k2-fsa/web-assembly-zh-tts-matcha
[wasm-ms-tts-matcha-zh-en]: https://modelscope.cn/studios/csukuangfj/web-assembly-zh-en-tts-matcha
[wasm-ms-tts-matcha-zh]: https://modelscope.cn/studios/csukuangfj/web-assembly-zh-tts-matcha
[wasm-hf-tts-matcha-en]: https://huggingface.co/spaces/k2-fsa/web-assembly-en-tts-matcha
[wasm-ms-tts-matcha-en]: https://modelscope.cn/studios/csukuangfj/web-assembly-en-tts-matcha
[wasm-hf-tts-piper-en]: https://huggingface.co/spaces/k2-fsa/web-assembly-tts-sherpa-onnx-en
[wasm-ms-tts-piper-en]: https://modelscope.cn/studios/k2-fsa/web-assembly-tts-sherpa-onnx-en
[wasm-hf-tts-piper-de]: https://huggingface.co/spaces/k2-fsa/web-assembly-tts-sherpa-onnx-de
[wasm-ms-tts-piper-de]: https://modelscope.cn/studios/k2-fsa/web-assembly-tts-sherpa-onnx-de
[wasm-hf-speaker-diarization]: https://huggingface.co/spaces/k2-fsa/web-assembly-speaker-diarization-sherpa-onnx
[wasm-ms-speaker-diarization]: https://www.modelscope.cn/studios/csukuangfj/web-assembly-speaker-diarization-sherpa-onnx
[wasm-hf-voice-cloning-zipvoice]: https://huggingface.co/spaces/k2-fsa/web-assembly-zh-en-tts-zipvoice
[wasm-ms-voice-cloning-zipvoice]: https://modelscope.cn/studios/csukuangfj/web-assembly-zh-en-tts-zipvoice
[wasm-hf-voice-cloning-pocket]: https://huggingface.co/spaces/k2-fsa/web-assembly-en-tts-pocket
[wasm-ms-voice-cloning-pocket]: https://modelscope.cn/studios/csukuangfj/web-assembly-en-tts-pocket
[apk-speaker-diarization]: https://k2-fsa.github.io/sherpa/onnx/speaker-diarization/apk.html
[apk-speaker-diarization-cn]: https://k2-fsa.github.io/sherpa/onnx/speaker-diarization/apk-cn.html
[apk-streaming-asr]: https://k2-fsa.github.io/sherpa/onnx/android/apk.html
[apk-streaming-asr-cn]: https://k2-fsa.github.io/sherpa/onnx/android/apk-cn.html
[apk-simula-streaming-asr]: https://k2-fsa.github.io/sherpa/onnx/android/apk-simulate-streaming-asr.html
[apk-simula-streaming-asr-cn]: https://k2-fsa.github.io/sherpa/onnx/android/apk-simulate-streaming-asr-cn.html
[apk-tts]: https://k2-fsa.github.io/sherpa/onnx/tts/apk-engine.html
[apk-tts-cn]: https://k2-fsa.github.io/sherpa/onnx/tts/apk-engine-cn.html
[apk-vad]: https://k2-fsa.github.io/sherpa/onnx/vad/apk.html
[apk-vad-cn]: https://k2-fsa.github.io/sherpa/onnx/vad/apk-cn.html
[apk-vad-asr]: https://k2-fsa.github.io/sherpa/onnx/vad/apk-asr.html
[apk-vad-asr-cn]: https://k2-fsa.github.io/sherpa/onnx/vad/apk-asr-cn.html
[apk-2pass]: https://k2-fsa.github.io/sherpa/onnx/android/apk-2pass.html
[apk-2pass-cn]: https://k2-fsa.github.io/sherpa/onnx/android/apk-2pass-cn.html
[apk-at]: https://k2-fsa.github.io/sherpa/onnx/audio-tagging/apk.html
[apk-at-cn]: https://k2-fsa.github.io/sherpa/onnx/audio-tagging/apk-cn.html
[apk-at-wearos]: https://k2-fsa.github.io/sherpa/onnx/audio-tagging/apk-wearos.html
[apk-at-wearos-cn]: https://k2-fsa.github.io/sherpa/onnx/audio-tagging/apk-wearos-cn.html
[apk-sid]: https://k2-fsa.github.io/sherpa/onnx/speaker-identification/apk.html
[apk-sid-cn]: https://k2-fsa.github.io/sherpa/onnx/speaker-identification/apk-cn.html
[apk-slid]: https://k2-fsa.github.io/sherpa/onnx/spoken-language-identification/apk.html
[apk-slid-cn]: https://k2-fsa.github.io/sherpa/onnx/spoken-language-identification/apk-cn.html
[apk-kws]: https://k2-fsa.github.io/sherpa/onnx/kws/apk.html
[apk-kws-cn]: https://k2-fsa.github.io/sherpa/onnx/kws/apk-cn.html
[apk-flutter-streaming-asr]: https://k2-fsa.github.io/sherpa/onnx/flutter/pre-built-app.html#streaming-speech-recognition-stt-asr
[apk-flutter-streaming-asr-cn]: https://k2-fsa.github.io/sherpa/onnx/flutter/pre-built-app.html#streaming-speech-recognition-stt-asr
[flutter-tts-android]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-android.html
[flutter-tts-android-cn]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-android-cn.html
[flutter-tts-linux]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-linux.html
[flutter-tts-linux-cn]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-linux-cn.html
[flutter-tts-macos-x64]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-macos-x64.html
[flutter-tts-macos-arm64-cn]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-macos-arm64-cn.html
[flutter-tts-macos-arm64]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-macos-arm64.html
[flutter-tts-macos-x64-cn]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-macos-x64-cn.html
[flutter-tts-win-x64]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-win.html
[flutter-tts-win-x64-cn]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-win-cn.html
[lazarus-subtitle]: https://k2-fsa.github.io/sherpa/onnx/lazarus/download-generated-subtitles.html
[lazarus-subtitle-cn]: https://k2-fsa.github.io/sherpa/onnx/lazarus/download-generated-subtitles-cn.html
[asr-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/asr-models
[tts-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/tts-models
[vad-models]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/silero_vad.onnx
[kws-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/kws-models
[at-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/audio-tagging-models
[sid-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/speaker-recongition-models
[slid-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/speaker-recongition-models
[punct-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/punctuation-models
[speaker-segmentation-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/speaker-segmentation-models
[GigaSpeech]: https://github.com/SpeechColab/GigaSpeech
[WenetSpeech]: https://github.com/wenet-e2e/WenetSpeech
[sherpa-onnx-streaming-zipformer-bilingual-zh-en-2023-02-20]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-streaming-zipformer-bilingual-zh-en-2023-02-20.tar.bz2
[sherpa-onnx-streaming-zipformer-small-bilingual-zh-en-2023-02-16]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-streaming-zipformer-small-bilingual-zh-en-2023-02-16.tar.bz2
[sherpa-onnx-streaming-zipformer-korean-2024-06-16]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-streaming-zipformer-korean-2024-06-16.tar.bz2
[sherpa-onnx-streaming-zipformer-zh-14M-2023-02-23]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-streaming-zipformer-zh-14M-2023-02-23.tar.bz2
[sherpa-onnx-streaming-zipformer-en-20M-2023-02-17]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-streaming-zipformer-en-20M-2023-02-17.tar.bz2
[sherpa-onnx-zipformer-ja-reazonspeech-2024-08-01]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-zipformer-ja-reazonspeech-2024-08-01.tar.bz2
[sherpa-onnx-zipformer-ru-2024-09-18]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-zipformer-ru-2024-09-18.tar.bz2
[sherpa-onnx-zipformer-korean-2024-06-24]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-zipformer-korean-2024-06-24.tar.bz2
[sherpa-onnx-zipformer-thai-2024-06-20]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-zipformer-thai-2024-06-20.tar.bz2
[sherpa-onnx-nemo-transducer-giga-am-russian-2024-10-24]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-nemo-transducer-giga-am-russian-2024-10-24.tar.bz2
[sherpa-onnx-paraformer-zh-2024-03-09]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-paraformer-zh-2024-03-09.tar.bz2
[sherpa-onnx-nemo-ctc-giga-am-russian-2024-10-24]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-nemo-ctc-giga-am-russian-2024-10-24.tar.bz2
[sherpa-onnx-telespeech-ctc-int8-zh-2024-06-04]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-telespeech-ctc-int8-zh-2024-06-04.tar.bz2
[sherpa-onnx-sense-voice-zh-en-ja-ko-yue-2024-07-17]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-sense-voice-zh-en-ja-ko-yue-2024-07-17.tar.bz2
[sherpa-onnx-streaming-zipformer-fr-2023-04-14]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-streaming-zipformer-fr-2023-04-14.tar.bz2
[Moonshine tiny]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-moonshine-tiny-en-int8.tar.bz2
[NVIDIA Jetson Orin NX]: https://developer.download.nvidia.com/assets/embedded/secure/jetson/orin_nx/docs/Jetson_Orin_NX_DS-10712-001_v0.5.pdf?RCPGu9Q6OVAOv7a7vgtwc9-BLScXRIWq6cSLuditMALECJ_dOj27DgnqAPGVnT2VpiNpQan9SyFy-9zRykR58CokzbXwjSA7Gj819e91AXPrWkGZR3oS1VLxiDEpJa_Y0lr7UT-N4GnXtb8NlUkP4GkCkkF_FQivGPrAucCUywL481GH_WpP_p7ziHU1Wg==&t=eyJscyI6ImdzZW8iLCJsc2QiOiJodHRwczovL3d3dy5nb29nbGUuY29tLmhrLyJ9
[NVIDIA Jetson Nano B01]: https://www.seeedstudio.com/blog/2020/01/16/new-revision-of-jetson-nano-dev-kit-now-supports-new-jetson-nano-module/
[speech-enhancement-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/speech-enhancement-models
[source-separation-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/source-separation-models
[RK3588]: https://www.rock-chips.com/uploads/pdf/2022.8.26/192/RK3588%20Brief%20Datasheet.pdf
[spleeter]: https://github.com/deezer/spleeter
[UVR]: https://github.com/Anjok07/ultimatevocalremovergui
[gtcrn]: https://github.com/Xiaobin-Rong/gtcrn
[tts-url]: https://k2-fsa.github.io/sherpa/onnx/tts/all-in-one.html
[ss-url]: https://k2-fsa.github.io/sherpa/onnx/source-separation/index.html
[sd-url]: https://k2-fsa.github.io/sherpa/onnx/speaker-diarization/index.html
[slid-url]: https://k2-fsa.github.io/sherpa/onnx/spoken-language-identification/index.html
[at-url]: https://k2-fsa.github.io/sherpa/onnx/audio-tagging/index.html
[vad-url]: https://k2-fsa.github.io/sherpa/onnx/vad/index.html
[kws-url]: https://k2-fsa.github.io/sherpa/onnx/kws/index.html
[punct-url]: https://k2-fsa.github.io/sherpa/onnx/punctuation/index.html
[se-url]: https://k2-fsa.github.io/sherpa/onnx/speech-enhancement/index.html
[rknpu-doc]: https://k2-fsa.github.io/sherpa/onnx/rknn/index.html
[qnn-doc]: https://k2-fsa.github.io/sherpa/onnx/qnn/index.html
[ascend-doc]: https://k2-fsa.github.io/sherpa/onnx/ascend/index.html
[axera-npu]: https://axera-tech.com/Skill/166.html
[SpacemiT-K1]: https://cdn-resource.spacemit.com/file/chip/K1/K1_brief_zh.pdf
[SpacemiT-K3]: https://cdn-resource.spacemit.com/file/chip/K3/K3_brief_zh.pdf
+475
View File
@@ -0,0 +1,475 @@
use std::env;
use std::error::Error;
use std::ffi::OsStr;
use std::fs;
use std::fs::File;
use std::io;
use std::path::{Path, PathBuf};
use std::{collections::HashSet, ffi::OsString};
use bzip2::read::BzDecoder;
use tar::Archive;
const RELEASE_BASE_URL: &str = "https://github.com/k2-fsa/sherpa-onnx/releases/download";
const SHERPA_ONNX_STATIC_LIBS: &[&str] = &[
"sherpa-onnx-c-api",
"sherpa-onnx-core",
"kaldi-decoder-core",
"sherpa-onnx-kaldifst-core",
"sherpa-onnx-fstfar",
"sherpa-onnx-fst",
"kaldi-native-fbank-core",
"kissfft-float",
"piper_phonemize",
"espeak-ng",
"ucd",
"onnxruntime",
"ssentencepiece_core",
];
type DynError = Box<dyn Error>;
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
enum LinkMode {
Static,
Shared,
}
fn main() {
if let Err(err) = try_main() {
panic!("{err}");
}
}
fn try_main() -> Result<(), DynError> {
println!("cargo:rerun-if-env-changed=SHERPA_ONNX_LIB_DIR");
println!("cargo:rerun-if-env-changed=SHERPA_ONNX_ARCHIVE_DIR");
println!("cargo:rerun-if-env-changed=DOCS_RS");
if env::var_os("DOCS_RS").is_some() {
// docs.rs sets DOCS_RS=1; skip downloading/linking native libraries
// so that `cargo doc` can succeed without the real C artifacts.
return Ok(());
}
let target_os = env::var("CARGO_CFG_TARGET_OS")?;
let target_arch = env::var("CARGO_CFG_TARGET_ARCH")?;
let link_mode = resolve_link_mode()?;
let lib_dir = resolve_lib_dir(link_mode, &target_os, &target_arch)?;
println!("cargo:rustc-link-search=native={}", lib_dir.display());
if link_mode == LinkMode::Shared && matches!(target_os.as_str(), "linux" | "macos") {
println!("cargo:rustc-link-arg=-Wl,-rpath,{}", lib_dir.display());
emit_relative_rpath(&target_os);
copy_unix_runtime_libs(&lib_dir, &target_os)?;
}
if link_mode == LinkMode::Shared && target_os == "windows" {
copy_windows_runtime_dlls(&lib_dir)?;
}
match (link_mode, target_os.as_str()) {
(LinkMode::Static, "ios") => emit_ios_link_directives(),
(LinkMode::Shared, "android") => emit_android_link_directives(),
(LinkMode::Static, _) => emit_static_link_directives(&target_os),
(LinkMode::Shared, _) => emit_shared_link_directives(),
}
Ok(())
}
fn resolve_link_mode() -> Result<LinkMode, DynError> {
let static_enabled = env::var_os("CARGO_FEATURE_STATIC").is_some();
let shared_enabled = env::var_os("CARGO_FEATURE_SHARED").is_some();
if static_enabled && shared_enabled {
return Err("Features `static` and `shared` cannot be enabled at the same time".into());
}
if shared_enabled {
Ok(LinkMode::Shared)
} else {
Ok(LinkMode::Static)
}
}
fn resolve_lib_dir(
link_mode: LinkMode,
target_os: &str,
target_arch: &str,
) -> Result<PathBuf, DynError> {
if let Some(path) = env::var_os("SHERPA_ONNX_LIB_DIR") {
let path = PathBuf::from(path);
if !path.is_dir() {
return Err(format!(
"SHERPA_ONNX_LIB_DIR does not exist or is not a directory: {}",
path.display()
)
.into());
}
return Ok(path);
}
if matches!(target_os, "android" | "ios") {
return Err(format!(
"SHERPA_ONNX_LIB_DIR must point at the verified sherpa-onnx v{} \
mobile libraries when building for os={target_os}, arch={target_arch}",
env!("CARGO_PKG_VERSION")
)
.into());
}
download_prebuilt_libs(link_mode, target_os, target_arch)
}
fn download_prebuilt_libs(
link_mode: LinkMode,
target_os: &str,
target_arch: &str,
) -> Result<PathBuf, DynError> {
let archive_name = archive_name(link_mode, target_os, target_arch)?;
let archive_stem = archive_name.trim_end_matches(".tar.bz2");
let out_dir = PathBuf::from(env::var("OUT_DIR")?);
let cache_root = target_dir_from_out_dir(&out_dir)?.join("sherpa-onnx-prebuilt");
let extracted_dir = cache_root.join(archive_stem);
let lib_dir = extracted_dir.join("lib");
if lib_dir.is_dir() {
return Ok(lib_dir);
}
fs::create_dir_all(&cache_root)?;
let archive_path = cache_root.join(&archive_name);
if !archive_path.is_file() {
if let Some(local_archive_dir) = env::var_os("SHERPA_ONNX_ARCHIVE_DIR") {
let local_archive_path = PathBuf::from(local_archive_dir).join(&archive_name);
if !local_archive_path.is_file() {
return Err(format!(
"SHERPA_ONNX_ARCHIVE_DIR does not contain expected archive: {}",
local_archive_path.display()
)
.into());
}
copy_file_atomically(&local_archive_path, &archive_path)?;
} else {
let version = env!("CARGO_PKG_VERSION");
let url = format!("{RELEASE_BASE_URL}/v{version}/{archive_name}");
eprintln!("Downloading sherpa-onnx libs from {url}");
let response = ureq::builder()
.try_proxy_from_env(true)
.build()
.get(&url)
.call()
.map_err(|e| format!("Failed to download sherpa-onnx archive from {url}: {e}"))?;
let mut reader = response.into_reader();
write_reader_atomically(&mut reader, &archive_path)?;
}
}
if extracted_dir.exists() {
fs::remove_dir_all(&extracted_dir)?;
}
let unpack_result: Result<(), DynError> = (|| {
let tar_file = File::open(&archive_path)?;
let decoder = BzDecoder::new(tar_file);
let mut archive = Archive::new(decoder);
archive.unpack(&cache_root)?;
Ok(())
})();
if let Err(err) = unpack_result {
let _ = fs::remove_file(&archive_path);
let _ = fs::remove_dir_all(&extracted_dir);
return Err(format!(
"Failed to unpack cached archive {}: {err}",
archive_path.display()
)
.into());
}
if !lib_dir.is_dir() {
return Err(format!(
"Downloaded archive did not contain a lib directory: {}",
lib_dir.display()
)
.into());
}
eprintln!("Downloaded sherpa-onnx libs to {}", extracted_dir.display());
Ok(lib_dir)
}
fn archive_name(
link_mode: LinkMode,
target_os: &str,
target_arch: &str,
) -> Result<String, DynError> {
let version = env!("CARGO_PKG_VERSION");
let name = match (link_mode, target_os, target_arch) {
(LinkMode::Static, "linux", "x86_64") => {
format!("sherpa-onnx-v{version}-linux-x64-static-lib.tar.bz2")
}
(LinkMode::Static, "linux", "aarch64") => {
format!("sherpa-onnx-v{version}-linux-aarch64-static-lib.tar.bz2")
}
(LinkMode::Static, "macos", "x86_64") => {
format!("sherpa-onnx-v{version}-osx-x64-static-lib.tar.bz2")
}
(LinkMode::Static, "macos", "aarch64") => {
format!("sherpa-onnx-v{version}-osx-arm64-static-lib.tar.bz2")
}
(LinkMode::Static, "windows", "x86_64") => {
format!("sherpa-onnx-v{version}-win-x64-static-MT-Release-lib.tar.bz2")
}
(LinkMode::Shared, "linux", "x86_64") => {
format!("sherpa-onnx-v{version}-linux-x64-shared-lib.tar.bz2")
}
(LinkMode::Shared, "linux", "aarch64") => {
format!("sherpa-onnx-v{version}-linux-aarch64-shared-cpu-lib.tar.bz2")
}
(LinkMode::Shared, "macos", "x86_64") => {
format!("sherpa-onnx-v{version}-osx-x64-shared-lib.tar.bz2")
}
(LinkMode::Shared, "macos", "aarch64") => {
format!("sherpa-onnx-v{version}-osx-arm64-shared-lib.tar.bz2")
}
(LinkMode::Shared, "windows", "x86_64") => {
format!("sherpa-onnx-v{version}-win-x64-shared-MT-Release-lib.tar.bz2")
}
_ => {
return Err(format!(
"Unsupported target for sherpa-onnx prebuilt libs: os={target_os}, arch={target_arch}"
)
.into())
}
};
Ok(name)
}
fn emit_shared_link_directives() {
println!("cargo:rustc-link-lib=dylib=sherpa-onnx-c-api");
println!("cargo:rustc-link-lib=dylib=onnxruntime");
}
fn emit_android_link_directives() {
println!("cargo:rustc-link-lib=dylib=sherpa-onnx-c-api");
println!("cargo:rustc-link-lib=dylib=onnxruntime");
println!("cargo:rustc-link-lib=dylib=log");
}
fn emit_ios_link_directives() {
println!("cargo:rustc-link-lib=static=sherpa-onnx");
println!("cargo:rustc-link-lib=static=onnxruntime");
println!("cargo:rustc-link-lib=dylib=c++");
for framework in [
"Accelerate",
"CoreML",
"Foundation",
"Metal",
"QuartzCore",
"Security",
] {
println!("cargo:rustc-link-lib=framework={framework}");
}
}
fn emit_static_link_directives(target_os: &str) {
for lib in SHERPA_ONNX_STATIC_LIBS {
println!("cargo:rustc-link-lib=static={lib}");
}
match target_os {
"linux" => {
println!("cargo:rustc-link-lib=dylib=stdc++");
println!("cargo:rustc-link-lib=dylib=m");
println!("cargo:rustc-link-lib=dylib=pthread");
println!("cargo:rustc-link-lib=dylib=dl");
}
"macos" => {
println!("cargo:rustc-link-lib=dylib=c++");
println!("cargo:rustc-link-lib=framework=Foundation");
}
_ => {}
}
}
fn target_dir_from_out_dir(out_dir: &Path) -> Result<PathBuf, DynError> {
if let Ok(explicit_target_dir) = env::var("CARGO_TARGET_DIR") {
return Ok(PathBuf::from(explicit_target_dir));
}
if let Some(target_dir) = out_dir
.ancestors()
.find(|path| path.file_name() == Some(OsStr::new("target")))
{
return Ok(target_dir.to_path_buf());
}
Ok(out_dir.to_path_buf())
}
fn emit_relative_rpath(target_os: &str) {
match target_os {
"linux" => println!("cargo:rustc-link-arg=-Wl,-rpath,$ORIGIN"),
"macos" => println!("cargo:rustc-link-arg=-Wl,-rpath,@loader_path"),
_ => {}
}
}
fn profile_output_dirs() -> Result<[PathBuf; 2], DynError> {
let out_dir = PathBuf::from(env::var("OUT_DIR")?);
let profile = env::var("PROFILE")?;
let profile_dir = out_dir
.ancestors()
.find(|path| path.file_name() == Some(OsStr::new(&profile)))
.ok_or_else(|| {
format!(
"Could not locate Cargo profile directory from {}",
out_dir.display()
)
})?
.to_path_buf();
Ok([profile_dir.clone(), profile_dir.join("examples")])
}
fn copy_unix_runtime_libs(lib_dir: &Path, target_os: &str) -> Result<(), DynError> {
let runtime_libs: Vec<PathBuf> = fs::read_dir(lib_dir)?
.filter_map(|entry| entry.ok().map(|e| e.path()))
.filter(|path| {
path.file_name()
.and_then(OsStr::to_str)
.map(|name| match target_os {
"linux" => name.contains(".so"),
"macos" => name.ends_with(".dylib"),
_ => false,
})
.unwrap_or(false)
})
.collect();
if runtime_libs.is_empty() {
return Err(format!("No shared runtime libraries found in {}", lib_dir.display()).into());
}
let mut copy_plan = Vec::<(PathBuf, OsString)>::new();
let mut planned_names = HashSet::<OsString>::new();
for lib in runtime_libs {
if !lib.exists() {
continue;
}
let lib_name = lib
.file_name()
.ok_or_else(|| format!("Invalid runtime library path: {}", lib.display()))?
.to_os_string();
let source = fs::canonicalize(&lib).unwrap_or(lib.clone());
if planned_names.insert(lib_name.clone()) {
copy_plan.push((source.clone(), lib_name));
}
if let Some(source_name) = source.file_name() {
let source_name = source_name.to_os_string();
if planned_names.insert(source_name.clone()) {
copy_plan.push((source.clone(), source_name));
}
}
}
if copy_plan.is_empty() {
return Err(format!(
"No usable shared runtime libraries found in {}",
lib_dir.display()
)
.into());
}
for dest_dir in profile_output_dirs()? {
fs::create_dir_all(&dest_dir)?;
for (source, dest_name) in &copy_plan {
let dest = dest_dir.join(dest_name);
fs::copy(source, &dest)?;
}
}
Ok(())
}
fn temp_path_for(path: &Path) -> PathBuf {
let mut temp_name = path
.file_name()
.map(OsStr::to_os_string)
.unwrap_or_else(|| OsString::from("tmp"));
temp_name.push(".part");
path.with_file_name(temp_name)
}
fn copy_file_atomically(src: &Path, dst: &Path) -> Result<(), DynError> {
let temp_path = temp_path_for(dst);
if temp_path.exists() {
let _ = fs::remove_file(&temp_path);
}
fs::copy(src, &temp_path)?;
fs::rename(&temp_path, dst)?;
Ok(())
}
fn write_reader_atomically(reader: &mut dyn io::Read, dst: &Path) -> Result<(), DynError> {
let temp_path = temp_path_for(dst);
if temp_path.exists() {
let _ = fs::remove_file(&temp_path);
}
{
let mut file = File::create(&temp_path)?;
io::copy(reader, &mut file)?;
file.sync_all()?;
}
fs::rename(&temp_path, dst)?;
Ok(())
}
fn copy_windows_runtime_dlls(lib_dir: &Path) -> Result<(), DynError> {
let dlls: Vec<PathBuf> = fs::read_dir(lib_dir)?
.filter_map(|entry| entry.ok().map(|e| e.path()))
.filter(|path| path.extension() == Some(OsStr::new("dll")))
.collect();
if dlls.is_empty() {
println!(
"cargo:warning=No runtime DLLs found in {}",
lib_dir.display()
);
return Ok(());
}
let [profile_dir, examples_dir] = profile_output_dirs()?;
for dest_dir in [profile_dir.clone(), examples_dir] {
fs::create_dir_all(&dest_dir)?;
for dll in &dlls {
let dest = dest_dir.join(
dll.file_name()
.ok_or_else(|| format!("Invalid DLL path: {}", dll.display()))?,
);
fs::copy(dll, &dest)?;
}
}
println!(
"cargo:warning=Copied Windows runtime DLLs to {} and {}/examples",
profile_dir.display(),
profile_dir.display()
);
Ok(())
}
@@ -0,0 +1,60 @@
#![allow(non_camel_case_types)]
#![allow(non_snake_case)]
#![allow(non_upper_case_globals)]
use std::os::raw::{c_char, c_float};
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineZipformerAudioTaggingModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct AudioTaggingModelConfig {
pub zipformer: OfflineZipformerAudioTaggingModelConfig,
pub ced: *const c_char,
pub num_threads: i32,
pub debug: i32,
pub provider: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct AudioTaggingConfig {
pub model: AudioTaggingModelConfig,
pub labels: *const c_char,
pub top_k: i32,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct AudioEvent {
pub name: *const c_char,
pub index: i32,
pub prob: c_float,
}
#[repr(C)]
pub struct AudioTagging {
_private: [u8; 0],
}
extern "C" {
pub fn SherpaOnnxCreateAudioTagging(config: *const AudioTaggingConfig) -> *const AudioTagging;
pub fn SherpaOnnxDestroyAudioTagging(tagger: *const AudioTagging);
pub fn SherpaOnnxAudioTaggingCreateOfflineStream(
tagger: *const AudioTagging,
) -> *const crate::offline_asr::OfflineStream;
pub fn SherpaOnnxAudioTaggingCompute(
tagger: *const AudioTagging,
s: *const crate::offline_asr::OfflineStream,
top_k: i32,
) -> *const *const AudioEvent;
pub fn SherpaOnnxAudioTaggingFreeResults(p: *const *const AudioEvent);
}
+88
View File
@@ -0,0 +1,88 @@
#![allow(non_camel_case_types)]
#![allow(non_snake_case)]
#![allow(non_upper_case_globals)]
use std::os::raw::{c_char, c_float};
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct KeywordResult {
pub keyword: *const c_char,
pub tokens: *const c_char,
pub tokens_arr: *const *const c_char,
pub count: i32,
pub timestamps: *mut c_float,
pub start_time: c_float,
pub json: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct KeywordSpotterConfig {
pub feat_config: super::online_asr::FeatureConfig,
pub model_config: super::online_asr::OnlineModelConfig,
pub max_active_paths: i32,
pub num_trailing_blanks: i32,
pub keywords_score: c_float,
pub keywords_threshold: c_float,
pub keywords_file: *const c_char,
pub keywords_buf: *const c_char,
pub keywords_buf_size: i32,
}
#[repr(C)]
pub struct KeywordSpotter {
_private: [u8; 0],
}
extern "C" {
pub fn SherpaOnnxCreateKeywordSpotter(
config: *const KeywordSpotterConfig,
) -> *const KeywordSpotter;
pub fn SherpaOnnxDestroyKeywordSpotter(spotter: *const KeywordSpotter);
pub fn SherpaOnnxCreateKeywordStream(
spotter: *const KeywordSpotter,
) -> *const super::online_asr::OnlineStream;
pub fn SherpaOnnxCreateKeywordStreamWithKeywords(
spotter: *const KeywordSpotter,
keywords: *const c_char,
) -> *const super::online_asr::OnlineStream;
pub fn SherpaOnnxIsKeywordStreamReady(
spotter: *const KeywordSpotter,
stream: *const super::online_asr::OnlineStream,
) -> i32;
pub fn SherpaOnnxDecodeKeywordStream(
spotter: *const KeywordSpotter,
stream: *const super::online_asr::OnlineStream,
);
pub fn SherpaOnnxResetKeywordStream(
spotter: *const KeywordSpotter,
stream: *const super::online_asr::OnlineStream,
);
pub fn SherpaOnnxDecodeMultipleKeywordStreams(
spotter: *const KeywordSpotter,
streams: *const *const super::online_asr::OnlineStream,
n: i32,
);
pub fn SherpaOnnxGetKeywordResult(
spotter: *const KeywordSpotter,
stream: *const super::online_asr::OnlineStream,
) -> *const KeywordResult;
pub fn SherpaOnnxDestroyKeywordResult(r: *const KeywordResult);
pub fn SherpaOnnxGetKeywordResultAsJson(
spotter: *const KeywordSpotter,
stream: *const super::online_asr::OnlineStream,
) -> *const c_char;
pub fn SherpaOnnxFreeKeywordResultJson(s: *const c_char);
}
+42
View File
@@ -0,0 +1,42 @@
#![allow(non_camel_case_types)]
#![allow(non_snake_case)]
#![allow(non_upper_case_globals)]
use std::os::raw::c_char;
extern "C" {
pub fn SherpaOnnxGetVersionStr() -> *const c_char;
pub fn SherpaOnnxGetGitSha1() -> *const c_char;
pub fn SherpaOnnxGetGitDate() -> *const c_char;
pub fn SherpaOnnxFileExists(filename: *const c_char) -> i32;
}
pub mod audio_tagging;
pub mod kws;
pub mod offline_asr;
pub mod offline_punctuation;
pub mod offline_speaker_diarization;
pub mod online_asr;
pub mod online_punctuation;
pub mod resampler;
pub mod speaker_embedding;
pub mod speech_denoiser;
pub mod spoken_language_identification;
pub mod tts;
pub mod vad;
pub mod wave;
pub use audio_tagging::*;
pub use kws::*;
pub use offline_asr::*;
pub use offline_punctuation::*;
pub use offline_speaker_diarization::*;
pub use online_asr::*;
pub use online_punctuation::*;
pub use resampler::*;
pub use speaker_embedding::*;
pub use speech_denoiser::*;
pub use spoken_language_identification::*;
pub use tts::*;
pub use vad::*;
pub use wave::*;
+281
View File
@@ -0,0 +1,281 @@
#![allow(non_camel_case_types)]
#![allow(non_snake_case)]
#![allow(non_upper_case_globals)]
use std::os::raw::{c_char, c_float};
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineTransducerModelConfig {
pub encoder: *const c_char,
pub decoder: *const c_char,
pub joiner: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineParaformerModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineNemoEncDecCtcModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineWhisperModelConfig {
pub encoder: *const c_char,
pub decoder: *const c_char,
pub language: *const c_char,
pub task: *const c_char,
pub tail_paddings: i32,
pub enable_token_timestamps: i32,
pub enable_segment_timestamps: i32,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineCanaryModelConfig {
pub encoder: *const c_char,
pub decoder: *const c_char,
pub src_lang: *const c_char,
pub tgt_lang: *const c_char,
pub use_pnc: i32,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineFireRedAsrModelConfig {
pub encoder: *const c_char,
pub decoder: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineMoonshineModelConfig {
pub preprocessor: *const c_char,
pub encoder: *const c_char,
pub uncached_decoder: *const c_char,
pub cached_decoder: *const c_char,
pub merged_decoder: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineTdnnModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineLMConfig {
pub model: *const c_char,
pub scale: c_float,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineSenseVoiceModelConfig {
pub model: *const c_char,
pub language: *const c_char,
pub use_itn: i32,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineDolphinModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineZipformerCtcModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineWenetCtcModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineOmnilingualAsrCtcModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineFunASRNanoModelConfig {
pub encoder_adaptor: *const c_char,
pub llm: *const c_char,
pub embedding: *const c_char,
pub tokenizer: *const c_char,
pub system_prompt: *const c_char,
pub user_prompt: *const c_char,
pub max_new_tokens: i32,
pub temperature: c_float,
pub top_p: c_float,
pub seed: i32,
pub language: *const c_char,
pub itn: i32,
pub hotwords: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineMedAsrCtcModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineFireRedAsrCtcModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineQwen3ASRModelConfig {
pub conv_frontend: *const c_char,
pub encoder: *const c_char,
pub decoder: *const c_char,
pub tokenizer: *const c_char,
pub max_total_len: i32,
pub max_new_tokens: i32,
pub temperature: c_float,
pub top_p: c_float,
pub seed: i32,
pub hotwords: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineCohereTranscribeModelConfig {
pub encoder: *const c_char,
pub decoder: *const c_char,
pub language: *const c_char,
pub use_punct: i32,
pub use_itn: i32,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineModelConfig {
pub transducer: OfflineTransducerModelConfig,
pub paraformer: OfflineParaformerModelConfig,
pub nemo_ctc: OfflineNemoEncDecCtcModelConfig,
pub whisper: OfflineWhisperModelConfig,
pub tdnn: OfflineTdnnModelConfig,
pub tokens: *const c_char,
pub num_threads: i32,
pub debug: i32,
pub provider: *const c_char,
pub model_type: *const c_char,
pub modeling_unit: *const c_char,
pub bpe_vocab: *const c_char,
pub telespeech_ctc: *const c_char,
pub sense_voice: OfflineSenseVoiceModelConfig,
pub moonshine: OfflineMoonshineModelConfig,
pub fire_red_asr: OfflineFireRedAsrModelConfig,
pub dolphin: OfflineDolphinModelConfig,
pub zipformer_ctc: OfflineZipformerCtcModelConfig,
pub canary: OfflineCanaryModelConfig,
pub wenet_ctc: OfflineWenetCtcModelConfig,
pub omnilingual: OfflineOmnilingualAsrCtcModelConfig,
pub medasr: OfflineMedAsrCtcModelConfig,
pub funasr_nano: OfflineFunASRNanoModelConfig,
pub fire_red_asr_ctc: OfflineFireRedAsrCtcModelConfig,
pub qwen3_asr: OfflineQwen3ASRModelConfig,
pub cohere_transcribe: OfflineCohereTranscribeModelConfig,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineRecognizerConfig {
pub feat_config: super::online_asr::FeatureConfig,
pub model_config: OfflineModelConfig,
pub lm_config: OfflineLMConfig,
pub decoding_method: *const c_char,
pub max_active_paths: i32,
pub hotwords_file: *const c_char,
pub hotwords_score: c_float,
pub rule_fsts: *const c_char,
pub rule_fars: *const c_char,
pub blank_penalty: c_float,
pub hr: super::online_asr::HomophoneReplacerConfig,
}
#[repr(C)]
pub struct OfflineRecognizer {
_private: [u8; 0],
}
#[repr(C)]
pub struct OfflineStream {
_private: [u8; 0],
}
extern "C" {
pub fn SherpaOnnxCreateOfflineRecognizer(
config: *const OfflineRecognizerConfig,
) -> *const OfflineRecognizer;
pub fn SherpaOnnxDestroyOfflineRecognizer(recognizer: *const OfflineRecognizer);
pub fn SherpaOnnxCreateOfflineStream(
recognizer: *const OfflineRecognizer,
) -> *const OfflineStream;
pub fn SherpaOnnxCreateOfflineStreamWithHotwords(
recognizer: *const OfflineRecognizer,
hotwords: *const c_char,
) -> *const OfflineStream;
pub fn SherpaOnnxDestroyOfflineStream(stream: *const OfflineStream);
pub fn SherpaOnnxAcceptWaveformOffline(
stream: *const OfflineStream,
sample_rate: i32,
samples: *const f32,
n: i32,
);
pub fn SherpaOnnxOfflineStreamSetOption(
stream: *const OfflineStream,
key: *const c_char,
value: *const c_char,
);
pub fn SherpaOnnxOfflineStreamGetOption(
stream: *const OfflineStream,
key: *const c_char,
) -> *const c_char;
pub fn SherpaOnnxOfflineStreamHasOption(
stream: *const OfflineStream,
key: *const c_char,
) -> i32;
pub fn SherpaOnnxDecodeOfflineStream(
recognizer: *const OfflineRecognizer,
stream: *const OfflineStream,
);
pub fn SherpaOnnxDecodeMultipleOfflineStreams(
recognizer: *const OfflineRecognizer,
streams: *const *const OfflineStream,
n: i32,
);
pub fn SherpaOnnxGetOfflineStreamResultAsJson(stream: *const OfflineStream) -> *const c_char;
pub fn SherpaOnnxDestroyOfflineStreamResultJson(s: *const c_char);
}
@@ -0,0 +1,40 @@
#![allow(non_camel_case_types)]
#![allow(non_snake_case)]
#![allow(non_upper_case_globals)]
use std::os::raw::c_char;
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflinePunctuationModelConfig {
pub ct_transformer: *const c_char,
pub num_threads: i32,
pub debug: i32,
pub provider: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflinePunctuationConfig {
pub model: OfflinePunctuationModelConfig,
}
#[repr(C)]
pub struct OfflinePunctuation {
_private: [u8; 0],
}
extern "C" {
pub fn SherpaOnnxCreateOfflinePunctuation(
config: *const OfflinePunctuationConfig,
) -> *const OfflinePunctuation;
pub fn SherpaOnnxDestroyOfflinePunctuation(punct: *const OfflinePunctuation);
pub fn SherpaOfflinePunctuationAddPunct(
punct: *const OfflinePunctuation,
text: *const c_char,
) -> *const c_char;
pub fn SherpaOfflinePunctuationFreeText(text: *const c_char);
}
@@ -0,0 +1,98 @@
#![allow(non_camel_case_types)]
#![allow(non_snake_case)]
#![allow(non_upper_case_globals)]
use std::os::raw::{c_char, c_float};
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineSpeakerSegmentationPyannoteModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineSpeakerSegmentationModelConfig {
pub pyannote: OfflineSpeakerSegmentationPyannoteModelConfig,
pub num_threads: i32,
pub debug: i32,
pub provider: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct FastClusteringConfig {
pub num_clusters: i32,
pub threshold: c_float,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineSpeakerDiarizationConfig {
pub segmentation: OfflineSpeakerSegmentationModelConfig,
pub embedding: crate::speaker_embedding::SpeakerEmbeddingExtractorConfig,
pub clustering: FastClusteringConfig,
pub min_duration_on: c_float,
pub min_duration_off: c_float,
}
#[repr(C)]
pub struct OfflineSpeakerDiarization {
_private: [u8; 0],
}
#[repr(C)]
pub struct OfflineSpeakerDiarizationResult {
_private: [u8; 0],
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineSpeakerDiarizationSegment {
pub start: c_float,
pub end: c_float,
pub speaker: i32,
}
extern "C" {
pub fn SherpaOnnxCreateOfflineSpeakerDiarization(
config: *const OfflineSpeakerDiarizationConfig,
) -> *const OfflineSpeakerDiarization;
pub fn SherpaOnnxDestroyOfflineSpeakerDiarization(sd: *const OfflineSpeakerDiarization);
pub fn SherpaOnnxOfflineSpeakerDiarizationGetSampleRate(
sd: *const OfflineSpeakerDiarization,
) -> i32;
pub fn SherpaOnnxOfflineSpeakerDiarizationSetConfig(
sd: *const OfflineSpeakerDiarization,
config: *const OfflineSpeakerDiarizationConfig,
);
pub fn SherpaOnnxOfflineSpeakerDiarizationResultGetNumSpeakers(
r: *const OfflineSpeakerDiarizationResult,
) -> i32;
pub fn SherpaOnnxOfflineSpeakerDiarizationResultGetNumSegments(
r: *const OfflineSpeakerDiarizationResult,
) -> i32;
pub fn SherpaOnnxOfflineSpeakerDiarizationResultSortByStartTime(
r: *const OfflineSpeakerDiarizationResult,
) -> *const OfflineSpeakerDiarizationSegment;
pub fn SherpaOnnxOfflineSpeakerDiarizationDestroySegment(
s: *const OfflineSpeakerDiarizationSegment,
);
pub fn SherpaOnnxOfflineSpeakerDiarizationProcess(
sd: *const OfflineSpeakerDiarization,
samples: *const c_float,
n: i32,
) -> *const OfflineSpeakerDiarizationResult;
pub fn SherpaOnnxOfflineSpeakerDiarizationDestroyResult(
r: *const OfflineSpeakerDiarizationResult,
);
}
+202
View File
@@ -0,0 +1,202 @@
#![allow(non_camel_case_types)]
#![allow(non_snake_case)]
#![allow(non_upper_case_globals)]
use std::os::raw::{c_char, c_float};
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OnlineTransducerModelConfig {
pub encoder: *const c_char,
pub decoder: *const c_char,
pub joiner: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OnlineParaformerModelConfig {
pub encoder: *const c_char,
pub decoder: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OnlineZipformer2CtcModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OnlineNemoCtcModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OnlineToneCtcModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OnlineModelConfig {
pub transducer: OnlineTransducerModelConfig,
pub paraformer: OnlineParaformerModelConfig,
pub zipformer2_ctc: OnlineZipformer2CtcModelConfig,
pub tokens: *const c_char,
pub num_threads: i32,
pub provider: *const c_char,
pub debug: i32,
pub model_type: *const c_char,
// cjkchar | bpe | cjkchar+bpe
pub modeling_unit: *const c_char,
pub bpe_vocab: *const c_char,
pub tokens_buf: *const u8,
pub tokens_buf_size: i32,
pub nemo_ctc: OnlineNemoCtcModelConfig,
pub t_one_ctc: OnlineToneCtcModelConfig,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct FeatureConfig {
pub sample_rate: i32,
pub feature_dim: i32,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OnlineCtcFstDecoderConfig {
pub graph: *const c_char,
pub max_active: i32,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct HomophoneReplacerConfig {
pub dict_dir: *const c_char,
pub lexicon: *const c_char,
pub rule_fsts: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OnlineRecognizerConfig {
pub feat_config: FeatureConfig,
pub model_config: OnlineModelConfig,
// greedy_search | modified_beam_search
pub decoding_method: *const c_char,
pub max_active_paths: i32,
pub enable_endpoint: i32,
pub rule1_min_trailing_silence: c_float,
pub rule2_min_trailing_silence: c_float,
pub rule3_min_utterance_length: c_float,
pub hotwords_file: *const c_char,
pub hotwords_score: c_float,
pub ctc_fst_decoder_config: OnlineCtcFstDecoderConfig,
pub rule_fsts: *const c_char,
pub rule_fars: *const c_char,
pub blank_penalty: c_float,
pub hotwords_buf: *const u8,
pub hotwords_buf_size: i32,
pub hr: HomophoneReplacerConfig,
}
#[repr(C)]
pub struct OnlineRecognizer {
_private: [u8; 0],
}
#[repr(C)]
pub struct OnlineStream {
_private: [u8; 0],
}
extern "C" {
pub fn SherpaOnnxCreateOnlineRecognizer(
config: *const OnlineRecognizerConfig,
) -> *const OnlineRecognizer;
pub fn SherpaOnnxDestroyOnlineRecognizer(recognizer: *const OnlineRecognizer);
pub fn SherpaOnnxCreateOnlineStream(recognizer: *const OnlineRecognizer)
-> *const OnlineStream;
pub fn SherpaOnnxCreateOnlineStreamWithHotwords(
recognizer: *const OnlineRecognizer,
hotwords: *const c_char,
) -> *const OnlineStream;
pub fn SherpaOnnxDestroyOnlineStream(stream: *const OnlineStream);
pub fn SherpaOnnxOnlineStreamAcceptWaveform(
stream: *const OnlineStream,
sample_rate: i32,
samples: *const f32,
n: i32,
);
pub fn SherpaOnnxIsOnlineStreamReady(
recognizer: *const OnlineRecognizer,
stream: *const OnlineStream,
) -> i32;
pub fn SherpaOnnxDecodeOnlineStream(
recognizer: *const OnlineRecognizer,
stream: *const OnlineStream,
);
pub fn SherpaOnnxDecodeMultipleOnlineStreams(
recognizer: *const OnlineRecognizer,
streams: *const *const OnlineStream,
n: i32,
);
pub fn SherpaOnnxGetOnlineStreamResultAsJson(
recognizer: *const OnlineRecognizer,
stream: *const OnlineStream,
) -> *const c_char;
pub fn SherpaOnnxDestroyOnlineStreamResultJson(s: *const c_char);
pub fn SherpaOnnxOnlineStreamReset(
recognizer: *const OnlineRecognizer,
stream: *const OnlineStream,
);
pub fn SherpaOnnxOnlineStreamInputFinished(stream: *const OnlineStream);
pub fn SherpaOnnxOnlineStreamSetOption(
stream: *const OnlineStream,
key: *const c_char,
value: *const c_char,
);
pub fn SherpaOnnxOnlineStreamGetOption(
stream: *const OnlineStream,
key: *const c_char,
) -> *const c_char;
pub fn SherpaOnnxOnlineStreamHasOption(stream: *const OnlineStream, key: *const c_char) -> i32;
pub fn SherpaOnnxOnlineStreamIsEndpoint(
recognizer: *const OnlineRecognizer,
stream: *const OnlineStream,
) -> i32;
}
@@ -0,0 +1,41 @@
#![allow(non_camel_case_types)]
#![allow(non_snake_case)]
#![allow(non_upper_case_globals)]
use std::os::raw::c_char;
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OnlinePunctuationModelConfig {
pub cnn_bilstm: *const c_char,
pub bpe_vocab: *const c_char,
pub num_threads: i32,
pub debug: i32,
pub provider: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OnlinePunctuationConfig {
pub model: OnlinePunctuationModelConfig,
}
#[repr(C)]
pub struct OnlinePunctuation {
_private: [u8; 0],
}
extern "C" {
pub fn SherpaOnnxCreateOnlinePunctuation(
config: *const OnlinePunctuationConfig,
) -> *const OnlinePunctuation;
pub fn SherpaOnnxDestroyOnlinePunctuation(punctuation: *const OnlinePunctuation);
pub fn SherpaOnnxOnlinePunctuationAddPunct(
punctuation: *const OnlinePunctuation,
text: *const c_char,
) -> *const c_char;
pub fn SherpaOnnxOnlinePunctuationFreeText(text: *const c_char);
}
+42
View File
@@ -0,0 +1,42 @@
use std::os::raw::c_float;
#[repr(C)]
pub struct SherpaOnnxLinearResampler {
_private: [u8; 0],
}
#[repr(C)]
pub struct SherpaOnnxResampleOut {
pub samples: *const f32,
pub n: i32,
}
extern "C" {
pub fn SherpaOnnxCreateLinearResampler(
samp_rate_in_hz: i32,
samp_rate_out_hz: i32,
filter_cutoff_hz: c_float,
num_zeros: i32,
) -> *const SherpaOnnxLinearResampler;
pub fn SherpaOnnxDestroyLinearResampler(p: *const SherpaOnnxLinearResampler);
pub fn SherpaOnnxLinearResamplerReset(p: *const SherpaOnnxLinearResampler);
pub fn SherpaOnnxLinearResamplerResample(
p: *const SherpaOnnxLinearResampler,
input: *const f32,
input_dim: i32,
flush: i32,
) -> *const SherpaOnnxResampleOut;
pub fn SherpaOnnxLinearResamplerResampleFree(p: *const SherpaOnnxResampleOut);
pub fn SherpaOnnxLinearResamplerResampleGetInputSampleRate(
p: *const SherpaOnnxLinearResampler,
) -> i32;
pub fn SherpaOnnxLinearResamplerResampleGetOutputSampleRate(
p: *const SherpaOnnxLinearResampler,
) -> i32;
}
@@ -0,0 +1,131 @@
#![allow(non_camel_case_types)]
#![allow(non_snake_case)]
#![allow(non_upper_case_globals)]
use std::os::raw::{c_char, c_float};
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct SpeakerEmbeddingExtractorConfig {
pub model: *const c_char,
pub num_threads: i32,
pub debug: i32,
pub provider: *const c_char,
}
#[repr(C)]
pub struct SpeakerEmbeddingExtractor {
_private: [u8; 0],
}
#[repr(C)]
pub struct SpeakerEmbeddingManager {
_private: [u8; 0],
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct SpeakerEmbeddingManagerSpeakerMatch {
pub score: c_float,
pub name: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct SpeakerEmbeddingManagerBestMatchesResult {
pub matches: *const SpeakerEmbeddingManagerSpeakerMatch,
pub count: i32,
}
extern "C" {
pub fn SherpaOnnxCreateSpeakerEmbeddingExtractor(
config: *const SpeakerEmbeddingExtractorConfig,
) -> *const SpeakerEmbeddingExtractor;
pub fn SherpaOnnxDestroySpeakerEmbeddingExtractor(p: *const SpeakerEmbeddingExtractor);
pub fn SherpaOnnxSpeakerEmbeddingExtractorDim(p: *const SpeakerEmbeddingExtractor) -> i32;
pub fn SherpaOnnxSpeakerEmbeddingExtractorCreateStream(
p: *const SpeakerEmbeddingExtractor,
) -> *const crate::online_asr::OnlineStream;
pub fn SherpaOnnxSpeakerEmbeddingExtractorIsReady(
p: *const SpeakerEmbeddingExtractor,
s: *const crate::online_asr::OnlineStream,
) -> i32;
pub fn SherpaOnnxSpeakerEmbeddingExtractorComputeEmbedding(
p: *const SpeakerEmbeddingExtractor,
s: *const crate::online_asr::OnlineStream,
) -> *const c_float;
pub fn SherpaOnnxSpeakerEmbeddingExtractorDestroyEmbedding(v: *const c_float);
pub fn SherpaOnnxCreateSpeakerEmbeddingManager(dim: i32) -> *const SpeakerEmbeddingManager;
pub fn SherpaOnnxDestroySpeakerEmbeddingManager(p: *const SpeakerEmbeddingManager);
pub fn SherpaOnnxSpeakerEmbeddingManagerAdd(
p: *const SpeakerEmbeddingManager,
name: *const c_char,
v: *const c_float,
) -> i32;
pub fn SherpaOnnxSpeakerEmbeddingManagerAddList(
p: *const SpeakerEmbeddingManager,
name: *const c_char,
v: *const *const c_float,
) -> i32;
pub fn SherpaOnnxSpeakerEmbeddingManagerAddListFlattened(
p: *const SpeakerEmbeddingManager,
name: *const c_char,
v: *const c_float,
n: i32,
) -> i32;
pub fn SherpaOnnxSpeakerEmbeddingManagerRemove(
p: *const SpeakerEmbeddingManager,
name: *const c_char,
) -> i32;
pub fn SherpaOnnxSpeakerEmbeddingManagerSearch(
p: *const SpeakerEmbeddingManager,
v: *const c_float,
threshold: c_float,
) -> *const c_char;
pub fn SherpaOnnxSpeakerEmbeddingManagerFreeSearch(name: *const c_char);
pub fn SherpaOnnxSpeakerEmbeddingManagerGetBestMatches(
p: *const SpeakerEmbeddingManager,
v: *const c_float,
threshold: c_float,
n: i32,
) -> *const SpeakerEmbeddingManagerBestMatchesResult;
pub fn SherpaOnnxSpeakerEmbeddingManagerFreeBestMatches(
r: *const SpeakerEmbeddingManagerBestMatchesResult,
);
pub fn SherpaOnnxSpeakerEmbeddingManagerVerify(
p: *const SpeakerEmbeddingManager,
name: *const c_char,
v: *const c_float,
threshold: c_float,
) -> i32;
pub fn SherpaOnnxSpeakerEmbeddingManagerContains(
p: *const SpeakerEmbeddingManager,
name: *const c_char,
) -> i32;
pub fn SherpaOnnxSpeakerEmbeddingManagerNumSpeakers(p: *const SpeakerEmbeddingManager) -> i32;
pub fn SherpaOnnxSpeakerEmbeddingManagerGetAllSpeakers(
p: *const SpeakerEmbeddingManager,
) -> *const *const c_char;
pub fn SherpaOnnxSpeakerEmbeddingManagerFreeAllSpeakers(names: *const *const c_char);
}
@@ -0,0 +1,88 @@
use std::os::raw::c_char;
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineSpeechDenoiserGtcrnModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineSpeechDenoiserDpdfNetModelConfig {
pub model: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineSpeechDenoiserModelConfig {
pub gtcrn: OfflineSpeechDenoiserGtcrnModelConfig,
pub num_threads: i32,
pub debug: i32,
pub provider: *const c_char,
pub dpdfnet: OfflineSpeechDenoiserDpdfNetModelConfig,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineSpeechDenoiserConfig {
pub model: OfflineSpeechDenoiserModelConfig,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OnlineSpeechDenoiserConfig {
pub model: OfflineSpeechDenoiserModelConfig,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct DenoisedAudio {
pub samples: *const f32,
pub n: i32,
pub sample_rate: i32,
}
#[repr(C)]
pub struct OfflineSpeechDenoiser {
_private: [u8; 0],
}
#[repr(C)]
pub struct OnlineSpeechDenoiser {
_private: [u8; 0],
}
extern "C" {
pub fn SherpaOnnxCreateOfflineSpeechDenoiser(
config: *const OfflineSpeechDenoiserConfig,
) -> *const OfflineSpeechDenoiser;
pub fn SherpaOnnxDestroyOfflineSpeechDenoiser(p: *const OfflineSpeechDenoiser);
pub fn SherpaOnnxOfflineSpeechDenoiserGetSampleRate(p: *const OfflineSpeechDenoiser) -> i32;
pub fn SherpaOnnxOfflineSpeechDenoiserRun(
p: *const OfflineSpeechDenoiser,
samples: *const f32,
n: i32,
sample_rate: i32,
) -> *const DenoisedAudio;
pub fn SherpaOnnxCreateOnlineSpeechDenoiser(
config: *const OnlineSpeechDenoiserConfig,
) -> *const OnlineSpeechDenoiser;
pub fn SherpaOnnxDestroyOnlineSpeechDenoiser(p: *const OnlineSpeechDenoiser);
pub fn SherpaOnnxOnlineSpeechDenoiserGetSampleRate(p: *const OnlineSpeechDenoiser) -> i32;
pub fn SherpaOnnxOnlineSpeechDenoiserGetFrameShiftInSamples(
p: *const OnlineSpeechDenoiser,
) -> i32;
pub fn SherpaOnnxOnlineSpeechDenoiserRun(
p: *const OnlineSpeechDenoiser,
samples: *const f32,
n: i32,
sample_rate: i32,
) -> *const DenoisedAudio;
pub fn SherpaOnnxOnlineSpeechDenoiserFlush(
p: *const OnlineSpeechDenoiser,
) -> *const DenoisedAudio;
pub fn SherpaOnnxOnlineSpeechDenoiserReset(p: *const OnlineSpeechDenoiser);
pub fn SherpaOnnxDestroyDenoisedAudio(audio: *const DenoisedAudio);
}
@@ -0,0 +1,54 @@
#![allow(non_camel_case_types)]
#![allow(non_snake_case)]
#![allow(non_upper_case_globals)]
use std::os::raw::c_char;
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct SpokenLanguageIdentificationWhisperConfig {
pub encoder: *const c_char,
pub decoder: *const c_char,
pub tail_paddings: i32,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct SpokenLanguageIdentificationConfig {
pub whisper: SpokenLanguageIdentificationWhisperConfig,
pub num_threads: i32,
pub debug: i32,
pub provider: *const c_char,
}
#[repr(C)]
pub struct SpokenLanguageIdentification {
_private: [u8; 0],
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct SpokenLanguageIdentificationResult {
pub lang: *const c_char,
}
extern "C" {
pub fn SherpaOnnxCreateSpokenLanguageIdentification(
config: *const SpokenLanguageIdentificationConfig,
) -> *const SpokenLanguageIdentification;
pub fn SherpaOnnxDestroySpokenLanguageIdentification(slid: *const SpokenLanguageIdentification);
pub fn SherpaOnnxSpokenLanguageIdentificationCreateOfflineStream(
slid: *const SpokenLanguageIdentification,
) -> *const super::offline_asr::OfflineStream;
pub fn SherpaOnnxSpokenLanguageIdentificationCompute(
slid: *const SpokenLanguageIdentification,
stream: *const super::offline_asr::OfflineStream,
) -> *const SpokenLanguageIdentificationResult;
pub fn SherpaOnnxDestroySpokenLanguageIdentificationResult(
r: *const SpokenLanguageIdentificationResult,
);
}
+172
View File
@@ -0,0 +1,172 @@
#![allow(non_camel_case_types)]
#![allow(non_snake_case)]
#![allow(non_upper_case_globals)]
use std::os::raw::{c_char, c_float, c_void};
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineTtsVitsModelConfig {
pub model: *const c_char,
pub lexicon: *const c_char,
pub tokens: *const c_char,
pub data_dir: *const c_char,
pub noise_scale: c_float,
pub noise_scale_w: c_float,
pub length_scale: c_float,
pub dict_dir: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineTtsMatchaModelConfig {
pub acoustic_model: *const c_char,
pub vocoder: *const c_char,
pub lexicon: *const c_char,
pub tokens: *const c_char,
pub data_dir: *const c_char,
pub noise_scale: c_float,
pub length_scale: c_float,
pub dict_dir: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineTtsKokoroModelConfig {
pub model: *const c_char,
pub voices: *const c_char,
pub tokens: *const c_char,
pub data_dir: *const c_char,
pub length_scale: c_float,
pub dict_dir: *const c_char,
pub lexicon: *const c_char,
pub lang: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineTtsKittenModelConfig {
pub model: *const c_char,
pub voices: *const c_char,
pub tokens: *const c_char,
pub data_dir: *const c_char,
pub length_scale: c_float,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineTtsZipvoiceModelConfig {
pub tokens: *const c_char,
pub encoder: *const c_char,
pub decoder: *const c_char,
pub vocoder: *const c_char,
pub data_dir: *const c_char,
pub lexicon: *const c_char,
pub feat_scale: c_float,
pub t_shift: c_float,
pub target_rms: c_float,
pub guidance_scale: c_float,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineTtsPocketModelConfig {
pub lm_flow: *const c_char,
pub lm_main: *const c_char,
pub encoder: *const c_char,
pub decoder: *const c_char,
pub text_conditioner: *const c_char,
pub vocab_json: *const c_char,
pub token_scores_json: *const c_char,
pub voice_embedding_cache_capacity: i32,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineTtsSupertonicModelConfig {
pub duration_predictor: *const c_char,
pub text_encoder: *const c_char,
pub vector_estimator: *const c_char,
pub vocoder: *const c_char,
pub tts_json: *const c_char,
pub unicode_indexer: *const c_char,
pub voice_style: *const c_char,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineTtsModelConfig {
pub vits: OfflineTtsVitsModelConfig,
pub num_threads: i32,
pub debug: i32,
pub provider: *const c_char,
pub matcha: OfflineTtsMatchaModelConfig,
pub kokoro: OfflineTtsKokoroModelConfig,
pub kitten: OfflineTtsKittenModelConfig,
pub zipvoice: OfflineTtsZipvoiceModelConfig,
pub pocket: OfflineTtsPocketModelConfig,
pub supertonic: OfflineTtsSupertonicModelConfig,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct OfflineTtsConfig {
pub model: OfflineTtsModelConfig,
pub rule_fsts: *const c_char,
pub max_num_sentences: i32,
pub rule_fars: *const c_char,
pub silence_scale: c_float,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct SherpaOnnxGeneratedAudio {
pub samples: *const f32,
pub n: i32,
pub sample_rate: i32,
}
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct SherpaOnnxGenerationConfig {
pub silence_scale: c_float,
pub speed: c_float,
pub sid: i32,
pub reference_audio: *const f32,
pub reference_audio_len: i32,
pub reference_sample_rate: i32,
pub reference_text: *const c_char,
pub num_steps: i32,
pub extra: *const c_char,
}
pub type SherpaOnnxGeneratedAudioProgressCallbackWithArg = Option<
unsafe extern "C" fn(samples: *const f32, n: i32, progress: c_float, arg: *mut c_void) -> i32,
>;
#[repr(C)]
pub struct SherpaOnnxOfflineTts {
_private: [u8; 0],
}
extern "C" {
pub fn SherpaOnnxCreateOfflineTts(
config: *const OfflineTtsConfig,
) -> *const SherpaOnnxOfflineTts;
pub fn SherpaOnnxDestroyOfflineTts(tts: *const SherpaOnnxOfflineTts);
pub fn SherpaOnnxOfflineTtsSampleRate(tts: *const SherpaOnnxOfflineTts) -> i32;
pub fn SherpaOnnxOfflineTtsNumSpeakers(tts: *const SherpaOnnxOfflineTts) -> i32;
pub fn SherpaOnnxOfflineTtsGenerateWithConfig(
tts: *const SherpaOnnxOfflineTts,
text: *const c_char,
config: *const SherpaOnnxGenerationConfig,
callback: SherpaOnnxGeneratedAudioProgressCallbackWithArg,
arg: *mut c_void,
) -> *const SherpaOnnxGeneratedAudio;
pub fn SherpaOnnxDestroyOfflineTtsGeneratedAudio(p: *const SherpaOnnxGeneratedAudio);
}
+85
View File
@@ -0,0 +1,85 @@
use std::os::raw::{c_char, c_float};
#[repr(C)]
pub struct SileroVadModelConfig {
pub model: *const c_char,
pub threshold: c_float,
pub min_silence_duration: c_float,
pub min_speech_duration: c_float,
pub window_size: i32,
pub max_speech_duration: c_float,
}
#[repr(C)]
pub struct TenVadModelConfig {
pub model: *const c_char,
pub threshold: c_float,
pub min_silence_duration: c_float,
pub min_speech_duration: c_float,
pub window_size: i32,
pub max_speech_duration: c_float,
}
#[repr(C)]
pub struct VadModelConfig {
pub silero_vad: SileroVadModelConfig,
pub sample_rate: i32,
pub num_threads: i32,
pub provider: *const c_char,
pub debug: i32,
pub ten_vad: TenVadModelConfig,
}
#[repr(C)]
pub struct CircularBuffer {
_private: [u8; 0],
}
#[repr(C)]
pub struct SpeechSegment {
pub start: i32,
pub samples: *mut f32,
pub n: i32,
}
#[repr(C)]
pub struct VoiceActivityDetector {
_private: [u8; 0],
}
extern "C" {
pub fn SherpaOnnxCreateCircularBuffer(capacity: i32) -> *const CircularBuffer;
pub fn SherpaOnnxDestroyCircularBuffer(buffer: *const CircularBuffer);
pub fn SherpaOnnxCircularBufferPush(buffer: *const CircularBuffer, p: *const f32, n: i32);
pub fn SherpaOnnxCircularBufferGet(
buffer: *const CircularBuffer,
start_index: i32,
n: i32,
) -> *const f32;
pub fn SherpaOnnxCircularBufferFree(p: *const f32);
pub fn SherpaOnnxCircularBufferPop(buffer: *const CircularBuffer, n: i32);
pub fn SherpaOnnxCircularBufferSize(buffer: *const CircularBuffer) -> i32;
pub fn SherpaOnnxCircularBufferHead(buffer: *const CircularBuffer) -> i32;
pub fn SherpaOnnxCircularBufferReset(buffer: *const CircularBuffer);
pub fn SherpaOnnxCreateVoiceActivityDetector(
config: *const VadModelConfig,
buffer_size_in_seconds: c_float,
) -> *const VoiceActivityDetector;
pub fn SherpaOnnxDestroyVoiceActivityDetector(p: *const VoiceActivityDetector);
pub fn SherpaOnnxVoiceActivityDetectorAcceptWaveform(
p: *const VoiceActivityDetector,
samples: *const f32,
n: i32,
);
pub fn SherpaOnnxVoiceActivityDetectorEmpty(p: *const VoiceActivityDetector) -> i32;
pub fn SherpaOnnxVoiceActivityDetectorDetected(p: *const VoiceActivityDetector) -> i32;
pub fn SherpaOnnxVoiceActivityDetectorPop(p: *const VoiceActivityDetector);
pub fn SherpaOnnxVoiceActivityDetectorClear(p: *const VoiceActivityDetector);
pub fn SherpaOnnxVoiceActivityDetectorFront(
p: *const VoiceActivityDetector,
) -> *const SpeechSegment;
pub fn SherpaOnnxDestroySpeechSegment(p: *const SpeechSegment);
pub fn SherpaOnnxVoiceActivityDetectorReset(p: *const VoiceActivityDetector);
pub fn SherpaOnnxVoiceActivityDetectorFlush(p: *const VoiceActivityDetector);
}
+30
View File
@@ -0,0 +1,30 @@
#![allow(non_camel_case_types)]
#![allow(non_snake_case)]
#![allow(non_upper_case_globals)]
use std::os::raw::c_char;
#[repr(C)]
#[derive(Debug, Copy, Clone)]
pub struct SherpaOnnxWave {
/// Samples normalized to [-1, 1]
pub samples: *const f32,
pub sample_rate: i32,
pub num_samples: i32,
}
extern "C" {
/// Read a WAV file. Returns NULL on error.
pub fn SherpaOnnxReadWave(filename: *const c_char) -> *const SherpaOnnxWave;
/// Free memory allocated by SherpaOnnxReadWave
pub fn SherpaOnnxFreeWave(wave: *const SherpaOnnxWave);
/// Write a WAV file. Returns 1 on success, 0 on failure.
pub fn SherpaOnnxWriteWave(
samples: *const f32,
n: i32,
sample_rate: i32,
filename: *const c_char,
) -> i32;
}