mirror of
https://github.com/block/buzz.git
synced 2026-08-18 06:50:31 +02:00
feat(voice): add mobile linking and cancellation
Signed-off-by: John Tennant <johnmatthewtennant@gmail.com>
This commit is contained in:
committed by
John Tennant
parent
388cf0ccdc
commit
b1c899fa6d
Generated
-2
@@ -8446,8 +8446,6 @@ dependencies = [
|
||||
[[package]]
|
||||
name = "sherpa-onnx-sys"
|
||||
version = "1.13.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ffc951af03dc0653c0622158ca8a585a6f2bc43b7b06048cf0e5b5020005c227"
|
||||
dependencies = [
|
||||
"bzip2",
|
||||
"tar",
|
||||
|
||||
+2
-1
@@ -29,7 +29,7 @@ members = [
|
||||
"crates/buzz-voice",
|
||||
"examples/countdown-bot",
|
||||
]
|
||||
exclude = ["desktop/src-tauri"]
|
||||
exclude = ["desktop/src-tauri", "crates/sherpa-onnx-sys"]
|
||||
resolver = "2"
|
||||
|
||||
[workspace.package]
|
||||
@@ -171,3 +171,4 @@ strip = true
|
||||
# allowlist for the auth token). Revert to crates.io once #449 lands upstream.
|
||||
[patch.crates-io]
|
||||
aws-creds = { git = "https://github.com/tlongwell-block/rust-s3", rev = "c9fce3620dd434c1f810101d672cf384268dbb0f" }
|
||||
sherpa-onnx-sys = { path = "crates/sherpa-onnx-sys" }
|
||||
|
||||
@@ -63,6 +63,10 @@ RUN apt-get update \
|
||||
# locations. The normal runtime strips it below; runtime-debug retains it.
|
||||
ENV CARGO_PROFILE_RELEASE_DEBUG=line-tables-only
|
||||
COPY --from=planner /build/recipe.json recipe.json
|
||||
# The sherpa sys patch keeps its published 1.13.4 version so Cargo can replace
|
||||
# the registry crate. It is excluded from the workspace recipe because
|
||||
# cargo-chef masks workspace package versions to 0.0.1.
|
||||
COPY --from=planner /build/crates/sherpa-onnx-sys crates/sherpa-onnx-sys
|
||||
# Cook the full workspace recipe — relay deps include workspace siblings, so
|
||||
# scoping to -p buzz-relay misses transitive deps and re-builds them later.
|
||||
RUN cargo chef cook --release --recipe-path recipe.json
|
||||
|
||||
@@ -15,6 +15,9 @@ RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends build-essential pkg-config libssl-dev ca-certificates \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
COPY --from=planner /build/recipe.json recipe.json
|
||||
# Keep the patched sherpa sys crate at its published version; cargo-chef masks
|
||||
# workspace package versions, so the patch remains outside the workspace recipe.
|
||||
COPY --from=planner /build/crates/sherpa-onnx-sys crates/sherpa-onnx-sys
|
||||
RUN cargo chef cook --release --recipe-path recipe.json
|
||||
COPY . .
|
||||
RUN cargo build --release --locked -p buzz-push-gateway --bin buzz-push-gateway \
|
||||
|
||||
@@ -7,6 +7,11 @@ license.workspace = true
|
||||
repository.workspace = true
|
||||
description = "Reusable local voice primitives for Buzz"
|
||||
|
||||
[features]
|
||||
default = ["static"]
|
||||
static = ["sherpa-onnx/static"]
|
||||
shared = ["sherpa-onnx/shared"]
|
||||
|
||||
[dependencies]
|
||||
ort = { version = "=2.0.0-rc.12", default-features = false, features = ["api-24", "ndarray", "std"] }
|
||||
ort-sys = { version = "=2.0.0-rc.12", features = ["disable-linking"] }
|
||||
@@ -14,5 +19,5 @@ rand = "0.10"
|
||||
sentencepiece-model = "0.1"
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde_json = "1"
|
||||
sherpa-onnx = "1.12"
|
||||
sherpa-onnx = { version = "1.12", default-features = false }
|
||||
tokenizers = { version = "0.22", default-features = false, features = ["fancy-regex"] }
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
//!
|
||||
//! `huddle::models` writes the complete attribution beside the cached bytes.
|
||||
|
||||
use std::cell::RefCell;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::Mutex;
|
||||
|
||||
@@ -39,6 +40,42 @@ pub const VOICE_FILE_EXT: &str = "wav";
|
||||
|
||||
const TTS_NUM_THREADS: usize = 1;
|
||||
|
||||
thread_local! {
|
||||
static ACTIVE_SYNTHESIS_ENGINES: RefCell<Vec<usize>> = const { RefCell::new(Vec::new()) };
|
||||
}
|
||||
|
||||
struct SynthesisCallGuard {
|
||||
engine_id: usize,
|
||||
}
|
||||
|
||||
impl SynthesisCallGuard {
|
||||
fn enter(engine_id: usize) -> Result<Self, String> {
|
||||
ACTIVE_SYNTHESIS_ENGINES.with(|active| {
|
||||
let mut active = active.borrow_mut();
|
||||
if active.contains(&engine_id) {
|
||||
return Err("Pocket TTS callback re-entered the active engine".to_string());
|
||||
}
|
||||
active.push(engine_id);
|
||||
Ok(Self { engine_id })
|
||||
})
|
||||
}
|
||||
|
||||
fn is_active(engine_id: usize) -> bool {
|
||||
ACTIVE_SYNTHESIS_ENGINES.with(|active| active.borrow().contains(&engine_id))
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for SynthesisCallGuard {
|
||||
fn drop(&mut self) {
|
||||
ACTIVE_SYNTHESIS_ENGINES.with(|active| {
|
||||
let mut active = active.borrow_mut();
|
||||
if let Some(index) = active.iter().rposition(|engine| *engine == self.engine_id) {
|
||||
active.remove(index);
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
/// Loaded reference voice samples and their original sample rate.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct VoiceStyle {
|
||||
@@ -90,6 +127,9 @@ impl PocketTts {
|
||||
/// Split text into synthesis units that satisfy the bundle's exact
|
||||
/// 50-token input limit.
|
||||
pub fn split_text_into_chunks(&self, text: &str) -> Result<Vec<String>, String> {
|
||||
if SynthesisCallGuard::is_active(self as *const Self as usize) {
|
||||
return Err("Pocket TTS callback re-entered the active engine".to_string());
|
||||
}
|
||||
let Some(prepared) = prepare_april_prompt(text) else {
|
||||
return Ok(Vec::new());
|
||||
};
|
||||
@@ -104,12 +144,36 @@ impl PocketTts {
|
||||
/// Pocket detects language from text and this model uses one synthesis
|
||||
/// step, so `_lang` and `_steps` intentionally do not affect output.
|
||||
pub fn synth_chunk(
|
||||
&self,
|
||||
text: &str,
|
||||
lang: &str,
|
||||
style: &VoiceStyle,
|
||||
steps: usize,
|
||||
) -> Result<Vec<f32>, String> {
|
||||
self.synth_chunk_with_callback(text, lang, style, steps, None::<fn(&[f32], f32) -> bool>)
|
||||
}
|
||||
|
||||
/// Synthesize text, allowing the caller to stop generation early.
|
||||
///
|
||||
/// The callback receives PCM accumulated after each decoded text chunk
|
||||
/// and progress in `[0, 1]`. During latent generation the callback is
|
||||
/// invoked with an empty sample slice so cancellation can remain
|
||||
/// responsive before PCM is available. Return `true` to continue or
|
||||
/// `false` to stop and return the audio generated so far. Progress is
|
||||
/// monotonic across split text chunks. Calls back into the same
|
||||
/// [`PocketTts`] return an error instead of blocking on its engine lock.
|
||||
pub fn synth_chunk_with_callback<F>(
|
||||
&self,
|
||||
text: &str,
|
||||
_lang: &str,
|
||||
style: &VoiceStyle,
|
||||
_steps: usize,
|
||||
) -> Result<Vec<f32>, String> {
|
||||
mut callback: Option<F>,
|
||||
) -> Result<Vec<f32>, String>
|
||||
where
|
||||
F: FnMut(&[f32], f32) -> bool + 'static,
|
||||
{
|
||||
let _call_guard = SynthesisCallGuard::enter(self as *const Self as usize)?;
|
||||
let Some(prepared) = prepare_april_prompt(text) else {
|
||||
return Ok(Vec::new());
|
||||
};
|
||||
@@ -119,15 +183,47 @@ impl PocketTts {
|
||||
.map_err(|_| "Pocket TTS engine lock poisoned".to_string())?;
|
||||
let chunks = engine.split_prompt(&prepared)?;
|
||||
let mut samples = Vec::new();
|
||||
for chunk in chunks {
|
||||
let chunk_count = chunks.len();
|
||||
for (index, chunk) in chunks.into_iter().enumerate() {
|
||||
let prepared = prepare_april_prompt(&chunk)
|
||||
.ok_or_else(|| "Pocket TTS prompt chunk became empty".to_string())?;
|
||||
samples.extend(engine.synth_chunk(&prepared, style)?);
|
||||
let progress_offset = index as f32 / chunk_count as f32;
|
||||
let progress_scale = 1.0 / chunk_count as f32;
|
||||
let (chunk_samples, cancelled) = engine.synth_chunk_with_callback(
|
||||
&prepared,
|
||||
style,
|
||||
&mut callback,
|
||||
progress_offset,
|
||||
progress_scale,
|
||||
)?;
|
||||
samples.extend(chunk_samples);
|
||||
if cancelled {
|
||||
break;
|
||||
}
|
||||
let progress = (index + 1) as f32 / chunk_count as f32;
|
||||
if !callback_allows_progress(&mut callback, &samples, progress)? {
|
||||
break;
|
||||
}
|
||||
}
|
||||
Ok(samples)
|
||||
}
|
||||
}
|
||||
|
||||
fn callback_allows_progress<F>(
|
||||
callback: &mut Option<F>,
|
||||
samples: &[f32],
|
||||
progress: f32,
|
||||
) -> Result<bool, String>
|
||||
where
|
||||
F: FnMut(&[f32], f32) -> bool,
|
||||
{
|
||||
let Some(callback) = callback.as_mut() else {
|
||||
return Ok(true);
|
||||
};
|
||||
std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| callback(samples, progress)))
|
||||
.map_err(|_| "Pocket TTS synthesis callback panicked".to_string())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -147,6 +243,34 @@ mod tests {
|
||||
.any(|artifact| artifact.filename == "flow_lm_main.onnx"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn callback_can_cancel_before_pcm_is_available() {
|
||||
let mut callback = Some(|samples: &[f32], progress: f32| {
|
||||
assert!(samples.is_empty());
|
||||
assert_eq!(progress, 0.25);
|
||||
false
|
||||
});
|
||||
assert!(!callback_allows_progress(&mut callback, &[], 0.25).expect("callback"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn callback_panic_is_reported_without_unwinding() {
|
||||
let mut callback = Some(|_: &[f32], _: f32| -> bool {
|
||||
panic!("callback failure");
|
||||
});
|
||||
assert_eq!(
|
||||
callback_allows_progress(&mut callback, &[], 0.0).unwrap_err(),
|
||||
"Pocket TTS synthesis callback panicked"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn active_engine_reentry_is_rejected() {
|
||||
let _guard = SynthesisCallGuard::enter(42).expect("first call");
|
||||
assert!(SynthesisCallGuard::enter(42).is_err());
|
||||
assert!(SynthesisCallGuard::is_active(42));
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[ignore = "requires BUZZ_POCKET_TEST_MODEL_DIR"]
|
||||
fn production_api_emits_non_silent_april_int8_pcm() {
|
||||
|
||||
@@ -304,11 +304,17 @@ impl AprilPocketTts {
|
||||
.collect()
|
||||
}
|
||||
|
||||
pub(crate) fn synth_chunk(
|
||||
pub(crate) fn synth_chunk_with_callback<F>(
|
||||
&mut self,
|
||||
prepared: &AprilPreparedPrompt,
|
||||
style: &VoiceStyle,
|
||||
) -> Result<Vec<f32>, String> {
|
||||
callback: &mut Option<F>,
|
||||
progress_offset: f32,
|
||||
progress_scale: f32,
|
||||
) -> Result<(Vec<f32>, bool), String>
|
||||
where
|
||||
F: FnMut(&[f32], f32) -> bool,
|
||||
{
|
||||
let voice_embeddings = self.voice_embeddings(style)?;
|
||||
let mut flow_state = self.condition_voice(&voice_embeddings)?;
|
||||
let token_ids = self
|
||||
@@ -321,7 +327,7 @@ impl AprilPocketTts {
|
||||
.map(i64::from)
|
||||
.collect::<Vec<_>>();
|
||||
if token_ids.is_empty() {
|
||||
return Ok(Vec::new());
|
||||
return Ok((Vec::new(), false));
|
||||
}
|
||||
if token_ids.len() > self.bundle.max_token_per_chunk {
|
||||
return Err(format!(
|
||||
@@ -335,9 +341,15 @@ impl AprilPocketTts {
|
||||
let text_embeddings = self.text_embeddings(token_ids)?;
|
||||
self.run_flow_main_prefix(&text_embeddings, &mut flow_state)?;
|
||||
let max_frames = estimate_max_frames(token_count, self.bundle.frame_rate);
|
||||
let latents =
|
||||
self.generate_latents(max_frames, prepared.frames_after_eos, &mut flow_state)?;
|
||||
self.decode_latents(&latents)
|
||||
let (latents, cancelled) = self.generate_latents(
|
||||
max_frames,
|
||||
prepared.frames_after_eos,
|
||||
&mut flow_state,
|
||||
callback,
|
||||
progress_offset,
|
||||
progress_scale,
|
||||
)?;
|
||||
Ok((self.decode_latents(&latents)?, cancelled))
|
||||
}
|
||||
|
||||
fn prepared_token_count(&self, text: &str) -> Result<usize, String> {
|
||||
@@ -497,18 +509,29 @@ impl AprilPocketTts {
|
||||
replace_state_from_outputs(state, &mut outputs)
|
||||
}
|
||||
|
||||
fn generate_latents(
|
||||
fn generate_latents<F>(
|
||||
&mut self,
|
||||
max_frames: usize,
|
||||
frames_after_eos: usize,
|
||||
state: &mut [StateValue],
|
||||
) -> Result<Vec<f32>, String> {
|
||||
callback: &mut Option<F>,
|
||||
progress_offset: f32,
|
||||
progress_scale: f32,
|
||||
) -> Result<(Vec<f32>, bool), String>
|
||||
where
|
||||
F: FnMut(&[f32], f32) -> bool,
|
||||
{
|
||||
let mut current = vec![f32::NAN; self.bundle.latent_dim];
|
||||
let mut latents = Vec::with_capacity(max_frames * self.bundle.latent_dim);
|
||||
let mut eos_step = None;
|
||||
let mut rng = rand::rng();
|
||||
|
||||
for step in 0..max_frames {
|
||||
let local_progress = step as f32 / max_frames as f32;
|
||||
let progress = progress_offset + local_progress * progress_scale;
|
||||
if !super::callback_allows_progress(callback, &[], progress)? {
|
||||
return Ok((latents, true));
|
||||
}
|
||||
let sequence = Tensor::from_array((
|
||||
vec![1_i64, 1, self.bundle.latent_dim as i64],
|
||||
current.clone().into_boxed_slice(),
|
||||
@@ -594,7 +617,7 @@ impl AprilPocketTts {
|
||||
current.clone_from(&noise);
|
||||
latents.extend_from_slice(&noise);
|
||||
}
|
||||
Ok(latents)
|
||||
Ok((latents, false))
|
||||
}
|
||||
|
||||
fn decode_latents(&mut self, latents: &[f32]) -> Result<Vec<f32>, String> {
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
# THIS FILE IS AUTOMATICALLY GENERATED BY CARGO
|
||||
#
|
||||
# When uploading crates to the registry Cargo will automatically
|
||||
# "normalize" Cargo.toml files for maximal compatibility
|
||||
# with all versions of Cargo and also rewrite `path` dependencies
|
||||
# to registry (e.g., crates.io) dependencies.
|
||||
#
|
||||
# If you are reading this file be aware that the original Cargo.toml
|
||||
# will likely look very different (and much more reasonable).
|
||||
# See Cargo.toml.orig for the original contents.
|
||||
|
||||
[package]
|
||||
edition = "2021"
|
||||
name = "sherpa-onnx-sys"
|
||||
version = "1.13.4"
|
||||
build = "build.rs"
|
||||
links = "sherpa-onnx"
|
||||
include = [
|
||||
"src/**",
|
||||
"build.rs",
|
||||
"Cargo.toml",
|
||||
"README.md",
|
||||
"LICENSE*",
|
||||
]
|
||||
autolib = false
|
||||
autobins = false
|
||||
autoexamples = false
|
||||
autotests = false
|
||||
autobenches = false
|
||||
description = "Raw FFI bindings to the sherpa-onnx C API"
|
||||
readme = "README.md"
|
||||
keywords = [
|
||||
"ffi",
|
||||
"speech",
|
||||
"sherpa-onnx",
|
||||
"bindings",
|
||||
]
|
||||
categories = ["external-ffi-bindings"]
|
||||
license = "Apache-2.0"
|
||||
repository = "https://github.com/k2-fsa/sherpa-onnx"
|
||||
|
||||
[package.metadata.docs.rs]
|
||||
default-features = false
|
||||
features = []
|
||||
|
||||
[features]
|
||||
default = ["static"]
|
||||
shared = []
|
||||
static = []
|
||||
|
||||
[lib]
|
||||
name = "sherpa_onnx_sys"
|
||||
path = "src/lib.rs"
|
||||
|
||||
[build-dependencies.bzip2]
|
||||
version = "0.4"
|
||||
|
||||
[build-dependencies.tar]
|
||||
version = "0.4"
|
||||
|
||||
[build-dependencies.ureq]
|
||||
version = "2.12"
|
||||
features = ["proxy-from-env"]
|
||||
Generated
+34
@@ -0,0 +1,34 @@
|
||||
[package]
|
||||
name = "sherpa-onnx-sys"
|
||||
version = "1.13.4"
|
||||
edition = "2021"
|
||||
description = "Raw FFI bindings to the sherpa-onnx C API"
|
||||
license = "Apache-2.0"
|
||||
repository = "https://github.com/k2-fsa/sherpa-onnx"
|
||||
readme = "README.md"
|
||||
links = "sherpa-onnx"
|
||||
|
||||
keywords = ["ffi", "speech", "sherpa-onnx", "bindings"]
|
||||
categories = ["external-ffi-bindings"]
|
||||
|
||||
include = [
|
||||
"src/**",
|
||||
"build.rs",
|
||||
"Cargo.toml",
|
||||
"README.md",
|
||||
"LICENSE*",
|
||||
]
|
||||
|
||||
[features]
|
||||
default = ["static"]
|
||||
static = []
|
||||
shared = []
|
||||
|
||||
[build-dependencies]
|
||||
bzip2 = "0.4"
|
||||
tar = "0.4"
|
||||
ureq = { version = "2.12", features = ["proxy-from-env"] }
|
||||
|
||||
[package.metadata.docs.rs]
|
||||
default-features = false
|
||||
features = []
|
||||
@@ -0,0 +1,202 @@
|
||||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
@@ -0,0 +1,671 @@
|
||||
<div align="center">
|
||||
|
||||
[](https://deepwiki.com/k2-fsa/sherpa-onnx)
|
||||
|
||||
</div>
|
||||
|
||||
Buzz carries the crates.io 1.13.4 FFI sources and locally patches the build
|
||||
script to link caller-supplied official iOS and Android libraries through
|
||||
`SHERPA_ONNX_LIB_DIR`. Desktop archive selection remains unchanged.
|
||||
|
||||
### Supported functions
|
||||
|
||||
|Speech recognition| [Speech synthesis][tts-url] | [Source separation][ss-url] |
|
||||
|------------------|------------------|-------------------|
|
||||
| ✔️ | ✔️ | ✔️ |
|
||||
|
||||
|Speaker identification| [Speaker diarization][sd-url] | Speaker verification |
|
||||
|----------------------|-------------------- |------------------------|
|
||||
| ✔️ | ✔️ | ✔️ |
|
||||
|
||||
| [Spoken Language identification][slid-url] | [Audio tagging][at-url] | [Voice activity detection][vad-url] |
|
||||
|--------------------------------|---------------|--------------------------|
|
||||
| ✔️ | ✔️ | ✔️ |
|
||||
|
||||
| [Keyword spotting][kws-url] | [Add punctuation][punct-url] | [Speech enhancement][se-url] |
|
||||
|------------------|-----------------|--------------------|
|
||||
| ✔️ | ✔️ | ✔️ |
|
||||
|
||||
|
||||
### Supported platforms
|
||||
|
||||
|Architecture| Android | iOS | Windows | macOS | linux | HarmonyOS |
|
||||
|------------|---------|---------|------------|-------|-------|-----------|
|
||||
| x64 | ✔️ | | ✔️ | ✔️ | ✔️ | ✔️ |
|
||||
| x86 | ✔️ | | ✔️ | | | |
|
||||
| arm64 | ✔️ | ✔️ | ✔️ | ✔️ | ✔️ | ✔️ |
|
||||
| arm32 | ✔️ | | | | ✔️ | ✔️ |
|
||||
| riscv64 | | | | | ✔️ | |
|
||||
|
||||
### Supported programming languages
|
||||
|
||||
| 1. C++ | 2. C | 3. Python | 4. JavaScript |
|
||||
|--------|-------|-----------|---------------|
|
||||
| ✔️ | ✔️ | ✔️ | ✔️ |
|
||||
|
||||
|5. Java | 6. C# | 7. Kotlin | 8. Swift |
|
||||
|--------|-------|-----------|----------|
|
||||
| ✔️ | ✔️ | ✔️ | ✔️ |
|
||||
|
||||
| 9. Go | 10. Dart | 11. Rust | 12. Pascal |
|
||||
|-------|----------|----------|------------|
|
||||
| ✔️ | ✔️ | ✔️ | ✔️ |
|
||||
|
||||
|
||||
It also supports WebAssembly.
|
||||
|
||||
### Supported NPUs
|
||||
|
||||
| [1. Rockchip NPU (RKNN)][rknpu-doc] | [2. Qualcomm NPU (QNN)][qnn-doc] | [3. Ascend NPU][ascend-doc] |
|
||||
|-------------------------------------|-----------------------------------|-----------------------------|
|
||||
| ✔️ | ✔️ | ✔️ |
|
||||
|
||||
| [4. Axera NPU][axera-npu] |
|
||||
|---------------------------|
|
||||
| ✔️ |
|
||||
|
||||
[Join our discord](https://discord.gg/fJdxzg2VbG)
|
||||
|
||||
|
||||
## Introduction
|
||||
|
||||
This repository supports running the following functions **locally**
|
||||
|
||||
- Speech-to-text (i.e., ASR); both streaming and non-streaming are supported
|
||||
- Text-to-speech (i.e., TTS)
|
||||
- Speaker diarization
|
||||
- Speaker identification
|
||||
- Speaker verification
|
||||
- Spoken language identification
|
||||
- Audio tagging
|
||||
- VAD (e.g., [silero-vad][silero-vad])
|
||||
- Speech enhancement (e.g., [gtcrn][gtcrn], [DPDFNet](https://github.com/ceva-ip/DPDFNet))
|
||||
- Keyword spotting
|
||||
- Source separation (e.g., [spleeter][spleeter], [UVR][UVR])
|
||||
|
||||
on the following platforms and operating systems:
|
||||
|
||||
- x86, ``x86_64``, 32-bit ARM, 64-bit ARM (arm64, aarch64), RISC-V (riscv64), **RK NPU**, **Ascend NPU**
|
||||
- Linux, macOS, Windows, openKylin
|
||||
- Android, WearOS
|
||||
- iOS
|
||||
- HarmonyOS
|
||||
- NodeJS
|
||||
- WebAssembly
|
||||
- [NVIDIA Jetson Orin NX][NVIDIA Jetson Orin NX] (Support running on both CPU and GPU)
|
||||
- [NVIDIA Jetson Nano B01][NVIDIA Jetson Nano B01] (Support running on both CPU and GPU)
|
||||
- [Raspberry Pi][Raspberry Pi]
|
||||
- [RV1126][RV1126]
|
||||
- [LicheePi4A][LicheePi4A]
|
||||
- [VisionFive 2][VisionFive 2]
|
||||
- [旭日X3派][旭日X3派]
|
||||
- [爱芯派][爱芯派]
|
||||
- [RK3588][RK3588]
|
||||
- [SpacemiT-K1][SpacemiT-K1]
|
||||
- [SpacemiT-K3][SpacemiT-K3]
|
||||
- etc
|
||||
|
||||
with the following APIs
|
||||
|
||||
- C++, C, Python, Go, ``C#``
|
||||
- Java, Kotlin, JavaScript
|
||||
- Swift, Rust
|
||||
- Dart, Object Pascal
|
||||
|
||||
### Links for Huggingface Spaces
|
||||
|
||||
<details>
|
||||
<summary>You can visit the following Huggingface spaces to try sherpa-onnx without
|
||||
installing anything. All you need is a browser.</summary>
|
||||
|
||||
| Description | URL | 中国镜像 |
|
||||
|-------------------------------------------------------|-----------------------------------------|----------------------------------------|
|
||||
| Speaker diarization | [Click me][hf-space-speaker-diarization]| [镜像][hf-space-speaker-diarization-cn]|
|
||||
| Speech recognition | [Click me][hf-space-asr] | [镜像][hf-space-asr-cn] |
|
||||
| Speech recognition with [Whisper][Whisper] | [Click me][hf-space-asr-whisper] | [镜像][hf-space-asr-whisper-cn] |
|
||||
| Speech synthesis | [Click me][hf-space-tts] | [镜像][hf-space-tts-cn] |
|
||||
| Generate subtitles | [Click me][hf-space-subtitle] | [镜像][hf-space-subtitle-cn] |
|
||||
| Audio tagging | [Click me][hf-space-audio-tagging] | [镜像][hf-space-audio-tagging-cn] |
|
||||
| Source separation | [Click me][hf-space-source-separation] | [镜像][hf-space-source-separation-cn] |
|
||||
| Spoken language identification with [Whisper][Whisper]| [Click me][hf-space-slid-whisper] | [镜像][hf-space-slid-whisper-cn] |
|
||||
|
||||
We also have spaces built using WebAssembly. They are listed below:
|
||||
|
||||
| Description | Huggingface space| ModelScope space|
|
||||
|------------------------------------------------------------------------------------------|------------------|-----------------|
|
||||
|Voice activity detection with [silero-vad][silero-vad] | [Click me][wasm-hf-vad]|[地址][wasm-ms-vad]|
|
||||
|Real-time speech recognition (Chinese + English) with Zipformer | [Click me][wasm-hf-streaming-asr-zh-en-zipformer]|[地址][wasm-hf-streaming-asr-zh-en-zipformer]|
|
||||
|Real-time speech recognition (Chinese + English) with Paraformer |[Click me][wasm-hf-streaming-asr-zh-en-paraformer]| [地址][wasm-ms-streaming-asr-zh-en-paraformer]|
|
||||
|Real-time speech recognition (Chinese + English + Cantonese) with [Paraformer-large][Paraformer-large]|[Click me][wasm-hf-streaming-asr-zh-en-yue-paraformer]| [地址][wasm-ms-streaming-asr-zh-en-yue-paraformer]|
|
||||
|Real-time speech recognition (English) |[Click me][wasm-hf-streaming-asr-en-zipformer] |[地址][wasm-ms-streaming-asr-en-zipformer]|
|
||||
|VAD + speech recognition (Chinese) with [Zipformer CTC](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-ctc/icefall/zipformer.html#sherpa-onnx-zipformer-ctc-zh-int8-2025-07-03-chinese)|[Click me][wasm-hf-vad-asr-zh-zipformer-ctc-07-03]| [地址][wasm-ms-vad-asr-zh-zipformer-ctc-07-03]|
|
||||
|VAD + speech recognition (Chinese + English + Korean + Japanese + Cantonese) with [SenseVoice][SenseVoice]|[Click me][wasm-hf-vad-asr-zh-en-ko-ja-yue-sense-voice]| [地址][wasm-ms-vad-asr-zh-en-ko-ja-yue-sense-voice]|
|
||||
|VAD + speech recognition (English) with [Whisper][Whisper] tiny.en|[Click me][wasm-hf-vad-asr-en-whisper-tiny-en]| [地址][wasm-ms-vad-asr-en-whisper-tiny-en]|
|
||||
|VAD + speech recognition (English) with [Moonshine tiny][Moonshine tiny]|[Click me][wasm-hf-vad-asr-en-moonshine-tiny-en]| [地址][wasm-ms-vad-asr-en-moonshine-tiny-en]|
|
||||
|VAD + speech recognition (English) with Zipformer trained with [GigaSpeech][GigaSpeech] |[Click me][wasm-hf-vad-asr-en-zipformer-gigaspeech]| [地址][wasm-ms-vad-asr-en-zipformer-gigaspeech]|
|
||||
|VAD + speech recognition (Chinese) with Zipformer trained with [WenetSpeech][WenetSpeech] |[Click me][wasm-hf-vad-asr-zh-zipformer-wenetspeech]| [地址][wasm-ms-vad-asr-zh-zipformer-wenetspeech]|
|
||||
|VAD + speech recognition (Japanese) with Zipformer trained with [ReazonSpeech][ReazonSpeech]|[Click me][wasm-hf-vad-asr-ja-zipformer-reazonspeech]| [地址][wasm-ms-vad-asr-ja-zipformer-reazonspeech]|
|
||||
|VAD + speech recognition (Thai) with Zipformer trained with [GigaSpeech2][GigaSpeech2] |[Click me][wasm-hf-vad-asr-th-zipformer-gigaspeech2]| [地址][wasm-ms-vad-asr-th-zipformer-gigaspeech2]|
|
||||
|VAD + speech recognition (Chinese 多种方言) with a [TeleSpeech-ASR][TeleSpeech-ASR] CTC model|[Click me][wasm-hf-vad-asr-zh-telespeech]| [地址][wasm-ms-vad-asr-zh-telespeech]|
|
||||
|VAD + speech recognition (English + Chinese, 及多种中文方言) with Paraformer-large |[Click me][wasm-hf-vad-asr-zh-en-paraformer-large]| [地址][wasm-ms-vad-asr-zh-en-paraformer-large]|
|
||||
|VAD + speech recognition (English + Chinese, 及多种中文方言) with Paraformer-small |[Click me][wasm-hf-vad-asr-zh-en-paraformer-small]| [地址][wasm-ms-vad-asr-zh-en-paraformer-small]|
|
||||
|VAD + speech recognition (多语种及多种中文方言) with [Dolphin][Dolphin]-base |[Click me][wasm-hf-vad-asr-multi-lang-dolphin-base]| [地址][wasm-ms-vad-asr-multi-lang-dolphin-base]|
|
||||
|Speech synthesis (Piper, English) |[Click me][wasm-hf-tts-piper-en]| [地址][wasm-ms-tts-piper-en]|
|
||||
|Speech synthesis (Piper, German) |[Click me][wasm-hf-tts-piper-de]| [地址][wasm-ms-tts-piper-de]|
|
||||
|Speech synthesis (Matcha, Chinese) |[Click me][wasm-hf-tts-matcha-zh]| [地址][wasm-ms-tts-matcha-zh]|
|
||||
|Speech synthesis (Matcha, English) |[Click me][wasm-hf-tts-matcha-en]| [地址][wasm-ms-tts-matcha-en]|
|
||||
|Speech synthesis (Matcha, Chinese+English) |[Click me][wasm-hf-tts-matcha-zh-en]| [地址][wasm-ms-tts-matcha-zh-en]|
|
||||
|Speaker diarization |[Click me][wasm-hf-speaker-diarization]|[地址][wasm-ms-speaker-diarization]|
|
||||
|Voice cloning with ZipVoice (Chinese+English) |[Click me][wasm-hf-voice-cloning-zipvoice]|[地址][wasm-ms-voice-cloning-zipvoice]|
|
||||
|Voice cloning with Pocket TTS (English) |[Click me][wasm-hf-voice-cloning-pocket]|[地址][wasm-ms-voice-cloning-pocket]|
|
||||
|
||||
</details>
|
||||
|
||||
### Links for pre-built Android APKs
|
||||
|
||||
<details>
|
||||
|
||||
<summary>You can find pre-built Android APKs for this repository in the following table</summary>
|
||||
|
||||
| Description | URL | 中国用户 |
|
||||
|----------------------------------------|------------------------------------|-----------------------------------|
|
||||
| Speaker diarization | [Address][apk-speaker-diarization] | [点此][apk-speaker-diarization-cn]|
|
||||
| Streaming speech recognition | [Address][apk-streaming-asr] | [点此][apk-streaming-asr-cn] |
|
||||
| Simulated-streaming speech recognition | [Address][apk-simula-streaming-asr]| [点此][apk-simula-streaming-asr-cn]|
|
||||
| Text-to-speech | [Address][apk-tts] | [点此][apk-tts-cn] |
|
||||
| Voice activity detection (VAD) | [Address][apk-vad] | [点此][apk-vad-cn] |
|
||||
| VAD + non-streaming speech recognition | [Address][apk-vad-asr] | [点此][apk-vad-asr-cn] |
|
||||
| Two-pass speech recognition | [Address][apk-2pass] | [点此][apk-2pass-cn] |
|
||||
| Audio tagging | [Address][apk-at] | [点此][apk-at-cn] |
|
||||
| Audio tagging (WearOS) | [Address][apk-at-wearos] | [点此][apk-at-wearos-cn] |
|
||||
| Speaker identification | [Address][apk-sid] | [点此][apk-sid-cn] |
|
||||
| Spoken language identification | [Address][apk-slid] | [点此][apk-slid-cn] |
|
||||
| Keyword spotting | [Address][apk-kws] | [点此][apk-kws-cn] |
|
||||
|
||||
</details>
|
||||
|
||||
### Links for pre-built Flutter APPs
|
||||
|
||||
<details>
|
||||
|
||||
#### Real-time speech recognition
|
||||
|
||||
| Description | URL | 中国用户 |
|
||||
|--------------------------------|-------------------------------------|-------------------------------------|
|
||||
| Streaming speech recognition | [Address][apk-flutter-streaming-asr]| [点此][apk-flutter-streaming-asr-cn]|
|
||||
|
||||
#### Text-to-speech
|
||||
|
||||
| Description | URL | 中国用户 |
|
||||
|------------------------------------------|------------------------------------|------------------------------------|
|
||||
| Android (arm64-v8a, armeabi-v7a, x86_64) | [Address][flutter-tts-android] | [点此][flutter-tts-android-cn] |
|
||||
| Linux (x64) | [Address][flutter-tts-linux] | [点此][flutter-tts-linux-cn] |
|
||||
| macOS (x64) | [Address][flutter-tts-macos-x64] | [点此][flutter-tts-macos-x64-cn] |
|
||||
| macOS (arm64) | [Address][flutter-tts-macos-arm64] | [点此][flutter-tts-macos-arm64-cn] |
|
||||
| Windows (x64) | [Address][flutter-tts-win-x64] | [点此][flutter-tts-win-x64-cn] |
|
||||
|
||||
> Note: You need to build from source for iOS.
|
||||
|
||||
</details>
|
||||
|
||||
### Links for pre-built Lazarus APPs
|
||||
|
||||
<details>
|
||||
|
||||
#### Generating subtitles
|
||||
|
||||
| Description | URL | 中国用户 |
|
||||
|--------------------------------|----------------------------|----------------------------|
|
||||
| Generate subtitles (生成字幕) | [Address][lazarus-subtitle]| [点此][lazarus-subtitle-cn]|
|
||||
|
||||
</details>
|
||||
|
||||
### Links for pre-trained models
|
||||
|
||||
<details>
|
||||
|
||||
| Description | URL |
|
||||
|---------------------------------------------|---------------------------------------------------------------------------------------|
|
||||
| Speech recognition (speech to text, ASR) | [Address][asr-models] |
|
||||
| Text-to-speech (TTS) | [Address][tts-models] |
|
||||
| VAD | [Address][vad-models] |
|
||||
| Keyword spotting | [Address][kws-models] |
|
||||
| Audio tagging | [Address][at-models] |
|
||||
| Speaker identification (Speaker ID) | [Address][sid-models] |
|
||||
| Spoken language identification (Language ID)| See multi-lingual [Whisper][Whisper] ASR models from [Speech recognition][asr-models]|
|
||||
| Punctuation | [Address][punct-models] |
|
||||
| Speaker segmentation | [Address][speaker-segmentation-models] |
|
||||
| Speech enhancement | [Address][speech-enhancement-models] |
|
||||
| Source separation | [Address][source-separation-models] |
|
||||
|
||||
</details>
|
||||
|
||||
#### Some pre-trained ASR models (Streaming)
|
||||
|
||||
<details>
|
||||
|
||||
Please see
|
||||
|
||||
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/index.html>
|
||||
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-paraformer/index.html>
|
||||
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-ctc/index.html>
|
||||
|
||||
for more models. The following table lists only **SOME** of them.
|
||||
|
||||
|
||||
|Name | Supported Languages| Description|
|
||||
|-----|-----|----|
|
||||
|[sherpa-onnx-streaming-zipformer-bilingual-zh-en-2023-02-20][sherpa-onnx-streaming-zipformer-bilingual-zh-en-2023-02-20]| Chinese, English| See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/zipformer-transducer-models.html#csukuangfj-sherpa-onnx-streaming-zipformer-bilingual-zh-en-2023-02-20-bilingual-chinese-english)|
|
||||
|[sherpa-onnx-streaming-zipformer-small-bilingual-zh-en-2023-02-16][sherpa-onnx-streaming-zipformer-small-bilingual-zh-en-2023-02-16]| Chinese, English| See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/zipformer-transducer-models.html#sherpa-onnx-streaming-zipformer-small-bilingual-zh-en-2023-02-16-bilingual-chinese-english)|
|
||||
|[sherpa-onnx-streaming-zipformer-zh-14M-2023-02-23][sherpa-onnx-streaming-zipformer-zh-14M-2023-02-23]|Chinese| Suitable for Cortex A7 CPU. See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/zipformer-transducer-models.html#sherpa-onnx-streaming-zipformer-zh-14m-2023-02-23)|
|
||||
|[sherpa-onnx-streaming-zipformer-en-20M-2023-02-17][sherpa-onnx-streaming-zipformer-en-20M-2023-02-17]|English|Suitable for Cortex A7 CPU. See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/zipformer-transducer-models.html#sherpa-onnx-streaming-zipformer-en-20m-2023-02-17)|
|
||||
|[sherpa-onnx-streaming-zipformer-korean-2024-06-16][sherpa-onnx-streaming-zipformer-korean-2024-06-16]|Korean| See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/zipformer-transducer-models.html#sherpa-onnx-streaming-zipformer-korean-2024-06-16-korean)|
|
||||
|[sherpa-onnx-streaming-zipformer-fr-2023-04-14][sherpa-onnx-streaming-zipformer-fr-2023-04-14]|French| See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/online-transducer/zipformer-transducer-models.html#shaojieli-sherpa-onnx-streaming-zipformer-fr-2023-04-14-french)|
|
||||
|
||||
</details>
|
||||
|
||||
|
||||
#### Some pre-trained ASR models (Non-Streaming)
|
||||
|
||||
<details>
|
||||
|
||||
Please see
|
||||
|
||||
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/index.html>
|
||||
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-paraformer/index.html>
|
||||
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-ctc/index.html>
|
||||
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/telespeech/index.html>
|
||||
- <https://k2-fsa.github.io/sherpa/onnx/pretrained_models/whisper/index.html>
|
||||
|
||||
for more models. The following table lists only **SOME** of them.
|
||||
|
||||
|Name | Supported Languages| Description|
|
||||
|-----|-----|----|
|
||||
|[sherpa-onnx-nemo-parakeet-tdt-0.6b-v2-int8](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/nemo-transducer-models.html#sherpa-onnx-nemo-parakeet-tdt-0-6b-v2-int8-english)| English | It is converted from <https://huggingface.co/nvidia/parakeet-tdt-0.6b-v2>|
|
||||
|[Whisper tiny.en](https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-whisper-tiny.en.tar.bz2)|English| See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/whisper/tiny.en.html)|
|
||||
|[Moonshine tiny][Moonshine tiny]|English|See [also](https://github.com/usefulsensors/moonshine)|
|
||||
|[sherpa-onnx-zipformer-ctc-zh-int8-2025-07-03](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-ctc/icefall/zipformer.html#sherpa-onnx-zipformer-ctc-zh-int8-2025-07-03-chinese)|Chinese| A Zipformer CTC model|
|
||||
|[sherpa-onnx-sense-voice-zh-en-ja-ko-yue-2024-07-17][sherpa-onnx-sense-voice-zh-en-ja-ko-yue-2024-07-17]|Chinese, Cantonese, English, Korean, Japanese| 支持多种中文方言. See [also](https://k2-fsa.github.io/sherpa/onnx/sense-voice/index.html)|
|
||||
|[sherpa-onnx-paraformer-zh-2024-03-09][sherpa-onnx-paraformer-zh-2024-03-09]|Chinese, English| 也支持多种中文方言. See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-paraformer/paraformer-models.html#csukuangfj-sherpa-onnx-paraformer-zh-2024-03-09-chinese-english)|
|
||||
|[sherpa-onnx-zipformer-ja-reazonspeech-2024-08-01][sherpa-onnx-zipformer-ja-reazonspeech-2024-08-01]|Japanese|See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/zipformer-transducer-models.html#sherpa-onnx-zipformer-ja-reazonspeech-2024-08-01-japanese)|
|
||||
|[sherpa-onnx-nemo-transducer-giga-am-russian-2024-10-24][sherpa-onnx-nemo-transducer-giga-am-russian-2024-10-24]|Russian|See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/nemo-transducer-models.html#sherpa-onnx-nemo-transducer-giga-am-russian-2024-10-24-russian)|
|
||||
|[sherpa-onnx-nemo-ctc-giga-am-russian-2024-10-24][sherpa-onnx-nemo-ctc-giga-am-russian-2024-10-24]|Russian| See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-ctc/nemo/russian.html#sherpa-onnx-nemo-ctc-giga-am-russian-2024-10-24)|
|
||||
|[sherpa-onnx-zipformer-ru-2024-09-18][sherpa-onnx-zipformer-ru-2024-09-18]|Russian|See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/zipformer-transducer-models.html#sherpa-onnx-zipformer-ru-2024-09-18-russian)|
|
||||
|[sherpa-onnx-zipformer-korean-2024-06-24][sherpa-onnx-zipformer-korean-2024-06-24]|Korean|See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/zipformer-transducer-models.html#sherpa-onnx-zipformer-korean-2024-06-24-korean)|
|
||||
|[sherpa-onnx-zipformer-thai-2024-06-20][sherpa-onnx-zipformer-thai-2024-06-20]|Thai| See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/offline-transducer/zipformer-transducer-models.html#sherpa-onnx-zipformer-thai-2024-06-20-thai)|
|
||||
|[sherpa-onnx-telespeech-ctc-int8-zh-2024-06-04][sherpa-onnx-telespeech-ctc-int8-zh-2024-06-04]|Chinese| 支持多种方言. See [also](https://k2-fsa.github.io/sherpa/onnx/pretrained_models/telespeech/models.html#sherpa-onnx-telespeech-ctc-int8-zh-2024-06-04)|
|
||||
|
||||
</details>
|
||||
|
||||
### Useful links
|
||||
|
||||
- Documentation: https://k2-fsa.github.io/sherpa/onnx/
|
||||
- Bilibili 演示视频: https://search.bilibili.com/all?keyword=%E6%96%B0%E4%B8%80%E4%BB%A3Kaldi
|
||||
|
||||
### How to reach us
|
||||
|
||||
Please see
|
||||
https://k2-fsa.github.io/sherpa/social-groups.html
|
||||
for 新一代 Kaldi **微信交流群** and **QQ 交流群**.
|
||||
|
||||
## Projects using sherpa-onnx
|
||||
|
||||
### [Sherpa Voice / @siteed/sherpa-onnx.rn](https://github.com/deeeed/audiolab)
|
||||
|
||||
> React Native wrapper and demo app for validating sherpa-onnx on iOS,
|
||||
> Android, and Web, including ASR, TTS, VAD, KWS, speaker ID, diarization,
|
||||
> language ID, punctuation, audio tagging, and speech enhancement.
|
||||
|
||||
- [NPM package](https://www.npmjs.com/package/@siteed/sherpa-onnx.rn)
|
||||
- [Live demo](https://deeeed.github.io/audiolab/sherpa-voice/)
|
||||
|
||||
### [Speed of Sound](https://github.com/zugaldia/speedofsound)
|
||||
|
||||
> A voice-typing application for the Linux desktop (GTK4/Adwaita).
|
||||
> It captures microphone audio, transcribes it offline using Sherpa ONNX ASR models,
|
||||
> optionally polishes the text with an LLM, and types the result into the active window
|
||||
> via XDG Remote Desktop Portal keyboard simulation.
|
||||
|
||||
### [VoxSherpa TTS](https://github.com/CodeBySonu95/VoxSherpa-TTS)
|
||||
|
||||
> VoxSherpa TTS is a 100% offline Android Text-to-Speech app powered by Sherpa-ONNX.
|
||||
> It supports Kokoro-82M, Piper, and VITS engines with multilingual support including
|
||||
> Hindi, English, British English, Japanese, Chinese and 50+ more languages.
|
||||
|
||||
- [Download APK (All Versions)](https://github.com/CodeBySonu95/VoxSherpa-TTS/releases)
|
||||
- Android 11+ · 100% offline · No telemetry
|
||||
|
||||
<div align="center">
|
||||
|
||||
| Generate | Models | Library | Settings |
|
||||
|:---:|:---:|:---:|:---:|
|
||||
| <img src="https://raw.githubusercontent.com/CodeBySonu95/VoxSherpa-TTS/main/fastlane/metadata/android/en-US/images/phoneScreenshots/1.jpg" width="180"/> | <img src="https://raw.githubusercontent.com/CodeBySonu95/VoxSherpa-TTS/main/fastlane/metadata/android/en-US/images/phoneScreenshots/2.jpg" width="180"/> | <img src="https://raw.githubusercontent.com/CodeBySonu95/VoxSherpa-TTS/main/fastlane/metadata/android/en-US/images/phoneScreenshots/3.jpg" width="180"/> | <img src="https://raw.githubusercontent.com/CodeBySonu95/VoxSherpa-TTS/main/fastlane/metadata/android/en-US/images/phoneScreenshots/4.jpg" width="180"/> |
|
||||
|
||||
</div>
|
||||
|
||||
---
|
||||
### [BreezeApp](https://github.com/mtkresearch/BreezeApp) from [MediaTek Research](https://github.com/mtkresearch)
|
||||
|
||||
> BreezeAPP is a mobile AI application developed for both Android and iOS platforms.
|
||||
> Users can download it directly from the App Store and enjoy a variety of features
|
||||
> offline, including speech-to-text, text-to-speech, text-based chatbot interactions,
|
||||
> and image question-answering
|
||||
|
||||
- [Download APK for BreezeAPP](https://huggingface.co/MediaTek-Research/BreezeApp/resolve/main/BreezeApp.apk)
|
||||
- [APK 中国镜像](https://hf-mirror.com/MediaTek-Research/BreezeApp/blob/main/BreezeApp.apk)
|
||||
|
||||
| 1 | 2 | 3 |
|
||||
|---|---|---|
|
||||
||||
|
||||
|
||||
### [Open-LLM-VTuber](https://github.com/t41372/Open-LLM-VTuber)
|
||||
|
||||
Talk to any LLM with hands-free voice interaction, voice interruption, and Live2D taking
|
||||
face running locally across platforms
|
||||
|
||||
See also <https://github.com/t41372/Open-LLM-VTuber/pull/50>
|
||||
|
||||
### [voiceapi](https://github.com/ruzhila/voiceapi)
|
||||
|
||||
<details>
|
||||
<summary>Streaming ASR and TTS based on FastAPI</summary>
|
||||
|
||||
|
||||
It shows how to use the ASR and TTS Python APIs with FastAPI.
|
||||
</details>
|
||||
|
||||
### [腾讯会议摸鱼工具 TMSpeech](https://github.com/jxlpzqc/TMSpeech)
|
||||
|
||||
Uses streaming ASR in C# with graphical user interface.
|
||||
|
||||
Video demo in Chinese: [【开源】Windows实时字幕软件(网课/开会必备)](https://www.bilibili.com/video/BV1rX4y1p7Nx)
|
||||
|
||||
### [lol互动助手](https://github.com/l1veIn/lol-wom-electron)
|
||||
|
||||
It uses the JavaScript API of sherpa-onnx along with [Electron](https://electronjs.org/)
|
||||
|
||||
Video demo in Chinese: [爆了!炫神教你开打字挂!真正影响胜率的英雄联盟工具!英雄联盟的最后一块拼图!和游戏中的每个人无障碍沟通!](https://www.bilibili.com/video/BV142tje9E74)
|
||||
|
||||
### [Sherpa-ONNX 语音识别服务器](https://github.com/hfyydd/sherpa-onnx-server)
|
||||
|
||||
A server based on nodejs providing Restful API for speech recognition.
|
||||
|
||||
### [QSmartAssistant](https://github.com/xinhecuican/QSmartAssistant)
|
||||
|
||||
一个模块化,全过程可离线,低占用率的对话机器人/智能音箱
|
||||
|
||||
It uses QT. Both [ASR](https://github.com/xinhecuican/QSmartAssistant/blob/master/doc/%E5%AE%89%E8%A3%85.md#asr)
|
||||
and [TTS](https://github.com/xinhecuican/QSmartAssistant/blob/master/doc/%E5%AE%89%E8%A3%85.md#tts)
|
||||
are used.
|
||||
|
||||
### [Flutter-EasySpeechRecognition](https://github.com/Jason-chen-coder/Flutter-EasySpeechRecognition)
|
||||
|
||||
It extends [./flutter-examples/streaming_asr](./flutter-examples/streaming_asr) by
|
||||
downloading models inside the app to reduce the size of the app.
|
||||
|
||||
Note: [[Team B] Sherpa AI backend](https://github.com/umgc/spring2025/pull/82) also uses
|
||||
sherpa-onnx in a Flutter APP.
|
||||
|
||||
### [sherpa-onnx-unity](https://github.com/xue-fei/sherpa-onnx-unity)
|
||||
|
||||
sherpa-onnx in Unity. See also [#1695](https://github.com/k2-fsa/sherpa-onnx/issues/1695),
|
||||
[#1892](https://github.com/k2-fsa/sherpa-onnx/issues/1892), and [#1859](https://github.com/k2-fsa/sherpa-onnx/issues/1859)
|
||||
|
||||
### [xiaozhi-esp32-server](https://github.com/xinnan-tech/xiaozhi-esp32-server)
|
||||
|
||||
本项目为xiaozhi-esp32提供后端服务,帮助您快速搭建ESP32设备控制服务器
|
||||
Backend service for xiaozhi-esp32, helps you quickly build an ESP32 device control server.
|
||||
|
||||
See also
|
||||
|
||||
- [ASR新增轻量级sherpa-onnx-asr](https://github.com/xinnan-tech/xiaozhi-esp32-server/issues/315)
|
||||
- [feat: ASR增加sherpa-onnx模型](https://github.com/xinnan-tech/xiaozhi-esp32-server/pull/379)
|
||||
|
||||
### [KaithemAutomation](https://github.com/EternityForest/KaithemAutomation)
|
||||
|
||||
Pure Python, GUI-focused home automation/consumer grade SCADA.
|
||||
|
||||
It uses TTS from sherpa-onnx. See also [✨ Speak command that uses the new globally configured TTS model.](https://github.com/EternityForest/KaithemAutomation/commit/8e64d2b138725e426532f7d66bb69dd0b4f53693)
|
||||
|
||||
### [Open-XiaoAI KWS](https://github.com/idootop/open-xiaoai-kws)
|
||||
|
||||
Enable custom wake word for XiaoAi Speakers. 让小爱音箱支持自定义唤醒词。
|
||||
|
||||
Video demo in Chinese: [小爱同学启动~˶╹ꇴ╹˶!](https://www.bilibili.com/video/BV1YfVUz5EMj)
|
||||
|
||||
### [C++ WebSocket ASR Server](https://github.com/mawwalker/stt-server)
|
||||
|
||||
It provides a WebSocket server based on C++ for ASR using sherpa-onnx.
|
||||
|
||||
### [Go WebSocket Server](https://github.com/bbeyondllove/asr_server)
|
||||
|
||||
It provides a WebSocket server based on the Go programming language for sherpa-onnx.
|
||||
|
||||
### [Making robot Paimon, Ep10 "The AI Part 1"](https://www.youtube.com/watch?v=KxPKkwxGWZs)
|
||||
|
||||
It is a [YouTube video](https://www.youtube.com/watch?v=KxPKkwxGWZs),
|
||||
showing how the author tried to use AI so he can have a conversation with Paimon.
|
||||
|
||||
It uses sherpa-onnx for speech-to-text and text-to-speech.
|
||||
|1|
|
||||
|---|
|
||||
||
|
||||
|
||||
### [TtsReader - Desktop application](https://github.com/ys-pro-duction/TtsReader)
|
||||
|
||||
A desktop text-to-speech application built using Kotlin Multiplatform.
|
||||
|
||||
### [MentraOS](https://github.com/Mentra-Community/MentraOS)
|
||||
|
||||
> Smart glasses OS, with dozens of built-in apps. Users get AI assistant, notifications,
|
||||
> translation, screen mirror, captions, and more. Devs get to write 1 app that runs on
|
||||
> any pair of smart glasses.
|
||||
|
||||
It uses sherpa-onnx for real-time speech recognition on iOS and Android devices.
|
||||
See also <https://github.com/Mentra-Community/MentraOS/pull/861>
|
||||
|
||||
It uses Swift for iOS and Java for Android.
|
||||
|
||||
### [flet_sherpa_onnx](https://github.com/SamYuan1990/flet_sherpa_onnx)
|
||||
|
||||
Flet ASR/STT component based on sherpa-onnx.
|
||||
Example [a chat box agent](https://github.com/SamYuan1990/i18n-agent-action)
|
||||
|
||||
### [achatbot-go](https://github.com/ai-bot-pro/achatbot-go)
|
||||
|
||||
a multimodal chatbot based on go with sherpa-onnx's speech lib api.
|
||||
|
||||
### [fcitx5-vinput](https://github.com/xifan2333/fcitx5-vinput)
|
||||
|
||||
Local offline voice input plugin for [Fcitx5](https://github.com/fcitx/fcitx5) (Linux input method framework).
|
||||
It uses C++ with offline ASR for speech recognition, supporting push-to-talk,
|
||||
command mode, and optional LLM post-processing.
|
||||
|
||||
Video demo in Chinese: [fcitx5-vinput](https://www.bilibili.com/video/BV1a6cUzVE6F)
|
||||
|
||||
### [Wake Word](https://github.com/analyticsinmotion/wake-word)
|
||||
|
||||
A VS Code extension for hands-free voice-activated coding. It uses sherpa-onnx for real-time
|
||||
keyword spotting (KWS) to detect custom wake phrases and trigger VS Code commands by voice.
|
||||
Audio capture is handled by [decibri](https://github.com/analyticsinmotion/decibri), a
|
||||
cross-platform Node.js microphone streaming library with prebuilt native binaries.
|
||||
|
||||
- [VS Code Marketplace](https://marketplace.visualstudio.com/items?itemName=analytics-in-motion.wake-word)
|
||||
- [Open VSX](https://open-vsx.org/extension/analytics-in-motion/wake-word)
|
||||
- [decibri integration guides for sherpa-onnx](https://decibri.dev/docs/node/integrations/sherpa-onnx-stt.html)
|
||||
|
||||
### [SmartSub](https://github.com/buxuku/SmartSub)
|
||||
|
||||
> SmartSub is a local-first cross-platform desktop application for the complete subtitle production pipeline: audio/video transcription, subtitle translation, proofreading, and subtitle burning/muxing.
|
||||
>
|
||||
> It natively integrates sherpa-onnx to power three offline ASR engines — FunASR, Qwen3-ASR, and FireRedASR — delivering high-accuracy Chinese and multilingual speech recognition entirely on-device, with no file uploads required.
|
||||
|
||||
[silero-vad]: https://github.com/snakers4/silero-vad
|
||||
[Raspberry Pi]: https://www.raspberrypi.com/
|
||||
[RV1126]: https://www.rock-chips.com/uploads/pdf/2022.8.26/191/RV1126%20Brief%20Datasheet.pdf
|
||||
[LicheePi4A]: https://sipeed.com/licheepi4a
|
||||
[VisionFive 2]: https://www.starfivetech.com/en/site/boards
|
||||
[旭日X3派]: https://developer.horizon.ai/api/v1/fileData/documents_pi/index.html
|
||||
[爱芯派]: https://wiki.sipeed.com/hardware/zh/maixIII/ax-pi/axpi.html
|
||||
[hf-space-speaker-diarization]: https://huggingface.co/spaces/k2-fsa/speaker-diarization
|
||||
[hf-space-speaker-diarization-cn]: https://hf.qhduan.com/spaces/k2-fsa/speaker-diarization
|
||||
[hf-space-asr]: https://huggingface.co/spaces/k2-fsa/automatic-speech-recognition
|
||||
[hf-space-asr-cn]: https://hf.qhduan.com/spaces/k2-fsa/automatic-speech-recognition
|
||||
[Whisper]: https://github.com/openai/whisper
|
||||
[hf-space-asr-whisper]: https://huggingface.co/spaces/k2-fsa/automatic-speech-recognition-with-whisper
|
||||
[hf-space-asr-whisper-cn]: https://hf.qhduan.com/spaces/k2-fsa/automatic-speech-recognition-with-whisper
|
||||
[hf-space-tts]: https://huggingface.co/spaces/k2-fsa/text-to-speech
|
||||
[hf-space-tts-cn]: https://hf.qhduan.com/spaces/k2-fsa/text-to-speech
|
||||
[hf-space-subtitle]: https://huggingface.co/spaces/k2-fsa/generate-subtitles-for-videos
|
||||
[hf-space-subtitle-cn]: https://hf.qhduan.com/spaces/k2-fsa/generate-subtitles-for-videos
|
||||
[hf-space-audio-tagging]: https://huggingface.co/spaces/k2-fsa/audio-tagging
|
||||
[hf-space-audio-tagging-cn]: https://hf.qhduan.com/spaces/k2-fsa/audio-tagging
|
||||
[hf-space-source-separation]: https://huggingface.co/spaces/k2-fsa/source-separation
|
||||
[hf-space-source-separation-cn]: https://hf.qhduan.com/spaces/k2-fsa/source-separation
|
||||
[hf-space-slid-whisper]: https://huggingface.co/spaces/k2-fsa/spoken-language-identification
|
||||
[hf-space-slid-whisper-cn]: https://hf.qhduan.com/spaces/k2-fsa/spoken-language-identification
|
||||
[wasm-hf-vad]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-sherpa-onnx
|
||||
[wasm-ms-vad]: https://modelscope.cn/studios/csukuangfj/web-assembly-vad-sherpa-onnx
|
||||
[wasm-hf-streaming-asr-zh-en-zipformer]: https://huggingface.co/spaces/k2-fsa/web-assembly-asr-sherpa-onnx-zh-en
|
||||
[wasm-ms-streaming-asr-zh-en-zipformer]: https://modelscope.cn/studios/k2-fsa/web-assembly-asr-sherpa-onnx-zh-en
|
||||
[wasm-hf-streaming-asr-zh-en-paraformer]: https://huggingface.co/spaces/k2-fsa/web-assembly-asr-sherpa-onnx-zh-en-paraformer
|
||||
[wasm-ms-streaming-asr-zh-en-paraformer]: https://modelscope.cn/studios/k2-fsa/web-assembly-asr-sherpa-onnx-zh-en-paraformer
|
||||
[Paraformer-large]: https://www.modelscope.cn/models/damo/speech_paraformer-large_asr_nat-zh-cn-16k-common-vocab8404-pytorch/summary
|
||||
[wasm-hf-streaming-asr-zh-en-yue-paraformer]: https://huggingface.co/spaces/k2-fsa/web-assembly-asr-sherpa-onnx-zh-cantonese-en-paraformer
|
||||
[wasm-ms-streaming-asr-zh-en-yue-paraformer]: https://modelscope.cn/studios/k2-fsa/web-assembly-asr-sherpa-onnx-zh-cantonese-en-paraformer
|
||||
[wasm-hf-streaming-asr-en-zipformer]: https://huggingface.co/spaces/k2-fsa/web-assembly-asr-sherpa-onnx-en
|
||||
[wasm-ms-streaming-asr-en-zipformer]: https://modelscope.cn/studios/k2-fsa/web-assembly-asr-sherpa-onnx-en
|
||||
[SenseVoice]: https://github.com/FunAudioLLM/SenseVoice
|
||||
[wasm-hf-vad-asr-zh-zipformer-ctc-07-03]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-zipformer-ctc
|
||||
[wasm-ms-vad-asr-zh-zipformer-ctc-07-03]: https://modelscope.cn/studios/csukuangfj/web-assembly-vad-asr-sherpa-onnx-zh-zipformer-ctc/summary
|
||||
[wasm-hf-vad-asr-zh-en-ko-ja-yue-sense-voice]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-en-ja-ko-cantonese-sense-voice
|
||||
[wasm-ms-vad-asr-zh-en-ko-ja-yue-sense-voice]: https://www.modelscope.cn/studios/csukuangfj/web-assembly-vad-asr-sherpa-onnx-zh-en-jp-ko-cantonese-sense-voice
|
||||
[wasm-hf-vad-asr-en-whisper-tiny-en]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-en-whisper-tiny
|
||||
[wasm-ms-vad-asr-en-whisper-tiny-en]: https://www.modelscope.cn/studios/csukuangfj/web-assembly-vad-asr-sherpa-onnx-en-whisper-tiny
|
||||
[wasm-hf-vad-asr-en-moonshine-tiny-en]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-en-moonshine-tiny
|
||||
[wasm-ms-vad-asr-en-moonshine-tiny-en]: https://www.modelscope.cn/studios/csukuangfj/web-assembly-vad-asr-sherpa-onnx-en-moonshine-tiny
|
||||
[wasm-hf-vad-asr-en-zipformer-gigaspeech]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-en-zipformer-gigaspeech
|
||||
[wasm-ms-vad-asr-en-zipformer-gigaspeech]: https://www.modelscope.cn/studios/k2-fsa/web-assembly-vad-asr-sherpa-onnx-en-zipformer-gigaspeech
|
||||
[wasm-hf-vad-asr-zh-zipformer-wenetspeech]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-zipformer-wenetspeech
|
||||
[wasm-ms-vad-asr-zh-zipformer-wenetspeech]: https://www.modelscope.cn/studios/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-zipformer-wenetspeech
|
||||
[reazonspeech]: https://research.reazon.jp/_static/reazonspeech_nlp2023.pdf
|
||||
[wasm-hf-vad-asr-ja-zipformer-reazonspeech]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-ja-zipformer
|
||||
[wasm-ms-vad-asr-ja-zipformer-reazonspeech]: https://www.modelscope.cn/studios/csukuangfj/web-assembly-vad-asr-sherpa-onnx-ja-zipformer
|
||||
[gigaspeech2]: https://github.com/speechcolab/gigaspeech2
|
||||
[wasm-hf-vad-asr-th-zipformer-gigaspeech2]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-th-zipformer
|
||||
[wasm-ms-vad-asr-th-zipformer-gigaspeech2]: https://www.modelscope.cn/studios/csukuangfj/web-assembly-vad-asr-sherpa-onnx-th-zipformer
|
||||
[telespeech-asr]: https://github.com/tele-ai/telespeech-asr
|
||||
[wasm-hf-vad-asr-zh-telespeech]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-telespeech
|
||||
[wasm-ms-vad-asr-zh-telespeech]: https://www.modelscope.cn/studios/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-telespeech
|
||||
[wasm-hf-vad-asr-zh-en-paraformer-large]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-en-paraformer
|
||||
[wasm-ms-vad-asr-zh-en-paraformer-large]: https://www.modelscope.cn/studios/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-en-paraformer
|
||||
[wasm-hf-vad-asr-zh-en-paraformer-small]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-en-paraformer-small
|
||||
[wasm-ms-vad-asr-zh-en-paraformer-small]: https://www.modelscope.cn/studios/k2-fsa/web-assembly-vad-asr-sherpa-onnx-zh-en-paraformer-small
|
||||
[dolphin]: https://github.com/dataoceanai/dolphin
|
||||
[wasm-ms-vad-asr-multi-lang-dolphin-base]: https://modelscope.cn/studios/csukuangfj/web-assembly-vad-asr-sherpa-onnx-multi-lang-dophin-ctc
|
||||
[wasm-hf-vad-asr-multi-lang-dolphin-base]: https://huggingface.co/spaces/k2-fsa/web-assembly-vad-asr-sherpa-onnx-multi-lang-dophin-ctc
|
||||
|
||||
[wasm-hf-tts-matcha-zh-en]: https://huggingface.co/spaces/k2-fsa/web-assembly-zh-en-tts-matcha
|
||||
[wasm-hf-tts-matcha-zh]: https://huggingface.co/spaces/k2-fsa/web-assembly-zh-tts-matcha
|
||||
[wasm-ms-tts-matcha-zh-en]: https://modelscope.cn/studios/csukuangfj/web-assembly-zh-en-tts-matcha
|
||||
[wasm-ms-tts-matcha-zh]: https://modelscope.cn/studios/csukuangfj/web-assembly-zh-tts-matcha
|
||||
[wasm-hf-tts-matcha-en]: https://huggingface.co/spaces/k2-fsa/web-assembly-en-tts-matcha
|
||||
[wasm-ms-tts-matcha-en]: https://modelscope.cn/studios/csukuangfj/web-assembly-en-tts-matcha
|
||||
[wasm-hf-tts-piper-en]: https://huggingface.co/spaces/k2-fsa/web-assembly-tts-sherpa-onnx-en
|
||||
[wasm-ms-tts-piper-en]: https://modelscope.cn/studios/k2-fsa/web-assembly-tts-sherpa-onnx-en
|
||||
[wasm-hf-tts-piper-de]: https://huggingface.co/spaces/k2-fsa/web-assembly-tts-sherpa-onnx-de
|
||||
[wasm-ms-tts-piper-de]: https://modelscope.cn/studios/k2-fsa/web-assembly-tts-sherpa-onnx-de
|
||||
[wasm-hf-speaker-diarization]: https://huggingface.co/spaces/k2-fsa/web-assembly-speaker-diarization-sherpa-onnx
|
||||
[wasm-ms-speaker-diarization]: https://www.modelscope.cn/studios/csukuangfj/web-assembly-speaker-diarization-sherpa-onnx
|
||||
[wasm-hf-voice-cloning-zipvoice]: https://huggingface.co/spaces/k2-fsa/web-assembly-zh-en-tts-zipvoice
|
||||
[wasm-ms-voice-cloning-zipvoice]: https://modelscope.cn/studios/csukuangfj/web-assembly-zh-en-tts-zipvoice
|
||||
[wasm-hf-voice-cloning-pocket]: https://huggingface.co/spaces/k2-fsa/web-assembly-en-tts-pocket
|
||||
[wasm-ms-voice-cloning-pocket]: https://modelscope.cn/studios/csukuangfj/web-assembly-en-tts-pocket
|
||||
[apk-speaker-diarization]: https://k2-fsa.github.io/sherpa/onnx/speaker-diarization/apk.html
|
||||
[apk-speaker-diarization-cn]: https://k2-fsa.github.io/sherpa/onnx/speaker-diarization/apk-cn.html
|
||||
[apk-streaming-asr]: https://k2-fsa.github.io/sherpa/onnx/android/apk.html
|
||||
[apk-streaming-asr-cn]: https://k2-fsa.github.io/sherpa/onnx/android/apk-cn.html
|
||||
[apk-simula-streaming-asr]: https://k2-fsa.github.io/sherpa/onnx/android/apk-simulate-streaming-asr.html
|
||||
[apk-simula-streaming-asr-cn]: https://k2-fsa.github.io/sherpa/onnx/android/apk-simulate-streaming-asr-cn.html
|
||||
[apk-tts]: https://k2-fsa.github.io/sherpa/onnx/tts/apk-engine.html
|
||||
[apk-tts-cn]: https://k2-fsa.github.io/sherpa/onnx/tts/apk-engine-cn.html
|
||||
[apk-vad]: https://k2-fsa.github.io/sherpa/onnx/vad/apk.html
|
||||
[apk-vad-cn]: https://k2-fsa.github.io/sherpa/onnx/vad/apk-cn.html
|
||||
[apk-vad-asr]: https://k2-fsa.github.io/sherpa/onnx/vad/apk-asr.html
|
||||
[apk-vad-asr-cn]: https://k2-fsa.github.io/sherpa/onnx/vad/apk-asr-cn.html
|
||||
[apk-2pass]: https://k2-fsa.github.io/sherpa/onnx/android/apk-2pass.html
|
||||
[apk-2pass-cn]: https://k2-fsa.github.io/sherpa/onnx/android/apk-2pass-cn.html
|
||||
[apk-at]: https://k2-fsa.github.io/sherpa/onnx/audio-tagging/apk.html
|
||||
[apk-at-cn]: https://k2-fsa.github.io/sherpa/onnx/audio-tagging/apk-cn.html
|
||||
[apk-at-wearos]: https://k2-fsa.github.io/sherpa/onnx/audio-tagging/apk-wearos.html
|
||||
[apk-at-wearos-cn]: https://k2-fsa.github.io/sherpa/onnx/audio-tagging/apk-wearos-cn.html
|
||||
[apk-sid]: https://k2-fsa.github.io/sherpa/onnx/speaker-identification/apk.html
|
||||
[apk-sid-cn]: https://k2-fsa.github.io/sherpa/onnx/speaker-identification/apk-cn.html
|
||||
[apk-slid]: https://k2-fsa.github.io/sherpa/onnx/spoken-language-identification/apk.html
|
||||
[apk-slid-cn]: https://k2-fsa.github.io/sherpa/onnx/spoken-language-identification/apk-cn.html
|
||||
[apk-kws]: https://k2-fsa.github.io/sherpa/onnx/kws/apk.html
|
||||
[apk-kws-cn]: https://k2-fsa.github.io/sherpa/onnx/kws/apk-cn.html
|
||||
[apk-flutter-streaming-asr]: https://k2-fsa.github.io/sherpa/onnx/flutter/pre-built-app.html#streaming-speech-recognition-stt-asr
|
||||
[apk-flutter-streaming-asr-cn]: https://k2-fsa.github.io/sherpa/onnx/flutter/pre-built-app.html#streaming-speech-recognition-stt-asr
|
||||
[flutter-tts-android]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-android.html
|
||||
[flutter-tts-android-cn]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-android-cn.html
|
||||
[flutter-tts-linux]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-linux.html
|
||||
[flutter-tts-linux-cn]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-linux-cn.html
|
||||
[flutter-tts-macos-x64]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-macos-x64.html
|
||||
[flutter-tts-macos-arm64-cn]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-macos-arm64-cn.html
|
||||
[flutter-tts-macos-arm64]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-macos-arm64.html
|
||||
[flutter-tts-macos-x64-cn]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-macos-x64-cn.html
|
||||
[flutter-tts-win-x64]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-win.html
|
||||
[flutter-tts-win-x64-cn]: https://k2-fsa.github.io/sherpa/onnx/flutter/tts-win-cn.html
|
||||
[lazarus-subtitle]: https://k2-fsa.github.io/sherpa/onnx/lazarus/download-generated-subtitles.html
|
||||
[lazarus-subtitle-cn]: https://k2-fsa.github.io/sherpa/onnx/lazarus/download-generated-subtitles-cn.html
|
||||
[asr-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/asr-models
|
||||
[tts-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/tts-models
|
||||
[vad-models]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/silero_vad.onnx
|
||||
[kws-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/kws-models
|
||||
[at-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/audio-tagging-models
|
||||
[sid-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/speaker-recongition-models
|
||||
[slid-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/speaker-recongition-models
|
||||
[punct-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/punctuation-models
|
||||
[speaker-segmentation-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/speaker-segmentation-models
|
||||
[GigaSpeech]: https://github.com/SpeechColab/GigaSpeech
|
||||
[WenetSpeech]: https://github.com/wenet-e2e/WenetSpeech
|
||||
[sherpa-onnx-streaming-zipformer-bilingual-zh-en-2023-02-20]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-streaming-zipformer-bilingual-zh-en-2023-02-20.tar.bz2
|
||||
[sherpa-onnx-streaming-zipformer-small-bilingual-zh-en-2023-02-16]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-streaming-zipformer-small-bilingual-zh-en-2023-02-16.tar.bz2
|
||||
[sherpa-onnx-streaming-zipformer-korean-2024-06-16]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-streaming-zipformer-korean-2024-06-16.tar.bz2
|
||||
[sherpa-onnx-streaming-zipformer-zh-14M-2023-02-23]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-streaming-zipformer-zh-14M-2023-02-23.tar.bz2
|
||||
[sherpa-onnx-streaming-zipformer-en-20M-2023-02-17]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-streaming-zipformer-en-20M-2023-02-17.tar.bz2
|
||||
[sherpa-onnx-zipformer-ja-reazonspeech-2024-08-01]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-zipformer-ja-reazonspeech-2024-08-01.tar.bz2
|
||||
[sherpa-onnx-zipformer-ru-2024-09-18]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-zipformer-ru-2024-09-18.tar.bz2
|
||||
[sherpa-onnx-zipformer-korean-2024-06-24]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-zipformer-korean-2024-06-24.tar.bz2
|
||||
[sherpa-onnx-zipformer-thai-2024-06-20]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-zipformer-thai-2024-06-20.tar.bz2
|
||||
[sherpa-onnx-nemo-transducer-giga-am-russian-2024-10-24]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-nemo-transducer-giga-am-russian-2024-10-24.tar.bz2
|
||||
[sherpa-onnx-paraformer-zh-2024-03-09]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-paraformer-zh-2024-03-09.tar.bz2
|
||||
[sherpa-onnx-nemo-ctc-giga-am-russian-2024-10-24]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-nemo-ctc-giga-am-russian-2024-10-24.tar.bz2
|
||||
[sherpa-onnx-telespeech-ctc-int8-zh-2024-06-04]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-telespeech-ctc-int8-zh-2024-06-04.tar.bz2
|
||||
[sherpa-onnx-sense-voice-zh-en-ja-ko-yue-2024-07-17]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-sense-voice-zh-en-ja-ko-yue-2024-07-17.tar.bz2
|
||||
[sherpa-onnx-streaming-zipformer-fr-2023-04-14]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-streaming-zipformer-fr-2023-04-14.tar.bz2
|
||||
[Moonshine tiny]: https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/sherpa-onnx-moonshine-tiny-en-int8.tar.bz2
|
||||
[NVIDIA Jetson Orin NX]: https://developer.download.nvidia.com/assets/embedded/secure/jetson/orin_nx/docs/Jetson_Orin_NX_DS-10712-001_v0.5.pdf?RCPGu9Q6OVAOv7a7vgtwc9-BLScXRIWq6cSLuditMALECJ_dOj27DgnqAPGVnT2VpiNpQan9SyFy-9zRykR58CokzbXwjSA7Gj819e91AXPrWkGZR3oS1VLxiDEpJa_Y0lr7UT-N4GnXtb8NlUkP4GkCkkF_FQivGPrAucCUywL481GH_WpP_p7ziHU1Wg==&t=eyJscyI6ImdzZW8iLCJsc2QiOiJodHRwczovL3d3dy5nb29nbGUuY29tLmhrLyJ9
|
||||
[NVIDIA Jetson Nano B01]: https://www.seeedstudio.com/blog/2020/01/16/new-revision-of-jetson-nano-dev-kit-now-supports-new-jetson-nano-module/
|
||||
[speech-enhancement-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/speech-enhancement-models
|
||||
[source-separation-models]: https://github.com/k2-fsa/sherpa-onnx/releases/tag/source-separation-models
|
||||
[RK3588]: https://www.rock-chips.com/uploads/pdf/2022.8.26/192/RK3588%20Brief%20Datasheet.pdf
|
||||
[spleeter]: https://github.com/deezer/spleeter
|
||||
[UVR]: https://github.com/Anjok07/ultimatevocalremovergui
|
||||
[gtcrn]: https://github.com/Xiaobin-Rong/gtcrn
|
||||
[tts-url]: https://k2-fsa.github.io/sherpa/onnx/tts/all-in-one.html
|
||||
[ss-url]: https://k2-fsa.github.io/sherpa/onnx/source-separation/index.html
|
||||
[sd-url]: https://k2-fsa.github.io/sherpa/onnx/speaker-diarization/index.html
|
||||
[slid-url]: https://k2-fsa.github.io/sherpa/onnx/spoken-language-identification/index.html
|
||||
[at-url]: https://k2-fsa.github.io/sherpa/onnx/audio-tagging/index.html
|
||||
[vad-url]: https://k2-fsa.github.io/sherpa/onnx/vad/index.html
|
||||
[kws-url]: https://k2-fsa.github.io/sherpa/onnx/kws/index.html
|
||||
[punct-url]: https://k2-fsa.github.io/sherpa/onnx/punctuation/index.html
|
||||
[se-url]: https://k2-fsa.github.io/sherpa/onnx/speech-enhancement/index.html
|
||||
[rknpu-doc]: https://k2-fsa.github.io/sherpa/onnx/rknn/index.html
|
||||
[qnn-doc]: https://k2-fsa.github.io/sherpa/onnx/qnn/index.html
|
||||
[ascend-doc]: https://k2-fsa.github.io/sherpa/onnx/ascend/index.html
|
||||
[axera-npu]: https://axera-tech.com/Skill/166.html
|
||||
[SpacemiT-K1]: https://cdn-resource.spacemit.com/file/chip/K1/K1_brief_zh.pdf
|
||||
[SpacemiT-K3]: https://cdn-resource.spacemit.com/file/chip/K3/K3_brief_zh.pdf
|
||||
@@ -0,0 +1,475 @@
|
||||
use std::env;
|
||||
use std::error::Error;
|
||||
use std::ffi::OsStr;
|
||||
use std::fs;
|
||||
use std::fs::File;
|
||||
use std::io;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::{collections::HashSet, ffi::OsString};
|
||||
|
||||
use bzip2::read::BzDecoder;
|
||||
use tar::Archive;
|
||||
|
||||
const RELEASE_BASE_URL: &str = "https://github.com/k2-fsa/sherpa-onnx/releases/download";
|
||||
const SHERPA_ONNX_STATIC_LIBS: &[&str] = &[
|
||||
"sherpa-onnx-c-api",
|
||||
"sherpa-onnx-core",
|
||||
"kaldi-decoder-core",
|
||||
"sherpa-onnx-kaldifst-core",
|
||||
"sherpa-onnx-fstfar",
|
||||
"sherpa-onnx-fst",
|
||||
"kaldi-native-fbank-core",
|
||||
"kissfft-float",
|
||||
"piper_phonemize",
|
||||
"espeak-ng",
|
||||
"ucd",
|
||||
"onnxruntime",
|
||||
"ssentencepiece_core",
|
||||
];
|
||||
|
||||
type DynError = Box<dyn Error>;
|
||||
|
||||
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
||||
enum LinkMode {
|
||||
Static,
|
||||
Shared,
|
||||
}
|
||||
|
||||
fn main() {
|
||||
if let Err(err) = try_main() {
|
||||
panic!("{err}");
|
||||
}
|
||||
}
|
||||
|
||||
fn try_main() -> Result<(), DynError> {
|
||||
println!("cargo:rerun-if-env-changed=SHERPA_ONNX_LIB_DIR");
|
||||
println!("cargo:rerun-if-env-changed=SHERPA_ONNX_ARCHIVE_DIR");
|
||||
println!("cargo:rerun-if-env-changed=DOCS_RS");
|
||||
|
||||
if env::var_os("DOCS_RS").is_some() {
|
||||
// docs.rs sets DOCS_RS=1; skip downloading/linking native libraries
|
||||
// so that `cargo doc` can succeed without the real C artifacts.
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let target_os = env::var("CARGO_CFG_TARGET_OS")?;
|
||||
let target_arch = env::var("CARGO_CFG_TARGET_ARCH")?;
|
||||
let link_mode = resolve_link_mode()?;
|
||||
let lib_dir = resolve_lib_dir(link_mode, &target_os, &target_arch)?;
|
||||
|
||||
println!("cargo:rustc-link-search=native={}", lib_dir.display());
|
||||
|
||||
if link_mode == LinkMode::Shared && matches!(target_os.as_str(), "linux" | "macos") {
|
||||
println!("cargo:rustc-link-arg=-Wl,-rpath,{}", lib_dir.display());
|
||||
emit_relative_rpath(&target_os);
|
||||
copy_unix_runtime_libs(&lib_dir, &target_os)?;
|
||||
}
|
||||
|
||||
if link_mode == LinkMode::Shared && target_os == "windows" {
|
||||
copy_windows_runtime_dlls(&lib_dir)?;
|
||||
}
|
||||
|
||||
match (link_mode, target_os.as_str()) {
|
||||
(LinkMode::Static, "ios") => emit_ios_link_directives(),
|
||||
(LinkMode::Shared, "android") => emit_android_link_directives(),
|
||||
(LinkMode::Static, _) => emit_static_link_directives(&target_os),
|
||||
(LinkMode::Shared, _) => emit_shared_link_directives(),
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn resolve_link_mode() -> Result<LinkMode, DynError> {
|
||||
let static_enabled = env::var_os("CARGO_FEATURE_STATIC").is_some();
|
||||
let shared_enabled = env::var_os("CARGO_FEATURE_SHARED").is_some();
|
||||
|
||||
if static_enabled && shared_enabled {
|
||||
return Err("Features `static` and `shared` cannot be enabled at the same time".into());
|
||||
}
|
||||
|
||||
if shared_enabled {
|
||||
Ok(LinkMode::Shared)
|
||||
} else {
|
||||
Ok(LinkMode::Static)
|
||||
}
|
||||
}
|
||||
|
||||
fn resolve_lib_dir(
|
||||
link_mode: LinkMode,
|
||||
target_os: &str,
|
||||
target_arch: &str,
|
||||
) -> Result<PathBuf, DynError> {
|
||||
if let Some(path) = env::var_os("SHERPA_ONNX_LIB_DIR") {
|
||||
let path = PathBuf::from(path);
|
||||
if !path.is_dir() {
|
||||
return Err(format!(
|
||||
"SHERPA_ONNX_LIB_DIR does not exist or is not a directory: {}",
|
||||
path.display()
|
||||
)
|
||||
.into());
|
||||
}
|
||||
return Ok(path);
|
||||
}
|
||||
|
||||
if matches!(target_os, "android" | "ios") {
|
||||
return Err(format!(
|
||||
"SHERPA_ONNX_LIB_DIR must point at the verified sherpa-onnx v{} \
|
||||
mobile libraries when building for os={target_os}, arch={target_arch}",
|
||||
env!("CARGO_PKG_VERSION")
|
||||
)
|
||||
.into());
|
||||
}
|
||||
|
||||
download_prebuilt_libs(link_mode, target_os, target_arch)
|
||||
}
|
||||
|
||||
fn download_prebuilt_libs(
|
||||
link_mode: LinkMode,
|
||||
target_os: &str,
|
||||
target_arch: &str,
|
||||
) -> Result<PathBuf, DynError> {
|
||||
let archive_name = archive_name(link_mode, target_os, target_arch)?;
|
||||
let archive_stem = archive_name.trim_end_matches(".tar.bz2");
|
||||
|
||||
let out_dir = PathBuf::from(env::var("OUT_DIR")?);
|
||||
let cache_root = target_dir_from_out_dir(&out_dir)?.join("sherpa-onnx-prebuilt");
|
||||
let extracted_dir = cache_root.join(archive_stem);
|
||||
let lib_dir = extracted_dir.join("lib");
|
||||
|
||||
if lib_dir.is_dir() {
|
||||
return Ok(lib_dir);
|
||||
}
|
||||
|
||||
fs::create_dir_all(&cache_root)?;
|
||||
|
||||
let archive_path = cache_root.join(&archive_name);
|
||||
if !archive_path.is_file() {
|
||||
if let Some(local_archive_dir) = env::var_os("SHERPA_ONNX_ARCHIVE_DIR") {
|
||||
let local_archive_path = PathBuf::from(local_archive_dir).join(&archive_name);
|
||||
if !local_archive_path.is_file() {
|
||||
return Err(format!(
|
||||
"SHERPA_ONNX_ARCHIVE_DIR does not contain expected archive: {}",
|
||||
local_archive_path.display()
|
||||
)
|
||||
.into());
|
||||
}
|
||||
|
||||
copy_file_atomically(&local_archive_path, &archive_path)?;
|
||||
} else {
|
||||
let version = env!("CARGO_PKG_VERSION");
|
||||
let url = format!("{RELEASE_BASE_URL}/v{version}/{archive_name}");
|
||||
eprintln!("Downloading sherpa-onnx libs from {url}");
|
||||
|
||||
let response = ureq::builder()
|
||||
.try_proxy_from_env(true)
|
||||
.build()
|
||||
.get(&url)
|
||||
.call()
|
||||
.map_err(|e| format!("Failed to download sherpa-onnx archive from {url}: {e}"))?;
|
||||
let mut reader = response.into_reader();
|
||||
write_reader_atomically(&mut reader, &archive_path)?;
|
||||
}
|
||||
}
|
||||
|
||||
if extracted_dir.exists() {
|
||||
fs::remove_dir_all(&extracted_dir)?;
|
||||
}
|
||||
|
||||
let unpack_result: Result<(), DynError> = (|| {
|
||||
let tar_file = File::open(&archive_path)?;
|
||||
let decoder = BzDecoder::new(tar_file);
|
||||
let mut archive = Archive::new(decoder);
|
||||
archive.unpack(&cache_root)?;
|
||||
Ok(())
|
||||
})();
|
||||
if let Err(err) = unpack_result {
|
||||
let _ = fs::remove_file(&archive_path);
|
||||
let _ = fs::remove_dir_all(&extracted_dir);
|
||||
return Err(format!(
|
||||
"Failed to unpack cached archive {}: {err}",
|
||||
archive_path.display()
|
||||
)
|
||||
.into());
|
||||
}
|
||||
|
||||
if !lib_dir.is_dir() {
|
||||
return Err(format!(
|
||||
"Downloaded archive did not contain a lib directory: {}",
|
||||
lib_dir.display()
|
||||
)
|
||||
.into());
|
||||
}
|
||||
|
||||
eprintln!("Downloaded sherpa-onnx libs to {}", extracted_dir.display());
|
||||
|
||||
Ok(lib_dir)
|
||||
}
|
||||
|
||||
fn archive_name(
|
||||
link_mode: LinkMode,
|
||||
target_os: &str,
|
||||
target_arch: &str,
|
||||
) -> Result<String, DynError> {
|
||||
let version = env!("CARGO_PKG_VERSION");
|
||||
let name = match (link_mode, target_os, target_arch) {
|
||||
(LinkMode::Static, "linux", "x86_64") => {
|
||||
format!("sherpa-onnx-v{version}-linux-x64-static-lib.tar.bz2")
|
||||
}
|
||||
(LinkMode::Static, "linux", "aarch64") => {
|
||||
format!("sherpa-onnx-v{version}-linux-aarch64-static-lib.tar.bz2")
|
||||
}
|
||||
(LinkMode::Static, "macos", "x86_64") => {
|
||||
format!("sherpa-onnx-v{version}-osx-x64-static-lib.tar.bz2")
|
||||
}
|
||||
(LinkMode::Static, "macos", "aarch64") => {
|
||||
format!("sherpa-onnx-v{version}-osx-arm64-static-lib.tar.bz2")
|
||||
}
|
||||
(LinkMode::Static, "windows", "x86_64") => {
|
||||
format!("sherpa-onnx-v{version}-win-x64-static-MT-Release-lib.tar.bz2")
|
||||
}
|
||||
(LinkMode::Shared, "linux", "x86_64") => {
|
||||
format!("sherpa-onnx-v{version}-linux-x64-shared-lib.tar.bz2")
|
||||
}
|
||||
(LinkMode::Shared, "linux", "aarch64") => {
|
||||
format!("sherpa-onnx-v{version}-linux-aarch64-shared-cpu-lib.tar.bz2")
|
||||
}
|
||||
(LinkMode::Shared, "macos", "x86_64") => {
|
||||
format!("sherpa-onnx-v{version}-osx-x64-shared-lib.tar.bz2")
|
||||
}
|
||||
(LinkMode::Shared, "macos", "aarch64") => {
|
||||
format!("sherpa-onnx-v{version}-osx-arm64-shared-lib.tar.bz2")
|
||||
}
|
||||
(LinkMode::Shared, "windows", "x86_64") => {
|
||||
format!("sherpa-onnx-v{version}-win-x64-shared-MT-Release-lib.tar.bz2")
|
||||
}
|
||||
_ => {
|
||||
return Err(format!(
|
||||
"Unsupported target for sherpa-onnx prebuilt libs: os={target_os}, arch={target_arch}"
|
||||
)
|
||||
.into())
|
||||
}
|
||||
};
|
||||
|
||||
Ok(name)
|
||||
}
|
||||
|
||||
fn emit_shared_link_directives() {
|
||||
println!("cargo:rustc-link-lib=dylib=sherpa-onnx-c-api");
|
||||
println!("cargo:rustc-link-lib=dylib=onnxruntime");
|
||||
}
|
||||
|
||||
fn emit_android_link_directives() {
|
||||
println!("cargo:rustc-link-lib=dylib=sherpa-onnx-c-api");
|
||||
println!("cargo:rustc-link-lib=dylib=onnxruntime");
|
||||
println!("cargo:rustc-link-lib=dylib=log");
|
||||
}
|
||||
|
||||
fn emit_ios_link_directives() {
|
||||
println!("cargo:rustc-link-lib=static=sherpa-onnx");
|
||||
println!("cargo:rustc-link-lib=static=onnxruntime");
|
||||
println!("cargo:rustc-link-lib=dylib=c++");
|
||||
for framework in [
|
||||
"Accelerate",
|
||||
"CoreML",
|
||||
"Foundation",
|
||||
"Metal",
|
||||
"QuartzCore",
|
||||
"Security",
|
||||
] {
|
||||
println!("cargo:rustc-link-lib=framework={framework}");
|
||||
}
|
||||
}
|
||||
|
||||
fn emit_static_link_directives(target_os: &str) {
|
||||
for lib in SHERPA_ONNX_STATIC_LIBS {
|
||||
println!("cargo:rustc-link-lib=static={lib}");
|
||||
}
|
||||
|
||||
match target_os {
|
||||
"linux" => {
|
||||
println!("cargo:rustc-link-lib=dylib=stdc++");
|
||||
println!("cargo:rustc-link-lib=dylib=m");
|
||||
println!("cargo:rustc-link-lib=dylib=pthread");
|
||||
println!("cargo:rustc-link-lib=dylib=dl");
|
||||
}
|
||||
"macos" => {
|
||||
println!("cargo:rustc-link-lib=dylib=c++");
|
||||
println!("cargo:rustc-link-lib=framework=Foundation");
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
fn target_dir_from_out_dir(out_dir: &Path) -> Result<PathBuf, DynError> {
|
||||
if let Ok(explicit_target_dir) = env::var("CARGO_TARGET_DIR") {
|
||||
return Ok(PathBuf::from(explicit_target_dir));
|
||||
}
|
||||
|
||||
if let Some(target_dir) = out_dir
|
||||
.ancestors()
|
||||
.find(|path| path.file_name() == Some(OsStr::new("target")))
|
||||
{
|
||||
return Ok(target_dir.to_path_buf());
|
||||
}
|
||||
|
||||
Ok(out_dir.to_path_buf())
|
||||
}
|
||||
|
||||
fn emit_relative_rpath(target_os: &str) {
|
||||
match target_os {
|
||||
"linux" => println!("cargo:rustc-link-arg=-Wl,-rpath,$ORIGIN"),
|
||||
"macos" => println!("cargo:rustc-link-arg=-Wl,-rpath,@loader_path"),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
fn profile_output_dirs() -> Result<[PathBuf; 2], DynError> {
|
||||
let out_dir = PathBuf::from(env::var("OUT_DIR")?);
|
||||
let profile = env::var("PROFILE")?;
|
||||
let profile_dir = out_dir
|
||||
.ancestors()
|
||||
.find(|path| path.file_name() == Some(OsStr::new(&profile)))
|
||||
.ok_or_else(|| {
|
||||
format!(
|
||||
"Could not locate Cargo profile directory from {}",
|
||||
out_dir.display()
|
||||
)
|
||||
})?
|
||||
.to_path_buf();
|
||||
|
||||
Ok([profile_dir.clone(), profile_dir.join("examples")])
|
||||
}
|
||||
|
||||
fn copy_unix_runtime_libs(lib_dir: &Path, target_os: &str) -> Result<(), DynError> {
|
||||
let runtime_libs: Vec<PathBuf> = fs::read_dir(lib_dir)?
|
||||
.filter_map(|entry| entry.ok().map(|e| e.path()))
|
||||
.filter(|path| {
|
||||
path.file_name()
|
||||
.and_then(OsStr::to_str)
|
||||
.map(|name| match target_os {
|
||||
"linux" => name.contains(".so"),
|
||||
"macos" => name.ends_with(".dylib"),
|
||||
_ => false,
|
||||
})
|
||||
.unwrap_or(false)
|
||||
})
|
||||
.collect();
|
||||
|
||||
if runtime_libs.is_empty() {
|
||||
return Err(format!("No shared runtime libraries found in {}", lib_dir.display()).into());
|
||||
}
|
||||
|
||||
let mut copy_plan = Vec::<(PathBuf, OsString)>::new();
|
||||
let mut planned_names = HashSet::<OsString>::new();
|
||||
|
||||
for lib in runtime_libs {
|
||||
if !lib.exists() {
|
||||
continue;
|
||||
}
|
||||
|
||||
let lib_name = lib
|
||||
.file_name()
|
||||
.ok_or_else(|| format!("Invalid runtime library path: {}", lib.display()))?
|
||||
.to_os_string();
|
||||
|
||||
let source = fs::canonicalize(&lib).unwrap_or(lib.clone());
|
||||
if planned_names.insert(lib_name.clone()) {
|
||||
copy_plan.push((source.clone(), lib_name));
|
||||
}
|
||||
|
||||
if let Some(source_name) = source.file_name() {
|
||||
let source_name = source_name.to_os_string();
|
||||
if planned_names.insert(source_name.clone()) {
|
||||
copy_plan.push((source.clone(), source_name));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if copy_plan.is_empty() {
|
||||
return Err(format!(
|
||||
"No usable shared runtime libraries found in {}",
|
||||
lib_dir.display()
|
||||
)
|
||||
.into());
|
||||
}
|
||||
|
||||
for dest_dir in profile_output_dirs()? {
|
||||
fs::create_dir_all(&dest_dir)?;
|
||||
for (source, dest_name) in ©_plan {
|
||||
let dest = dest_dir.join(dest_name);
|
||||
fs::copy(source, &dest)?;
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn temp_path_for(path: &Path) -> PathBuf {
|
||||
let mut temp_name = path
|
||||
.file_name()
|
||||
.map(OsStr::to_os_string)
|
||||
.unwrap_or_else(|| OsString::from("tmp"));
|
||||
temp_name.push(".part");
|
||||
path.with_file_name(temp_name)
|
||||
}
|
||||
|
||||
fn copy_file_atomically(src: &Path, dst: &Path) -> Result<(), DynError> {
|
||||
let temp_path = temp_path_for(dst);
|
||||
if temp_path.exists() {
|
||||
let _ = fs::remove_file(&temp_path);
|
||||
}
|
||||
fs::copy(src, &temp_path)?;
|
||||
fs::rename(&temp_path, dst)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn write_reader_atomically(reader: &mut dyn io::Read, dst: &Path) -> Result<(), DynError> {
|
||||
let temp_path = temp_path_for(dst);
|
||||
if temp_path.exists() {
|
||||
let _ = fs::remove_file(&temp_path);
|
||||
}
|
||||
|
||||
{
|
||||
let mut file = File::create(&temp_path)?;
|
||||
io::copy(reader, &mut file)?;
|
||||
file.sync_all()?;
|
||||
}
|
||||
|
||||
fs::rename(&temp_path, dst)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn copy_windows_runtime_dlls(lib_dir: &Path) -> Result<(), DynError> {
|
||||
let dlls: Vec<PathBuf> = fs::read_dir(lib_dir)?
|
||||
.filter_map(|entry| entry.ok().map(|e| e.path()))
|
||||
.filter(|path| path.extension() == Some(OsStr::new("dll")))
|
||||
.collect();
|
||||
|
||||
if dlls.is_empty() {
|
||||
println!(
|
||||
"cargo:warning=No runtime DLLs found in {}",
|
||||
lib_dir.display()
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let [profile_dir, examples_dir] = profile_output_dirs()?;
|
||||
for dest_dir in [profile_dir.clone(), examples_dir] {
|
||||
fs::create_dir_all(&dest_dir)?;
|
||||
for dll in &dlls {
|
||||
let dest = dest_dir.join(
|
||||
dll.file_name()
|
||||
.ok_or_else(|| format!("Invalid DLL path: {}", dll.display()))?,
|
||||
);
|
||||
fs::copy(dll, &dest)?;
|
||||
}
|
||||
}
|
||||
|
||||
println!(
|
||||
"cargo:warning=Copied Windows runtime DLLs to {} and {}/examples",
|
||||
profile_dir.display(),
|
||||
profile_dir.display()
|
||||
);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
#![allow(non_camel_case_types)]
|
||||
#![allow(non_snake_case)]
|
||||
#![allow(non_upper_case_globals)]
|
||||
|
||||
use std::os::raw::{c_char, c_float};
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineZipformerAudioTaggingModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct AudioTaggingModelConfig {
|
||||
pub zipformer: OfflineZipformerAudioTaggingModelConfig,
|
||||
pub ced: *const c_char,
|
||||
pub num_threads: i32,
|
||||
pub debug: i32,
|
||||
pub provider: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct AudioTaggingConfig {
|
||||
pub model: AudioTaggingModelConfig,
|
||||
pub labels: *const c_char,
|
||||
pub top_k: i32,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct AudioEvent {
|
||||
pub name: *const c_char,
|
||||
pub index: i32,
|
||||
pub prob: c_float,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct AudioTagging {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
pub fn SherpaOnnxCreateAudioTagging(config: *const AudioTaggingConfig) -> *const AudioTagging;
|
||||
|
||||
pub fn SherpaOnnxDestroyAudioTagging(tagger: *const AudioTagging);
|
||||
|
||||
pub fn SherpaOnnxAudioTaggingCreateOfflineStream(
|
||||
tagger: *const AudioTagging,
|
||||
) -> *const crate::offline_asr::OfflineStream;
|
||||
|
||||
pub fn SherpaOnnxAudioTaggingCompute(
|
||||
tagger: *const AudioTagging,
|
||||
s: *const crate::offline_asr::OfflineStream,
|
||||
top_k: i32,
|
||||
) -> *const *const AudioEvent;
|
||||
|
||||
pub fn SherpaOnnxAudioTaggingFreeResults(p: *const *const AudioEvent);
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
#![allow(non_camel_case_types)]
|
||||
#![allow(non_snake_case)]
|
||||
#![allow(non_upper_case_globals)]
|
||||
|
||||
use std::os::raw::{c_char, c_float};
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct KeywordResult {
|
||||
pub keyword: *const c_char,
|
||||
pub tokens: *const c_char,
|
||||
pub tokens_arr: *const *const c_char,
|
||||
pub count: i32,
|
||||
pub timestamps: *mut c_float,
|
||||
pub start_time: c_float,
|
||||
pub json: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct KeywordSpotterConfig {
|
||||
pub feat_config: super::online_asr::FeatureConfig,
|
||||
pub model_config: super::online_asr::OnlineModelConfig,
|
||||
pub max_active_paths: i32,
|
||||
pub num_trailing_blanks: i32,
|
||||
pub keywords_score: c_float,
|
||||
pub keywords_threshold: c_float,
|
||||
pub keywords_file: *const c_char,
|
||||
pub keywords_buf: *const c_char,
|
||||
pub keywords_buf_size: i32,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct KeywordSpotter {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
pub fn SherpaOnnxCreateKeywordSpotter(
|
||||
config: *const KeywordSpotterConfig,
|
||||
) -> *const KeywordSpotter;
|
||||
|
||||
pub fn SherpaOnnxDestroyKeywordSpotter(spotter: *const KeywordSpotter);
|
||||
|
||||
pub fn SherpaOnnxCreateKeywordStream(
|
||||
spotter: *const KeywordSpotter,
|
||||
) -> *const super::online_asr::OnlineStream;
|
||||
|
||||
pub fn SherpaOnnxCreateKeywordStreamWithKeywords(
|
||||
spotter: *const KeywordSpotter,
|
||||
keywords: *const c_char,
|
||||
) -> *const super::online_asr::OnlineStream;
|
||||
|
||||
pub fn SherpaOnnxIsKeywordStreamReady(
|
||||
spotter: *const KeywordSpotter,
|
||||
stream: *const super::online_asr::OnlineStream,
|
||||
) -> i32;
|
||||
|
||||
pub fn SherpaOnnxDecodeKeywordStream(
|
||||
spotter: *const KeywordSpotter,
|
||||
stream: *const super::online_asr::OnlineStream,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxResetKeywordStream(
|
||||
spotter: *const KeywordSpotter,
|
||||
stream: *const super::online_asr::OnlineStream,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxDecodeMultipleKeywordStreams(
|
||||
spotter: *const KeywordSpotter,
|
||||
streams: *const *const super::online_asr::OnlineStream,
|
||||
n: i32,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxGetKeywordResult(
|
||||
spotter: *const KeywordSpotter,
|
||||
stream: *const super::online_asr::OnlineStream,
|
||||
) -> *const KeywordResult;
|
||||
|
||||
pub fn SherpaOnnxDestroyKeywordResult(r: *const KeywordResult);
|
||||
|
||||
pub fn SherpaOnnxGetKeywordResultAsJson(
|
||||
spotter: *const KeywordSpotter,
|
||||
stream: *const super::online_asr::OnlineStream,
|
||||
) -> *const c_char;
|
||||
|
||||
pub fn SherpaOnnxFreeKeywordResultJson(s: *const c_char);
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
#![allow(non_camel_case_types)]
|
||||
#![allow(non_snake_case)]
|
||||
#![allow(non_upper_case_globals)]
|
||||
|
||||
use std::os::raw::c_char;
|
||||
|
||||
extern "C" {
|
||||
pub fn SherpaOnnxGetVersionStr() -> *const c_char;
|
||||
pub fn SherpaOnnxGetGitSha1() -> *const c_char;
|
||||
pub fn SherpaOnnxGetGitDate() -> *const c_char;
|
||||
pub fn SherpaOnnxFileExists(filename: *const c_char) -> i32;
|
||||
}
|
||||
|
||||
pub mod audio_tagging;
|
||||
pub mod kws;
|
||||
pub mod offline_asr;
|
||||
pub mod offline_punctuation;
|
||||
pub mod offline_speaker_diarization;
|
||||
pub mod online_asr;
|
||||
pub mod online_punctuation;
|
||||
pub mod resampler;
|
||||
pub mod speaker_embedding;
|
||||
pub mod speech_denoiser;
|
||||
pub mod spoken_language_identification;
|
||||
pub mod tts;
|
||||
pub mod vad;
|
||||
pub mod wave;
|
||||
|
||||
pub use audio_tagging::*;
|
||||
pub use kws::*;
|
||||
pub use offline_asr::*;
|
||||
pub use offline_punctuation::*;
|
||||
pub use offline_speaker_diarization::*;
|
||||
pub use online_asr::*;
|
||||
pub use online_punctuation::*;
|
||||
pub use resampler::*;
|
||||
pub use speaker_embedding::*;
|
||||
pub use speech_denoiser::*;
|
||||
pub use spoken_language_identification::*;
|
||||
pub use tts::*;
|
||||
pub use vad::*;
|
||||
pub use wave::*;
|
||||
@@ -0,0 +1,281 @@
|
||||
#![allow(non_camel_case_types)]
|
||||
#![allow(non_snake_case)]
|
||||
#![allow(non_upper_case_globals)]
|
||||
|
||||
use std::os::raw::{c_char, c_float};
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineTransducerModelConfig {
|
||||
pub encoder: *const c_char,
|
||||
pub decoder: *const c_char,
|
||||
pub joiner: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineParaformerModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineNemoEncDecCtcModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineWhisperModelConfig {
|
||||
pub encoder: *const c_char,
|
||||
pub decoder: *const c_char,
|
||||
pub language: *const c_char,
|
||||
pub task: *const c_char,
|
||||
pub tail_paddings: i32,
|
||||
pub enable_token_timestamps: i32,
|
||||
pub enable_segment_timestamps: i32,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineCanaryModelConfig {
|
||||
pub encoder: *const c_char,
|
||||
pub decoder: *const c_char,
|
||||
pub src_lang: *const c_char,
|
||||
pub tgt_lang: *const c_char,
|
||||
pub use_pnc: i32,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineFireRedAsrModelConfig {
|
||||
pub encoder: *const c_char,
|
||||
pub decoder: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineMoonshineModelConfig {
|
||||
pub preprocessor: *const c_char,
|
||||
pub encoder: *const c_char,
|
||||
pub uncached_decoder: *const c_char,
|
||||
pub cached_decoder: *const c_char,
|
||||
pub merged_decoder: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineTdnnModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineLMConfig {
|
||||
pub model: *const c_char,
|
||||
pub scale: c_float,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineSenseVoiceModelConfig {
|
||||
pub model: *const c_char,
|
||||
pub language: *const c_char,
|
||||
pub use_itn: i32,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineDolphinModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineZipformerCtcModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineWenetCtcModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineOmnilingualAsrCtcModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineFunASRNanoModelConfig {
|
||||
pub encoder_adaptor: *const c_char,
|
||||
pub llm: *const c_char,
|
||||
pub embedding: *const c_char,
|
||||
pub tokenizer: *const c_char,
|
||||
pub system_prompt: *const c_char,
|
||||
pub user_prompt: *const c_char,
|
||||
pub max_new_tokens: i32,
|
||||
pub temperature: c_float,
|
||||
pub top_p: c_float,
|
||||
pub seed: i32,
|
||||
pub language: *const c_char,
|
||||
pub itn: i32,
|
||||
pub hotwords: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineMedAsrCtcModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineFireRedAsrCtcModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineQwen3ASRModelConfig {
|
||||
pub conv_frontend: *const c_char,
|
||||
pub encoder: *const c_char,
|
||||
pub decoder: *const c_char,
|
||||
pub tokenizer: *const c_char,
|
||||
pub max_total_len: i32,
|
||||
pub max_new_tokens: i32,
|
||||
pub temperature: c_float,
|
||||
pub top_p: c_float,
|
||||
pub seed: i32,
|
||||
pub hotwords: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineCohereTranscribeModelConfig {
|
||||
pub encoder: *const c_char,
|
||||
pub decoder: *const c_char,
|
||||
pub language: *const c_char,
|
||||
pub use_punct: i32,
|
||||
pub use_itn: i32,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineModelConfig {
|
||||
pub transducer: OfflineTransducerModelConfig,
|
||||
pub paraformer: OfflineParaformerModelConfig,
|
||||
pub nemo_ctc: OfflineNemoEncDecCtcModelConfig,
|
||||
pub whisper: OfflineWhisperModelConfig,
|
||||
pub tdnn: OfflineTdnnModelConfig,
|
||||
|
||||
pub tokens: *const c_char,
|
||||
pub num_threads: i32,
|
||||
pub debug: i32,
|
||||
pub provider: *const c_char,
|
||||
pub model_type: *const c_char,
|
||||
pub modeling_unit: *const c_char,
|
||||
pub bpe_vocab: *const c_char,
|
||||
pub telespeech_ctc: *const c_char,
|
||||
|
||||
pub sense_voice: OfflineSenseVoiceModelConfig,
|
||||
pub moonshine: OfflineMoonshineModelConfig,
|
||||
pub fire_red_asr: OfflineFireRedAsrModelConfig,
|
||||
pub dolphin: OfflineDolphinModelConfig,
|
||||
pub zipformer_ctc: OfflineZipformerCtcModelConfig,
|
||||
pub canary: OfflineCanaryModelConfig,
|
||||
pub wenet_ctc: OfflineWenetCtcModelConfig,
|
||||
pub omnilingual: OfflineOmnilingualAsrCtcModelConfig,
|
||||
pub medasr: OfflineMedAsrCtcModelConfig,
|
||||
pub funasr_nano: OfflineFunASRNanoModelConfig,
|
||||
pub fire_red_asr_ctc: OfflineFireRedAsrCtcModelConfig,
|
||||
pub qwen3_asr: OfflineQwen3ASRModelConfig,
|
||||
pub cohere_transcribe: OfflineCohereTranscribeModelConfig,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineRecognizerConfig {
|
||||
pub feat_config: super::online_asr::FeatureConfig,
|
||||
pub model_config: OfflineModelConfig,
|
||||
pub lm_config: OfflineLMConfig,
|
||||
|
||||
pub decoding_method: *const c_char,
|
||||
pub max_active_paths: i32,
|
||||
pub hotwords_file: *const c_char,
|
||||
pub hotwords_score: c_float,
|
||||
pub rule_fsts: *const c_char,
|
||||
pub rule_fars: *const c_char,
|
||||
pub blank_penalty: c_float,
|
||||
pub hr: super::online_asr::HomophoneReplacerConfig,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct OfflineRecognizer {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct OfflineStream {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
pub fn SherpaOnnxCreateOfflineRecognizer(
|
||||
config: *const OfflineRecognizerConfig,
|
||||
) -> *const OfflineRecognizer;
|
||||
|
||||
pub fn SherpaOnnxDestroyOfflineRecognizer(recognizer: *const OfflineRecognizer);
|
||||
|
||||
pub fn SherpaOnnxCreateOfflineStream(
|
||||
recognizer: *const OfflineRecognizer,
|
||||
) -> *const OfflineStream;
|
||||
|
||||
pub fn SherpaOnnxCreateOfflineStreamWithHotwords(
|
||||
recognizer: *const OfflineRecognizer,
|
||||
hotwords: *const c_char,
|
||||
) -> *const OfflineStream;
|
||||
|
||||
pub fn SherpaOnnxDestroyOfflineStream(stream: *const OfflineStream);
|
||||
|
||||
pub fn SherpaOnnxAcceptWaveformOffline(
|
||||
stream: *const OfflineStream,
|
||||
sample_rate: i32,
|
||||
samples: *const f32,
|
||||
n: i32,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxOfflineStreamSetOption(
|
||||
stream: *const OfflineStream,
|
||||
key: *const c_char,
|
||||
value: *const c_char,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxOfflineStreamGetOption(
|
||||
stream: *const OfflineStream,
|
||||
key: *const c_char,
|
||||
) -> *const c_char;
|
||||
|
||||
pub fn SherpaOnnxOfflineStreamHasOption(
|
||||
stream: *const OfflineStream,
|
||||
key: *const c_char,
|
||||
) -> i32;
|
||||
|
||||
pub fn SherpaOnnxDecodeOfflineStream(
|
||||
recognizer: *const OfflineRecognizer,
|
||||
stream: *const OfflineStream,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxDecodeMultipleOfflineStreams(
|
||||
recognizer: *const OfflineRecognizer,
|
||||
streams: *const *const OfflineStream,
|
||||
n: i32,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxGetOfflineStreamResultAsJson(stream: *const OfflineStream) -> *const c_char;
|
||||
|
||||
pub fn SherpaOnnxDestroyOfflineStreamResultJson(s: *const c_char);
|
||||
}
|
||||
@@ -0,0 +1,40 @@
|
||||
#![allow(non_camel_case_types)]
|
||||
#![allow(non_snake_case)]
|
||||
#![allow(non_upper_case_globals)]
|
||||
|
||||
use std::os::raw::c_char;
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflinePunctuationModelConfig {
|
||||
pub ct_transformer: *const c_char,
|
||||
pub num_threads: i32,
|
||||
pub debug: i32,
|
||||
pub provider: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflinePunctuationConfig {
|
||||
pub model: OfflinePunctuationModelConfig,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct OfflinePunctuation {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
pub fn SherpaOnnxCreateOfflinePunctuation(
|
||||
config: *const OfflinePunctuationConfig,
|
||||
) -> *const OfflinePunctuation;
|
||||
|
||||
pub fn SherpaOnnxDestroyOfflinePunctuation(punct: *const OfflinePunctuation);
|
||||
|
||||
pub fn SherpaOfflinePunctuationAddPunct(
|
||||
punct: *const OfflinePunctuation,
|
||||
text: *const c_char,
|
||||
) -> *const c_char;
|
||||
|
||||
pub fn SherpaOfflinePunctuationFreeText(text: *const c_char);
|
||||
}
|
||||
@@ -0,0 +1,98 @@
|
||||
#![allow(non_camel_case_types)]
|
||||
#![allow(non_snake_case)]
|
||||
#![allow(non_upper_case_globals)]
|
||||
|
||||
use std::os::raw::{c_char, c_float};
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineSpeakerSegmentationPyannoteModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineSpeakerSegmentationModelConfig {
|
||||
pub pyannote: OfflineSpeakerSegmentationPyannoteModelConfig,
|
||||
pub num_threads: i32,
|
||||
pub debug: i32,
|
||||
pub provider: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct FastClusteringConfig {
|
||||
pub num_clusters: i32,
|
||||
pub threshold: c_float,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineSpeakerDiarizationConfig {
|
||||
pub segmentation: OfflineSpeakerSegmentationModelConfig,
|
||||
pub embedding: crate::speaker_embedding::SpeakerEmbeddingExtractorConfig,
|
||||
pub clustering: FastClusteringConfig,
|
||||
pub min_duration_on: c_float,
|
||||
pub min_duration_off: c_float,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct OfflineSpeakerDiarization {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct OfflineSpeakerDiarizationResult {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineSpeakerDiarizationSegment {
|
||||
pub start: c_float,
|
||||
pub end: c_float,
|
||||
pub speaker: i32,
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
pub fn SherpaOnnxCreateOfflineSpeakerDiarization(
|
||||
config: *const OfflineSpeakerDiarizationConfig,
|
||||
) -> *const OfflineSpeakerDiarization;
|
||||
|
||||
pub fn SherpaOnnxDestroyOfflineSpeakerDiarization(sd: *const OfflineSpeakerDiarization);
|
||||
|
||||
pub fn SherpaOnnxOfflineSpeakerDiarizationGetSampleRate(
|
||||
sd: *const OfflineSpeakerDiarization,
|
||||
) -> i32;
|
||||
|
||||
pub fn SherpaOnnxOfflineSpeakerDiarizationSetConfig(
|
||||
sd: *const OfflineSpeakerDiarization,
|
||||
config: *const OfflineSpeakerDiarizationConfig,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxOfflineSpeakerDiarizationResultGetNumSpeakers(
|
||||
r: *const OfflineSpeakerDiarizationResult,
|
||||
) -> i32;
|
||||
|
||||
pub fn SherpaOnnxOfflineSpeakerDiarizationResultGetNumSegments(
|
||||
r: *const OfflineSpeakerDiarizationResult,
|
||||
) -> i32;
|
||||
|
||||
pub fn SherpaOnnxOfflineSpeakerDiarizationResultSortByStartTime(
|
||||
r: *const OfflineSpeakerDiarizationResult,
|
||||
) -> *const OfflineSpeakerDiarizationSegment;
|
||||
|
||||
pub fn SherpaOnnxOfflineSpeakerDiarizationDestroySegment(
|
||||
s: *const OfflineSpeakerDiarizationSegment,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxOfflineSpeakerDiarizationProcess(
|
||||
sd: *const OfflineSpeakerDiarization,
|
||||
samples: *const c_float,
|
||||
n: i32,
|
||||
) -> *const OfflineSpeakerDiarizationResult;
|
||||
|
||||
pub fn SherpaOnnxOfflineSpeakerDiarizationDestroyResult(
|
||||
r: *const OfflineSpeakerDiarizationResult,
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,202 @@
|
||||
#![allow(non_camel_case_types)]
|
||||
#![allow(non_snake_case)]
|
||||
#![allow(non_upper_case_globals)]
|
||||
|
||||
use std::os::raw::{c_char, c_float};
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OnlineTransducerModelConfig {
|
||||
pub encoder: *const c_char,
|
||||
pub decoder: *const c_char,
|
||||
pub joiner: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OnlineParaformerModelConfig {
|
||||
pub encoder: *const c_char,
|
||||
pub decoder: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OnlineZipformer2CtcModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OnlineNemoCtcModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OnlineToneCtcModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OnlineModelConfig {
|
||||
pub transducer: OnlineTransducerModelConfig,
|
||||
pub paraformer: OnlineParaformerModelConfig,
|
||||
pub zipformer2_ctc: OnlineZipformer2CtcModelConfig,
|
||||
|
||||
pub tokens: *const c_char,
|
||||
pub num_threads: i32,
|
||||
pub provider: *const c_char,
|
||||
pub debug: i32,
|
||||
|
||||
pub model_type: *const c_char,
|
||||
|
||||
// cjkchar | bpe | cjkchar+bpe
|
||||
pub modeling_unit: *const c_char,
|
||||
|
||||
pub bpe_vocab: *const c_char,
|
||||
|
||||
pub tokens_buf: *const u8,
|
||||
pub tokens_buf_size: i32,
|
||||
|
||||
pub nemo_ctc: OnlineNemoCtcModelConfig,
|
||||
pub t_one_ctc: OnlineToneCtcModelConfig,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct FeatureConfig {
|
||||
pub sample_rate: i32,
|
||||
pub feature_dim: i32,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OnlineCtcFstDecoderConfig {
|
||||
pub graph: *const c_char,
|
||||
pub max_active: i32,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct HomophoneReplacerConfig {
|
||||
pub dict_dir: *const c_char,
|
||||
pub lexicon: *const c_char,
|
||||
pub rule_fsts: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OnlineRecognizerConfig {
|
||||
pub feat_config: FeatureConfig,
|
||||
pub model_config: OnlineModelConfig,
|
||||
|
||||
// greedy_search | modified_beam_search
|
||||
pub decoding_method: *const c_char,
|
||||
|
||||
pub max_active_paths: i32,
|
||||
|
||||
pub enable_endpoint: i32,
|
||||
|
||||
pub rule1_min_trailing_silence: c_float,
|
||||
pub rule2_min_trailing_silence: c_float,
|
||||
pub rule3_min_utterance_length: c_float,
|
||||
|
||||
pub hotwords_file: *const c_char,
|
||||
pub hotwords_score: c_float,
|
||||
|
||||
pub ctc_fst_decoder_config: OnlineCtcFstDecoderConfig,
|
||||
|
||||
pub rule_fsts: *const c_char,
|
||||
pub rule_fars: *const c_char,
|
||||
|
||||
pub blank_penalty: c_float,
|
||||
|
||||
pub hotwords_buf: *const u8,
|
||||
pub hotwords_buf_size: i32,
|
||||
|
||||
pub hr: HomophoneReplacerConfig,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct OnlineRecognizer {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct OnlineStream {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
pub fn SherpaOnnxCreateOnlineRecognizer(
|
||||
config: *const OnlineRecognizerConfig,
|
||||
) -> *const OnlineRecognizer;
|
||||
|
||||
pub fn SherpaOnnxDestroyOnlineRecognizer(recognizer: *const OnlineRecognizer);
|
||||
|
||||
pub fn SherpaOnnxCreateOnlineStream(recognizer: *const OnlineRecognizer)
|
||||
-> *const OnlineStream;
|
||||
|
||||
pub fn SherpaOnnxCreateOnlineStreamWithHotwords(
|
||||
recognizer: *const OnlineRecognizer,
|
||||
hotwords: *const c_char,
|
||||
) -> *const OnlineStream;
|
||||
|
||||
pub fn SherpaOnnxDestroyOnlineStream(stream: *const OnlineStream);
|
||||
|
||||
pub fn SherpaOnnxOnlineStreamAcceptWaveform(
|
||||
stream: *const OnlineStream,
|
||||
sample_rate: i32,
|
||||
samples: *const f32,
|
||||
n: i32,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxIsOnlineStreamReady(
|
||||
recognizer: *const OnlineRecognizer,
|
||||
stream: *const OnlineStream,
|
||||
) -> i32;
|
||||
|
||||
pub fn SherpaOnnxDecodeOnlineStream(
|
||||
recognizer: *const OnlineRecognizer,
|
||||
stream: *const OnlineStream,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxDecodeMultipleOnlineStreams(
|
||||
recognizer: *const OnlineRecognizer,
|
||||
streams: *const *const OnlineStream,
|
||||
n: i32,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxGetOnlineStreamResultAsJson(
|
||||
recognizer: *const OnlineRecognizer,
|
||||
stream: *const OnlineStream,
|
||||
) -> *const c_char;
|
||||
|
||||
pub fn SherpaOnnxDestroyOnlineStreamResultJson(s: *const c_char);
|
||||
|
||||
pub fn SherpaOnnxOnlineStreamReset(
|
||||
recognizer: *const OnlineRecognizer,
|
||||
stream: *const OnlineStream,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxOnlineStreamInputFinished(stream: *const OnlineStream);
|
||||
|
||||
pub fn SherpaOnnxOnlineStreamSetOption(
|
||||
stream: *const OnlineStream,
|
||||
key: *const c_char,
|
||||
value: *const c_char,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxOnlineStreamGetOption(
|
||||
stream: *const OnlineStream,
|
||||
key: *const c_char,
|
||||
) -> *const c_char;
|
||||
|
||||
pub fn SherpaOnnxOnlineStreamHasOption(stream: *const OnlineStream, key: *const c_char) -> i32;
|
||||
|
||||
pub fn SherpaOnnxOnlineStreamIsEndpoint(
|
||||
recognizer: *const OnlineRecognizer,
|
||||
stream: *const OnlineStream,
|
||||
) -> i32;
|
||||
}
|
||||
@@ -0,0 +1,41 @@
|
||||
#![allow(non_camel_case_types)]
|
||||
#![allow(non_snake_case)]
|
||||
#![allow(non_upper_case_globals)]
|
||||
|
||||
use std::os::raw::c_char;
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OnlinePunctuationModelConfig {
|
||||
pub cnn_bilstm: *const c_char,
|
||||
pub bpe_vocab: *const c_char,
|
||||
pub num_threads: i32,
|
||||
pub debug: i32,
|
||||
pub provider: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OnlinePunctuationConfig {
|
||||
pub model: OnlinePunctuationModelConfig,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct OnlinePunctuation {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
pub fn SherpaOnnxCreateOnlinePunctuation(
|
||||
config: *const OnlinePunctuationConfig,
|
||||
) -> *const OnlinePunctuation;
|
||||
|
||||
pub fn SherpaOnnxDestroyOnlinePunctuation(punctuation: *const OnlinePunctuation);
|
||||
|
||||
pub fn SherpaOnnxOnlinePunctuationAddPunct(
|
||||
punctuation: *const OnlinePunctuation,
|
||||
text: *const c_char,
|
||||
) -> *const c_char;
|
||||
|
||||
pub fn SherpaOnnxOnlinePunctuationFreeText(text: *const c_char);
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
use std::os::raw::c_float;
|
||||
|
||||
#[repr(C)]
|
||||
pub struct SherpaOnnxLinearResampler {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct SherpaOnnxResampleOut {
|
||||
pub samples: *const f32,
|
||||
pub n: i32,
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
pub fn SherpaOnnxCreateLinearResampler(
|
||||
samp_rate_in_hz: i32,
|
||||
samp_rate_out_hz: i32,
|
||||
filter_cutoff_hz: c_float,
|
||||
num_zeros: i32,
|
||||
) -> *const SherpaOnnxLinearResampler;
|
||||
|
||||
pub fn SherpaOnnxDestroyLinearResampler(p: *const SherpaOnnxLinearResampler);
|
||||
|
||||
pub fn SherpaOnnxLinearResamplerReset(p: *const SherpaOnnxLinearResampler);
|
||||
|
||||
pub fn SherpaOnnxLinearResamplerResample(
|
||||
p: *const SherpaOnnxLinearResampler,
|
||||
input: *const f32,
|
||||
input_dim: i32,
|
||||
flush: i32,
|
||||
) -> *const SherpaOnnxResampleOut;
|
||||
|
||||
pub fn SherpaOnnxLinearResamplerResampleFree(p: *const SherpaOnnxResampleOut);
|
||||
|
||||
pub fn SherpaOnnxLinearResamplerResampleGetInputSampleRate(
|
||||
p: *const SherpaOnnxLinearResampler,
|
||||
) -> i32;
|
||||
|
||||
pub fn SherpaOnnxLinearResamplerResampleGetOutputSampleRate(
|
||||
p: *const SherpaOnnxLinearResampler,
|
||||
) -> i32;
|
||||
}
|
||||
@@ -0,0 +1,131 @@
|
||||
#![allow(non_camel_case_types)]
|
||||
#![allow(non_snake_case)]
|
||||
#![allow(non_upper_case_globals)]
|
||||
|
||||
use std::os::raw::{c_char, c_float};
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct SpeakerEmbeddingExtractorConfig {
|
||||
pub model: *const c_char,
|
||||
pub num_threads: i32,
|
||||
pub debug: i32,
|
||||
pub provider: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct SpeakerEmbeddingExtractor {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct SpeakerEmbeddingManager {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct SpeakerEmbeddingManagerSpeakerMatch {
|
||||
pub score: c_float,
|
||||
pub name: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct SpeakerEmbeddingManagerBestMatchesResult {
|
||||
pub matches: *const SpeakerEmbeddingManagerSpeakerMatch,
|
||||
pub count: i32,
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
pub fn SherpaOnnxCreateSpeakerEmbeddingExtractor(
|
||||
config: *const SpeakerEmbeddingExtractorConfig,
|
||||
) -> *const SpeakerEmbeddingExtractor;
|
||||
|
||||
pub fn SherpaOnnxDestroySpeakerEmbeddingExtractor(p: *const SpeakerEmbeddingExtractor);
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingExtractorDim(p: *const SpeakerEmbeddingExtractor) -> i32;
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingExtractorCreateStream(
|
||||
p: *const SpeakerEmbeddingExtractor,
|
||||
) -> *const crate::online_asr::OnlineStream;
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingExtractorIsReady(
|
||||
p: *const SpeakerEmbeddingExtractor,
|
||||
s: *const crate::online_asr::OnlineStream,
|
||||
) -> i32;
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingExtractorComputeEmbedding(
|
||||
p: *const SpeakerEmbeddingExtractor,
|
||||
s: *const crate::online_asr::OnlineStream,
|
||||
) -> *const c_float;
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingExtractorDestroyEmbedding(v: *const c_float);
|
||||
|
||||
pub fn SherpaOnnxCreateSpeakerEmbeddingManager(dim: i32) -> *const SpeakerEmbeddingManager;
|
||||
|
||||
pub fn SherpaOnnxDestroySpeakerEmbeddingManager(p: *const SpeakerEmbeddingManager);
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingManagerAdd(
|
||||
p: *const SpeakerEmbeddingManager,
|
||||
name: *const c_char,
|
||||
v: *const c_float,
|
||||
) -> i32;
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingManagerAddList(
|
||||
p: *const SpeakerEmbeddingManager,
|
||||
name: *const c_char,
|
||||
v: *const *const c_float,
|
||||
) -> i32;
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingManagerAddListFlattened(
|
||||
p: *const SpeakerEmbeddingManager,
|
||||
name: *const c_char,
|
||||
v: *const c_float,
|
||||
n: i32,
|
||||
) -> i32;
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingManagerRemove(
|
||||
p: *const SpeakerEmbeddingManager,
|
||||
name: *const c_char,
|
||||
) -> i32;
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingManagerSearch(
|
||||
p: *const SpeakerEmbeddingManager,
|
||||
v: *const c_float,
|
||||
threshold: c_float,
|
||||
) -> *const c_char;
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingManagerFreeSearch(name: *const c_char);
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingManagerGetBestMatches(
|
||||
p: *const SpeakerEmbeddingManager,
|
||||
v: *const c_float,
|
||||
threshold: c_float,
|
||||
n: i32,
|
||||
) -> *const SpeakerEmbeddingManagerBestMatchesResult;
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingManagerFreeBestMatches(
|
||||
r: *const SpeakerEmbeddingManagerBestMatchesResult,
|
||||
);
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingManagerVerify(
|
||||
p: *const SpeakerEmbeddingManager,
|
||||
name: *const c_char,
|
||||
v: *const c_float,
|
||||
threshold: c_float,
|
||||
) -> i32;
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingManagerContains(
|
||||
p: *const SpeakerEmbeddingManager,
|
||||
name: *const c_char,
|
||||
) -> i32;
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingManagerNumSpeakers(p: *const SpeakerEmbeddingManager) -> i32;
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingManagerGetAllSpeakers(
|
||||
p: *const SpeakerEmbeddingManager,
|
||||
) -> *const *const c_char;
|
||||
|
||||
pub fn SherpaOnnxSpeakerEmbeddingManagerFreeAllSpeakers(names: *const *const c_char);
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
use std::os::raw::c_char;
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineSpeechDenoiserGtcrnModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineSpeechDenoiserDpdfNetModelConfig {
|
||||
pub model: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineSpeechDenoiserModelConfig {
|
||||
pub gtcrn: OfflineSpeechDenoiserGtcrnModelConfig,
|
||||
pub num_threads: i32,
|
||||
pub debug: i32,
|
||||
pub provider: *const c_char,
|
||||
pub dpdfnet: OfflineSpeechDenoiserDpdfNetModelConfig,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineSpeechDenoiserConfig {
|
||||
pub model: OfflineSpeechDenoiserModelConfig,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OnlineSpeechDenoiserConfig {
|
||||
pub model: OfflineSpeechDenoiserModelConfig,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct DenoisedAudio {
|
||||
pub samples: *const f32,
|
||||
pub n: i32,
|
||||
pub sample_rate: i32,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct OfflineSpeechDenoiser {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct OnlineSpeechDenoiser {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
pub fn SherpaOnnxCreateOfflineSpeechDenoiser(
|
||||
config: *const OfflineSpeechDenoiserConfig,
|
||||
) -> *const OfflineSpeechDenoiser;
|
||||
pub fn SherpaOnnxDestroyOfflineSpeechDenoiser(p: *const OfflineSpeechDenoiser);
|
||||
pub fn SherpaOnnxOfflineSpeechDenoiserGetSampleRate(p: *const OfflineSpeechDenoiser) -> i32;
|
||||
pub fn SherpaOnnxOfflineSpeechDenoiserRun(
|
||||
p: *const OfflineSpeechDenoiser,
|
||||
samples: *const f32,
|
||||
n: i32,
|
||||
sample_rate: i32,
|
||||
) -> *const DenoisedAudio;
|
||||
|
||||
pub fn SherpaOnnxCreateOnlineSpeechDenoiser(
|
||||
config: *const OnlineSpeechDenoiserConfig,
|
||||
) -> *const OnlineSpeechDenoiser;
|
||||
pub fn SherpaOnnxDestroyOnlineSpeechDenoiser(p: *const OnlineSpeechDenoiser);
|
||||
pub fn SherpaOnnxOnlineSpeechDenoiserGetSampleRate(p: *const OnlineSpeechDenoiser) -> i32;
|
||||
pub fn SherpaOnnxOnlineSpeechDenoiserGetFrameShiftInSamples(
|
||||
p: *const OnlineSpeechDenoiser,
|
||||
) -> i32;
|
||||
pub fn SherpaOnnxOnlineSpeechDenoiserRun(
|
||||
p: *const OnlineSpeechDenoiser,
|
||||
samples: *const f32,
|
||||
n: i32,
|
||||
sample_rate: i32,
|
||||
) -> *const DenoisedAudio;
|
||||
pub fn SherpaOnnxOnlineSpeechDenoiserFlush(
|
||||
p: *const OnlineSpeechDenoiser,
|
||||
) -> *const DenoisedAudio;
|
||||
pub fn SherpaOnnxOnlineSpeechDenoiserReset(p: *const OnlineSpeechDenoiser);
|
||||
|
||||
pub fn SherpaOnnxDestroyDenoisedAudio(audio: *const DenoisedAudio);
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
#![allow(non_camel_case_types)]
|
||||
#![allow(non_snake_case)]
|
||||
#![allow(non_upper_case_globals)]
|
||||
|
||||
use std::os::raw::c_char;
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct SpokenLanguageIdentificationWhisperConfig {
|
||||
pub encoder: *const c_char,
|
||||
pub decoder: *const c_char,
|
||||
pub tail_paddings: i32,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct SpokenLanguageIdentificationConfig {
|
||||
pub whisper: SpokenLanguageIdentificationWhisperConfig,
|
||||
pub num_threads: i32,
|
||||
pub debug: i32,
|
||||
pub provider: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct SpokenLanguageIdentification {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct SpokenLanguageIdentificationResult {
|
||||
pub lang: *const c_char,
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
pub fn SherpaOnnxCreateSpokenLanguageIdentification(
|
||||
config: *const SpokenLanguageIdentificationConfig,
|
||||
) -> *const SpokenLanguageIdentification;
|
||||
|
||||
pub fn SherpaOnnxDestroySpokenLanguageIdentification(slid: *const SpokenLanguageIdentification);
|
||||
|
||||
pub fn SherpaOnnxSpokenLanguageIdentificationCreateOfflineStream(
|
||||
slid: *const SpokenLanguageIdentification,
|
||||
) -> *const super::offline_asr::OfflineStream;
|
||||
|
||||
pub fn SherpaOnnxSpokenLanguageIdentificationCompute(
|
||||
slid: *const SpokenLanguageIdentification,
|
||||
stream: *const super::offline_asr::OfflineStream,
|
||||
) -> *const SpokenLanguageIdentificationResult;
|
||||
|
||||
pub fn SherpaOnnxDestroySpokenLanguageIdentificationResult(
|
||||
r: *const SpokenLanguageIdentificationResult,
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,172 @@
|
||||
#![allow(non_camel_case_types)]
|
||||
#![allow(non_snake_case)]
|
||||
#![allow(non_upper_case_globals)]
|
||||
|
||||
use std::os::raw::{c_char, c_float, c_void};
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineTtsVitsModelConfig {
|
||||
pub model: *const c_char,
|
||||
pub lexicon: *const c_char,
|
||||
pub tokens: *const c_char,
|
||||
pub data_dir: *const c_char,
|
||||
pub noise_scale: c_float,
|
||||
pub noise_scale_w: c_float,
|
||||
pub length_scale: c_float,
|
||||
pub dict_dir: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineTtsMatchaModelConfig {
|
||||
pub acoustic_model: *const c_char,
|
||||
pub vocoder: *const c_char,
|
||||
pub lexicon: *const c_char,
|
||||
pub tokens: *const c_char,
|
||||
pub data_dir: *const c_char,
|
||||
pub noise_scale: c_float,
|
||||
pub length_scale: c_float,
|
||||
pub dict_dir: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineTtsKokoroModelConfig {
|
||||
pub model: *const c_char,
|
||||
pub voices: *const c_char,
|
||||
pub tokens: *const c_char,
|
||||
pub data_dir: *const c_char,
|
||||
pub length_scale: c_float,
|
||||
pub dict_dir: *const c_char,
|
||||
pub lexicon: *const c_char,
|
||||
pub lang: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineTtsKittenModelConfig {
|
||||
pub model: *const c_char,
|
||||
pub voices: *const c_char,
|
||||
pub tokens: *const c_char,
|
||||
pub data_dir: *const c_char,
|
||||
pub length_scale: c_float,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineTtsZipvoiceModelConfig {
|
||||
pub tokens: *const c_char,
|
||||
pub encoder: *const c_char,
|
||||
pub decoder: *const c_char,
|
||||
pub vocoder: *const c_char,
|
||||
pub data_dir: *const c_char,
|
||||
pub lexicon: *const c_char,
|
||||
pub feat_scale: c_float,
|
||||
pub t_shift: c_float,
|
||||
pub target_rms: c_float,
|
||||
pub guidance_scale: c_float,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineTtsPocketModelConfig {
|
||||
pub lm_flow: *const c_char,
|
||||
pub lm_main: *const c_char,
|
||||
pub encoder: *const c_char,
|
||||
pub decoder: *const c_char,
|
||||
pub text_conditioner: *const c_char,
|
||||
pub vocab_json: *const c_char,
|
||||
pub token_scores_json: *const c_char,
|
||||
pub voice_embedding_cache_capacity: i32,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineTtsSupertonicModelConfig {
|
||||
pub duration_predictor: *const c_char,
|
||||
pub text_encoder: *const c_char,
|
||||
pub vector_estimator: *const c_char,
|
||||
pub vocoder: *const c_char,
|
||||
pub tts_json: *const c_char,
|
||||
pub unicode_indexer: *const c_char,
|
||||
pub voice_style: *const c_char,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineTtsModelConfig {
|
||||
pub vits: OfflineTtsVitsModelConfig,
|
||||
pub num_threads: i32,
|
||||
pub debug: i32,
|
||||
pub provider: *const c_char,
|
||||
pub matcha: OfflineTtsMatchaModelConfig,
|
||||
pub kokoro: OfflineTtsKokoroModelConfig,
|
||||
pub kitten: OfflineTtsKittenModelConfig,
|
||||
pub zipvoice: OfflineTtsZipvoiceModelConfig,
|
||||
pub pocket: OfflineTtsPocketModelConfig,
|
||||
pub supertonic: OfflineTtsSupertonicModelConfig,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct OfflineTtsConfig {
|
||||
pub model: OfflineTtsModelConfig,
|
||||
pub rule_fsts: *const c_char,
|
||||
pub max_num_sentences: i32,
|
||||
pub rule_fars: *const c_char,
|
||||
pub silence_scale: c_float,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct SherpaOnnxGeneratedAudio {
|
||||
pub samples: *const f32,
|
||||
pub n: i32,
|
||||
pub sample_rate: i32,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct SherpaOnnxGenerationConfig {
|
||||
pub silence_scale: c_float,
|
||||
pub speed: c_float,
|
||||
pub sid: i32,
|
||||
pub reference_audio: *const f32,
|
||||
pub reference_audio_len: i32,
|
||||
pub reference_sample_rate: i32,
|
||||
pub reference_text: *const c_char,
|
||||
pub num_steps: i32,
|
||||
pub extra: *const c_char,
|
||||
}
|
||||
|
||||
pub type SherpaOnnxGeneratedAudioProgressCallbackWithArg = Option<
|
||||
unsafe extern "C" fn(samples: *const f32, n: i32, progress: c_float, arg: *mut c_void) -> i32,
|
||||
>;
|
||||
|
||||
#[repr(C)]
|
||||
pub struct SherpaOnnxOfflineTts {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
pub fn SherpaOnnxCreateOfflineTts(
|
||||
config: *const OfflineTtsConfig,
|
||||
) -> *const SherpaOnnxOfflineTts;
|
||||
|
||||
pub fn SherpaOnnxDestroyOfflineTts(tts: *const SherpaOnnxOfflineTts);
|
||||
|
||||
pub fn SherpaOnnxOfflineTtsSampleRate(tts: *const SherpaOnnxOfflineTts) -> i32;
|
||||
|
||||
pub fn SherpaOnnxOfflineTtsNumSpeakers(tts: *const SherpaOnnxOfflineTts) -> i32;
|
||||
|
||||
pub fn SherpaOnnxOfflineTtsGenerateWithConfig(
|
||||
tts: *const SherpaOnnxOfflineTts,
|
||||
text: *const c_char,
|
||||
config: *const SherpaOnnxGenerationConfig,
|
||||
callback: SherpaOnnxGeneratedAudioProgressCallbackWithArg,
|
||||
arg: *mut c_void,
|
||||
) -> *const SherpaOnnxGeneratedAudio;
|
||||
|
||||
pub fn SherpaOnnxDestroyOfflineTtsGeneratedAudio(p: *const SherpaOnnxGeneratedAudio);
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
use std::os::raw::{c_char, c_float};
|
||||
|
||||
#[repr(C)]
|
||||
pub struct SileroVadModelConfig {
|
||||
pub model: *const c_char,
|
||||
pub threshold: c_float,
|
||||
pub min_silence_duration: c_float,
|
||||
pub min_speech_duration: c_float,
|
||||
pub window_size: i32,
|
||||
pub max_speech_duration: c_float,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct TenVadModelConfig {
|
||||
pub model: *const c_char,
|
||||
pub threshold: c_float,
|
||||
pub min_silence_duration: c_float,
|
||||
pub min_speech_duration: c_float,
|
||||
pub window_size: i32,
|
||||
pub max_speech_duration: c_float,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct VadModelConfig {
|
||||
pub silero_vad: SileroVadModelConfig,
|
||||
pub sample_rate: i32,
|
||||
pub num_threads: i32,
|
||||
pub provider: *const c_char,
|
||||
pub debug: i32,
|
||||
pub ten_vad: TenVadModelConfig,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct CircularBuffer {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct SpeechSegment {
|
||||
pub start: i32,
|
||||
pub samples: *mut f32,
|
||||
pub n: i32,
|
||||
}
|
||||
|
||||
#[repr(C)]
|
||||
pub struct VoiceActivityDetector {
|
||||
_private: [u8; 0],
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
pub fn SherpaOnnxCreateCircularBuffer(capacity: i32) -> *const CircularBuffer;
|
||||
pub fn SherpaOnnxDestroyCircularBuffer(buffer: *const CircularBuffer);
|
||||
pub fn SherpaOnnxCircularBufferPush(buffer: *const CircularBuffer, p: *const f32, n: i32);
|
||||
pub fn SherpaOnnxCircularBufferGet(
|
||||
buffer: *const CircularBuffer,
|
||||
start_index: i32,
|
||||
n: i32,
|
||||
) -> *const f32;
|
||||
pub fn SherpaOnnxCircularBufferFree(p: *const f32);
|
||||
pub fn SherpaOnnxCircularBufferPop(buffer: *const CircularBuffer, n: i32);
|
||||
pub fn SherpaOnnxCircularBufferSize(buffer: *const CircularBuffer) -> i32;
|
||||
pub fn SherpaOnnxCircularBufferHead(buffer: *const CircularBuffer) -> i32;
|
||||
pub fn SherpaOnnxCircularBufferReset(buffer: *const CircularBuffer);
|
||||
|
||||
pub fn SherpaOnnxCreateVoiceActivityDetector(
|
||||
config: *const VadModelConfig,
|
||||
buffer_size_in_seconds: c_float,
|
||||
) -> *const VoiceActivityDetector;
|
||||
pub fn SherpaOnnxDestroyVoiceActivityDetector(p: *const VoiceActivityDetector);
|
||||
pub fn SherpaOnnxVoiceActivityDetectorAcceptWaveform(
|
||||
p: *const VoiceActivityDetector,
|
||||
samples: *const f32,
|
||||
n: i32,
|
||||
);
|
||||
pub fn SherpaOnnxVoiceActivityDetectorEmpty(p: *const VoiceActivityDetector) -> i32;
|
||||
pub fn SherpaOnnxVoiceActivityDetectorDetected(p: *const VoiceActivityDetector) -> i32;
|
||||
pub fn SherpaOnnxVoiceActivityDetectorPop(p: *const VoiceActivityDetector);
|
||||
pub fn SherpaOnnxVoiceActivityDetectorClear(p: *const VoiceActivityDetector);
|
||||
pub fn SherpaOnnxVoiceActivityDetectorFront(
|
||||
p: *const VoiceActivityDetector,
|
||||
) -> *const SpeechSegment;
|
||||
pub fn SherpaOnnxDestroySpeechSegment(p: *const SpeechSegment);
|
||||
pub fn SherpaOnnxVoiceActivityDetectorReset(p: *const VoiceActivityDetector);
|
||||
pub fn SherpaOnnxVoiceActivityDetectorFlush(p: *const VoiceActivityDetector);
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
#![allow(non_camel_case_types)]
|
||||
#![allow(non_snake_case)]
|
||||
#![allow(non_upper_case_globals)]
|
||||
|
||||
use std::os::raw::c_char;
|
||||
|
||||
#[repr(C)]
|
||||
#[derive(Debug, Copy, Clone)]
|
||||
pub struct SherpaOnnxWave {
|
||||
/// Samples normalized to [-1, 1]
|
||||
pub samples: *const f32,
|
||||
pub sample_rate: i32,
|
||||
pub num_samples: i32,
|
||||
}
|
||||
|
||||
extern "C" {
|
||||
/// Read a WAV file. Returns NULL on error.
|
||||
pub fn SherpaOnnxReadWave(filename: *const c_char) -> *const SherpaOnnxWave;
|
||||
|
||||
/// Free memory allocated by SherpaOnnxReadWave
|
||||
pub fn SherpaOnnxFreeWave(wave: *const SherpaOnnxWave);
|
||||
|
||||
/// Write a WAV file. Returns 1 on success, 0 on failure.
|
||||
pub fn SherpaOnnxWriteWave(
|
||||
samples: *const f32,
|
||||
n: i32,
|
||||
sample_rate: i32,
|
||||
filename: *const c_char,
|
||||
) -> i32;
|
||||
}
|
||||
Reference in New Issue
Block a user