From c104eecfb38620de2c35c7e20a716f8658b5a6b1 Mon Sep 17 00:00:00 2001 From: John Matthew Tennant Date: Fri, 31 Jul 2026 11:15:09 -0400 Subject: [PATCH] feat(desktop): import local Pocket voices (#3259) ## Context Pocket TTS currently offers bundled reference voices. People also need a local, private way to add a voice without sending audio to a cloud service. ## Summary Add a Pocket voice import flow to Voice settings. Buzz opens the native file picker, decodes common audio formats in the reusable `buzz-voice` crate, canonicalizes the selected audio, stores it under a content-derived identity in app data, selects it, and lets the user delete it later. ## Changes - Accept WAV, M4A, MP3, FLAC, OGG, and AIFF files between 2 and 30 seconds, including multichannel sources. - Decode and downmix accepted audio to canonical mono 32 kHz PCM16 WAV before hashing and storage. - Store imported voices behind stable `pocket:imported:` identities and content-addressed files. - Keep absolute file paths inside the native process and expose only voice metadata to React. - Include imported voices in Pocket preview and live huddle playback. - Add Add voice and delete controls while preserving the bundled Pocket voice catalog. - Fall back to Mary when the selected imported voice is deleted. - Keep durable import, selection, and deletion successful when a live TTS worker acknowledgement is delayed. - Preserve bundled voices when optional import metadata is unreadable and keep failed deletion retryable. ## Related issue None found. ## Testing Production decoding was exercised with WAV, M4A with AAC, MP3, FLAC, OGG Vorbis, and AIFF fixtures. Each format canonicalized to mono 32 kHz PCM16 WAV. Manual validation in the combined daily-driver build covered native-picker import, Preview, live-huddle playback, deletion, and Mary fallback. ## Screenshots The Voice settings card preserves the bundled Pocket catalog and adds the local Add voice action. ![Pocket TTS voice import](https://raw.githubusercontent.com/block/buzz/c03ba29060ca544c5ac3394c212f376651b386a3/pr-3259--pocket-voices.png) ## Reviewer-reproducible examples Create common-format fixtures and run them through the production importer: ```bash . ./bin/activate-hermit fixtures="$(mktemp -d)" ffmpeg -hide_banner -loglevel error -f lavfi -i "sine=frequency=220:duration=3" -ac 2 -ar 44100 "$fixtures/voice.wav" ffmpeg -hide_banner -loglevel error -i "$fixtures/voice.wav" -c:a aac "$fixtures/voice.m4a" ffmpeg -hide_banner -loglevel error -i "$fixtures/voice.wav" "$fixtures/voice.mp3" ffmpeg -hide_banner -loglevel error -i "$fixtures/voice.wav" "$fixtures/voice.flac" ffmpeg -hide_banner -loglevel error -i "$fixtures/voice.wav" -c:a libvorbis "$fixtures/voice.ogg" ffmpeg -hide_banner -loglevel error -i "$fixtures/voice.wav" -c:a pcm_s16be "$fixtures/voice.aiff" BUZZ_VOICE_IMPORT_TEST_DIR="$fixtures" \ cargo test -p buzz-voice imports_common_audio_format_fixtures -- --ignored --nocapture ``` Exercise import persistence, synthesis, deletion, and bundled-voice fallback with an installed Pocket model: ```bash BUZZ_POCKET_MODEL_DIR=/path/to/pocket-model-bundle \ cargo test -p buzz-voice --test pocket_import_audio \ objective_import_synthesis_delete_and_mary_fallback \ -- --ignored --nocapture ``` Exercise the native-picker boundary, selection, preview dispatch, deletion, cancellation, and invalid-file states: ```bash cd desktop pnpm build:e2e pnpm exec playwright test tests/e2e/voice-settings.spec.ts --project=smoke ``` --------- Signed-off-by: John Tennant Signed-off-by: John Tennant Signed-off-by: John Tennant Signed-off-by: npub1qyvc0c5kl4gqv2fd97fsk46tu378sqgy35vc83rvgfwne90sel7s0ed67d <011987e296fd5006292d2f930b574be47c7801048d1983c46c425d3c95f0cffd@buzz.block.builderlab.xyz> Signed-off-by: npub12gtutshhh76rx0jx697f32f9tffd4hhp3hx58fp4x6u4uemkm7sqf8f757 <5217c5c2f7bfb4333e46d17c98a9255a52dadee18dcd43a43536b95e6776dfa0@buzz.block.builderlab.xyz> Co-authored-by: John Tennant Co-authored-by: npub1qyvc0c5kl4gqv2fd97fsk46tu378sqgy35vc83rvgfwne90sel7s0ed67d <011987e296fd5006292d2f930b574be47c7801048d1983c46c425d3c95f0cffd@buzz.block.builderlab.xyz> Co-authored-by: npub12gtutshhh76rx0jx697f32f9tffd4hhp3hx58fp4x6u4uemkm7sqf8f757 <5217c5c2f7bfb4333e46d17c98a9255a52dadee18dcd43a43536b95e6776dfa0@buzz.block.builderlab.xyz> --- Cargo.lock | 191 +++++ crates/buzz-voice/Cargo.toml | 7 + crates/buzz-voice/src/imported.rs | 730 ++++++++++++++++++ crates/buzz-voice/src/lib.rs | 1 + .../buzz-voice/tests/pocket_import_audio.rs | 133 ++++ desktop/src-tauri/Cargo.lock | 15 + desktop/src-tauri/src/huddle/mod.rs | 1 + desktop/src-tauri/src/huddle/pipeline.rs | 63 +- desktop/src-tauri/src/huddle/tts_settings.rs | 212 ++++- .../src-tauri/src/huddle/tts_voice_import.rs | 59 ++ .../src/huddle/tts_voice_transition.rs | 11 +- desktop/src-tauri/src/lib.rs | 2 + .../settings/ui/VoiceSettingsCard.tsx | 120 ++- desktop/src/testing/e2eBridge.ts | 296 ++++--- desktop/tests/e2e/voice-settings.spec.ts | 93 +++ desktop/tests/helpers/bridge.ts | 2 + 16 files changed, 1778 insertions(+), 158 deletions(-) create mode 100644 crates/buzz-voice/src/imported.rs create mode 100644 crates/buzz-voice/tests/pocket_import_audio.rs create mode 100644 desktop/src-tauri/src/huddle/tts_voice_import.rs diff --git a/Cargo.lock b/Cargo.lock index 5d9a2d764..9a3f91671 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -410,6 +410,16 @@ version = "1.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1505bd5d3d116872e7271a6d4e16d81d0c8570876c8de68093a09ac269d8aac0" +[[package]] +name = "atomic-write-file" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "84790c55b5704b0d35130bf16a4ce22a8e70eb0ea773522557524d9a4852663d" +dependencies = [ + "nix 0.30.1", + "rand 0.9.4", +] + [[package]] name = "attohttpc" version = "0.30.1" @@ -1295,13 +1305,18 @@ dependencies = [ name = "buzz-voice" version = "0.1.0" dependencies = [ + "atomic-write-file", + "hex", "ort", "ort-sys", "rand 0.10.1", "sentencepiece-model", "serde", "serde_json", + "sha2 0.11.0", "sherpa-onnx", + "symphonia", + "tempfile", "tokenizers", ] @@ -2762,6 +2777,12 @@ dependencies = [ "smallvec", ] +[[package]] +name = "extended" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "af9673d8203fcb076b19dfd17e38b3d4ae9f44959416ea532ce72415a6020365" + [[package]] name = "fancy-regex" version = "0.11.0" @@ -5580,6 +5601,18 @@ dependencies = [ "memoffset", ] +[[package]] +name = "nix" +version = "0.30.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "74523f3a35e05aba87a1d978330aef40f67b0304ac79c1c00b294c9830543db6" +dependencies = [ + "bitflags 2.13.0", + "cfg-if 1.0.4", + "cfg_aliases", + "libc", +] + [[package]] name = "nix" version = "0.31.3" @@ -9016,6 +9049,164 @@ version = "0.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a7973cce6668464ea31f176d85b13c7ab3bba2cb3b77a2ed26abd7801688010a" +[[package]] +name = "symphonia" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5773a4c030a19d9bfaa090f49746ff35c75dfddfa700df7a5939d5e076a57039" +dependencies = [ + "lazy_static", + "symphonia-bundle-flac", + "symphonia-bundle-mp3", + "symphonia-codec-aac", + "symphonia-codec-alac", + "symphonia-codec-pcm", + "symphonia-codec-vorbis", + "symphonia-core", + "symphonia-format-isomp4", + "symphonia-format-ogg", + "symphonia-format-riff", + "symphonia-metadata", +] + +[[package]] +name = "symphonia-bundle-flac" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c91565e180aea25d9b80a910c546802526ffd0072d0b8974e3ebe59b686c9976" +dependencies = [ + "log", + "symphonia-core", + "symphonia-metadata", + "symphonia-utils-xiph", +] + +[[package]] +name = "symphonia-bundle-mp3" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4872dd6bb56bf5eac799e3e957aa1981086c3e613b27e0ac23b176054f7c57ed" +dependencies = [ + "lazy_static", + "log", + "symphonia-core", + "symphonia-metadata", +] + +[[package]] +name = "symphonia-codec-aac" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4c263845aa86881416849c1729a54c7f55164f8b96111dba59de46849e73a790" +dependencies = [ + "lazy_static", + "log", + "symphonia-core", +] + +[[package]] +name = "symphonia-codec-alac" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8413fa754942ac16a73634c9dfd1500ed5c61430956b33728567f667fdd393ab" +dependencies = [ + "log", + "symphonia-core", +] + +[[package]] +name = "symphonia-codec-pcm" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4e89d716c01541ad3ebe7c91ce4c8d38a7cf266a3f7b2f090b108fb0cb031d95" +dependencies = [ + "log", + "symphonia-core", +] + +[[package]] +name = "symphonia-codec-vorbis" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f025837c309cd69ffef572750b4a2257b59552c5399a5e49707cc5b1b85d1c73" +dependencies = [ + "log", + "symphonia-core", + "symphonia-utils-xiph", +] + +[[package]] +name = "symphonia-core" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ea00cc4f79b7f6bb7ff87eddc065a1066f3a43fe1875979056672c9ef948c2af" +dependencies = [ + "arrayvec", + "bitflags 1.3.2", + "bytemuck", + "lazy_static", + "log", +] + +[[package]] +name = "symphonia-format-isomp4" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "243739585d11f81daf8dac8d9f3d18cc7898f6c09a259675fc364b382c30e0a5" +dependencies = [ + "encoding_rs", + "log", + "symphonia-core", + "symphonia-metadata", + "symphonia-utils-xiph", +] + +[[package]] +name = "symphonia-format-ogg" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2b4955c67c1ed3aa8ae8428d04ca8397fbef6a19b2b051e73b5da8b1435639cb" +dependencies = [ + "log", + "symphonia-core", + "symphonia-metadata", + "symphonia-utils-xiph", +] + +[[package]] +name = "symphonia-format-riff" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c2d7c3df0e7d94efb68401d81906eae73c02b40d5ec1a141962c592d0f11a96f" +dependencies = [ + "extended", + "log", + "symphonia-core", + "symphonia-metadata", +] + +[[package]] +name = "symphonia-metadata" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "36306ff42b9ffe6e5afc99d49e121e0bd62fe79b9db7b9681d48e29fa19e6b16" +dependencies = [ + "encoding_rs", + "lazy_static", + "log", + "symphonia-core", +] + +[[package]] +name = "symphonia-utils-xiph" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ee27c85ab799a338446b68eec77abf42e1a6f1bb490656e121c6e27bfbab9f16" +dependencies = [ + "symphonia-core", + "symphonia-metadata", +] + [[package]] name = "syn" version = "1.0.109" diff --git a/crates/buzz-voice/Cargo.toml b/crates/buzz-voice/Cargo.toml index 3574c3291..beff5b4a5 100644 --- a/crates/buzz-voice/Cargo.toml +++ b/crates/buzz-voice/Cargo.toml @@ -8,11 +8,18 @@ repository.workspace = true description = "Reusable local voice primitives for Buzz" [dependencies] +atomic-write-file = "0.3" +hex = { workspace = true } ort = { version = "=2.0.0-rc.12", default-features = false, features = ["api-24", "ndarray", "std"] } ort-sys = { version = "=2.0.0-rc.12", features = ["disable-linking"] } rand = "0.10" sentencepiece-model = "0.1" serde = { version = "1", features = ["derive"] } serde_json = "1" +sha2 = { workspace = true } sherpa-onnx = "1.12" +symphonia = { version = "0.5", default-features = false, features = ["aac", "aiff", "alac", "flac", "isomp4", "mp3", "ogg", "pcm", "vorbis", "wav"] } tokenizers = { version = "0.22", default-features = false, features = ["fancy-regex"] } + +[dev-dependencies] +tempfile = "3" diff --git a/crates/buzz-voice/src/imported.rs b/crates/buzz-voice/src/imported.rs new file mode 100644 index 000000000..6f0ea71ca --- /dev/null +++ b/crates/buzz-voice/src/imported.rs @@ -0,0 +1,730 @@ +//! Device-local Pocket reference voice validation, canonicalization, and storage. + +use std::{ + fs, + io::Write, + path::{Path, PathBuf}, +}; + +use atomic_write_file::AtomicWriteFile; +use serde::{Deserialize, Serialize}; +use sha2::{Digest, Sha256}; +use symphonia::core::{ + audio::SampleBuffer, codecs::DecoderOptions, errors::Error as SymphoniaError, + formats::FormatOptions, io::MediaSourceStream, meta::MetadataOptions, probe::Hint, +}; + +const MAX_SOURCE_BYTES: u64 = 25 * 1024 * 1024; +const MIN_SAMPLE_RATE: u32 = 8_000; +const MAX_SAMPLE_RATE: u32 = 96_000; +const MIN_DURATION_SECONDS: f64 = 2.0; +const MAX_DURATION_SECONDS: f64 = 30.0; +pub const CANONICAL_SAMPLE_RATE: u32 = 32_000; +const REGISTRY_VERSION: u32 = 1; +const REGISTRY_FILE: &str = "registry.json"; + +#[derive(Clone, Debug, Deserialize, Serialize, PartialEq, Eq)] +#[serde(rename_all = "camelCase")] +pub struct ImportedVoice { + pub key: String, + pub display_name: String, + pub content_hash: String, + pub file_name: String, +} + +#[derive(Default, Deserialize, Serialize)] +#[serde(rename_all = "camelCase")] +struct ImportedVoiceRegistry { + version: u32, + voices: Vec, +} + +#[derive(Clone, Debug)] +pub struct PocketVoiceLibrary { + root: PathBuf, +} + +impl PocketVoiceLibrary { + pub fn new(root: impl Into) -> Self { + Self { root: root.into() } + } + + pub fn root(&self) -> &Path { + &self.root + } + + fn registry_path(&self) -> PathBuf { + self.root.join(REGISTRY_FILE) + } + + pub fn load(&self) -> Result, String> { + let path = self.registry_path(); + if !path.exists() { + return Ok(Vec::new()); + } + let bytes = + fs::read(&path).map_err(|error| format!("could not read imported voices: {error}"))?; + let registry: ImportedVoiceRegistry = serde_json::from_slice(&bytes) + .map_err(|error| format!("imported voice registry is invalid: {error}"))?; + if registry.version > REGISTRY_VERSION { + return Err(format!( + "imported voice registry version {} is newer than this Buzz build supports", + registry.version + )); + } + Ok(registry + .voices + .into_iter() + .filter(valid_identity) + .filter(|voice| self.resolve_file(voice).is_ok()) + .collect()) + } + + fn save(&self, voices: &[ImportedVoice]) -> Result<(), String> { + ensure_storage_dir(&self.root)?; + let payload = serde_json::to_vec_pretty(&ImportedVoiceRegistry { + version: REGISTRY_VERSION, + voices: voices.to_vec(), + }) + .map_err(|error| format!("could not encode imported voice registry: {error}"))?; + atomic_write_restricted(&self.registry_path(), &payload) + .map_err(|error| format!("could not save imported voice registry: {error}")) + } + + pub fn resolve_file(&self, voice: &ImportedVoice) -> Result { + if !valid_identity(voice) { + return Err("Imported voice registry contains an invalid file identity".to_string()); + } + let path = self.root.join(&voice.file_name); + if !is_regular_file_without_symlink(&path) { + return Err(format!("Imported voice {} is missing", voice.display_name)); + } + let bytes = + fs::read(&path).map_err(|error| format!("could not verify imported voice: {error}"))?; + if hex::encode(Sha256::digest(bytes)) != voice.content_hash { + return Err(format!( + "Imported voice {} does not match its content identity", + voice.display_name + )); + } + Ok(path) + } + + pub fn find(&self, key: &str) -> Result, String> { + Ok(self.load()?.into_iter().find(|voice| voice.key == key)) + } + + pub fn import_path(&self, source: &Path) -> Result { + let metadata = fs::metadata(source) + .map_err(|error| format!("could not inspect selected audio: {error}"))?; + if metadata.len() > MAX_SOURCE_BYTES { + return Err("Voice audio must be 25 MB or smaller".to_string()); + } + let extension = source + .extension() + .and_then(|extension| extension.to_str()) + .map(str::to_ascii_lowercase) + .ok_or_else(|| "Voice audio must have a supported file extension".to_string())?; + let samples = if extension == "wav" { + let source_bytes = fs::read(source) + .map_err(|error| format!("could not read selected audio: {error}"))?; + decode_wav(&source_bytes)? + } else { + decode_media(source, &extension)? + }; + let canonical_samples = resample_linear(&samples.samples, samples.sample_rate); + let canonical = encode_pcm16_wav(&canonical_samples, CANONICAL_SAMPLE_RATE); + let hash = hex::encode(Sha256::digest(&canonical)); + let key = format!("pocket:imported:{hash}"); + let file_name = format!("{hash}.wav"); + let display_name = source + .file_stem() + .and_then(|name| name.to_str()) + .map(str::trim) + .filter(|name| !name.is_empty()) + .unwrap_or("Imported voice") + .chars() + .take(80) + .collect::(); + + ensure_storage_dir(&self.root)?; + let file_path = self.root.join(&file_name); + let file_created = !file_path.exists(); + if file_created { + atomic_write_restricted(&file_path, &canonical) + .map_err(|error| format!("could not save imported voice audio: {error}"))?; + } else { + if !is_regular_file_without_symlink(&file_path) { + return Err("Imported voice storage contains an unsafe file entry".to_string()); + } + let existing = fs::read(&file_path) + .map_err(|error| format!("could not verify imported voice audio: {error}"))?; + if hex::encode(Sha256::digest(&existing)) != hash { + return Err("Imported voice storage contains mismatched audio data".to_string()); + } + } + + let mut imported = ImportedVoice { + key, + display_name, + content_hash: hash, + file_name, + }; + let mut voices = self.load()?; + if let Some(existing) = voices + .iter() + .find(|voice| voice.content_hash == imported.content_hash) + { + imported = existing.clone(); + } else { + voices.push(imported.clone()); + } + if let Err(error) = self.save(&voices) { + if file_created { + let _ = fs::remove_file(&file_path); + } + return Err(error); + } + Ok(imported) + } + + pub fn delete(&self, key: &str) -> Result<(), String> { + let mut voices = self.load()?; + let index = voices + .iter() + .position(|voice| voice.key == key) + .ok_or_else(|| format!("Unknown imported voice: {key}"))?; + let previous_voices = voices.clone(); + let removed = voices.remove(index); + self.save(&voices)?; + let path = self.root.join(removed.file_name); + match fs::remove_file(path) { + Ok(()) => Ok(()), + Err(error) if error.kind() == std::io::ErrorKind::NotFound => Ok(()), + Err(error) => { + self.save(&previous_voices).map_err(|rollback_error| { + format!( + "Imported voice audio could not be deleted ({error}), and its registry \ + entry could not be restored ({rollback_error})" + ) + })?; + Err(format!( + "Imported voice audio could not be deleted: {error}" + )) + } + } + } +} + +#[derive(Clone, Copy, Debug, PartialEq)] +pub struct PcmStats { + pub sample_count: usize, + pub sample_rate: u32, + pub duration_seconds: f64, + pub peak: f32, + pub rms: f32, + pub non_silent_samples: usize, +} + +impl PcmStats { + pub fn analyze(samples: &[f32], sample_rate: u32) -> Self { + let peak = samples + .iter() + .filter(|sample| sample.is_finite()) + .fold(0.0_f32, |peak, sample| peak.max(sample.abs())); + let square_sum = samples + .iter() + .filter(|sample| sample.is_finite()) + .map(|sample| sample * sample) + .sum::(); + let rms = if samples.is_empty() { + 0.0 + } else { + (square_sum / samples.len() as f32).sqrt() + }; + Self { + sample_count: samples.len(), + sample_rate, + duration_seconds: if sample_rate == 0 { + 0.0 + } else { + samples.len() as f64 / f64::from(sample_rate) + }, + peak, + rms, + non_silent_samples: samples + .iter() + .filter(|sample| sample.is_finite() && sample.abs() >= 0.001) + .count(), + } + } + + pub fn is_non_silent(self) -> bool { + self.peak >= 0.001 && self.rms >= 0.0001 && self.non_silent_samples > 0 + } +} + +pub fn write_pcm16_wav(path: &Path, samples: &[f32], sample_rate: u32) -> Result<(), String> { + let bytes = encode_pcm16_wav(samples, sample_rate); + fs::write(path, bytes).map_err(|error| format!("could not write PCM evidence: {error}")) +} + +fn ensure_storage_dir(path: &Path) -> Result<(), String> { + fs::create_dir_all(path) + .map_err(|error| format!("could not create local voice storage: {error}"))?; + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + fs::set_permissions(path, fs::Permissions::from_mode(0o700)) + .map_err(|error| format!("could not restrict local voice storage: {error}"))?; + } + Ok(()) +} + +fn atomic_write_restricted(path: &Path, payload: &[u8]) -> Result<(), String> { + let resolved = fs::canonicalize(path).unwrap_or_else(|_| path.to_path_buf()); + let mut file = AtomicWriteFile::open(&resolved) + .map_err(|error| format!("open {} for atomic write: {error}", resolved.display()))?; + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + file.set_permissions(fs::Permissions::from_mode(0o600)) + .map_err(|error| format!("set {} permissions: {error}", resolved.display()))?; + } + file.write_all(payload) + .map_err(|error| format!("write {}: {error}", resolved.display()))?; + file.commit() + .map_err(|error| format!("commit {}: {error}", resolved.display())) +} + +fn valid_hash(hash: &str) -> bool { + hash.len() == 64 && hash.bytes().all(|byte| byte.is_ascii_hexdigit()) +} + +fn valid_identity(voice: &ImportedVoice) -> bool { + valid_hash(&voice.content_hash) + && voice.key == format!("pocket:imported:{}", voice.content_hash) + && voice.file_name == format!("{}.wav", voice.content_hash) +} + +fn is_regular_file_without_symlink(path: &Path) -> bool { + fs::symlink_metadata(path) + .is_ok_and(|metadata| metadata.file_type().is_file() && !metadata.file_type().is_symlink()) +} + +#[derive(Debug)] +struct DecodedAudio { + sample_rate: u32, + samples: Vec, +} + +fn decode_wav(bytes: &[u8]) -> Result { + if bytes.len() < 12 || &bytes[..4] != b"RIFF" || &bytes[8..12] != b"WAVE" { + return Err("Selected file is not a valid RIFF/WAVE file".to_string()); + } + let mut offset = 12usize; + let mut format = None; + let mut data = None; + while offset.checked_add(8).is_some_and(|end| end <= bytes.len()) { + let id = &bytes[offset..offset + 4]; + let size = + u32::from_le_bytes(bytes[offset + 4..offset + 8].try_into().unwrap_or([0; 4])) as usize; + let start = offset + 8; + let end = start.checked_add(size).ok_or("WAV chunk size overflow")?; + if end > bytes.len() { + return Err("Selected WAV contains a truncated chunk".to_string()); + } + if id == b"fmt " { + format = Some(&bytes[start..end]); + } else if id == b"data" { + data = Some(&bytes[start..end]); + } + offset = end + (size & 1); + } + let format = format.ok_or("Selected WAV has no format chunk")?; + let data = data.ok_or("Selected WAV has no audio data")?; + if format.len() < 16 { + return Err("Selected WAV has an invalid format chunk".to_string()); + } + let encoding = u16::from_le_bytes(format[0..2].try_into().unwrap_or([0; 2])); + let encoding = if encoding == 0xfffe && format.len() >= 40 { + u16::from_le_bytes(format[24..26].try_into().unwrap_or([0; 2])) + } else { + encoding + }; + let channels = u16::from_le_bytes(format[2..4].try_into().unwrap_or([0; 2])); + let sample_rate = u32::from_le_bytes(format[4..8].try_into().unwrap_or([0; 4])); + let block_align = u16::from_le_bytes(format[12..14].try_into().unwrap_or([0; 2])) as usize; + let bits = u16::from_le_bytes(format[14..16].try_into().unwrap_or([0; 2])); + if channels == 0 || channels > 8 { + return Err("Voice WAV must contain between 1 and 8 channels".to_string()); + } + if !(MIN_SAMPLE_RATE..=MAX_SAMPLE_RATE).contains(&sample_rate) { + return Err("Voice WAV sample rate must be between 8 and 96 kHz".to_string()); + } + let bytes_per_sample = usize::from(bits.div_ceil(8)); + if block_align != bytes_per_sample * usize::from(channels) + || block_align == 0 + || data.len() % block_align != 0 + { + return Err("Voice WAV has invalid sample alignment".to_string()); + } + if !matches!((encoding, bits), (1, 8 | 16 | 24 | 32) | (3, 32)) { + return Err("Voice WAV must contain PCM or 32-bit float audio".to_string()); + } + let frames = data.len() / block_align; + let duration = frames as f64 / f64::from(sample_rate); + if !(MIN_DURATION_SECONDS..=MAX_DURATION_SECONDS).contains(&duration) { + return Err("Voice WAV must be between 2 and 30 seconds long".to_string()); + } + + let mut samples = Vec::with_capacity(frames); + for frame in data.chunks_exact(block_align) { + let mut mono = 0.0_f32; + for chunk in frame.chunks_exact(bytes_per_sample) { + let sample = match (encoding, bits) { + (1, 8) => (f32::from(chunk[0]) - 128.0) / 128.0, + (1, 16) => f32::from(i16::from_le_bytes([chunk[0], chunk[1]])) / 32768.0, + (1, 24) => { + let raw = i32::from_le_bytes([ + chunk[0], + chunk[1], + chunk[2], + if chunk[2] & 0x80 == 0 { 0 } else { 0xff }, + ]); + raw as f32 / 8_388_608.0 + } + (1, 32) => { + i32::from_le_bytes(chunk.try_into().map_err(|_| "invalid PCM sample")?) as f32 + / 2_147_483_648.0 + } + (3, 32) => f32::from_le_bytes( + chunk + .try_into() + .map_err(|_| "invalid floating-point sample")?, + ), + _ => unreachable!(), + }; + if !sample.is_finite() { + return Err("Voice WAV contains non-finite samples".to_string()); + } + mono += sample; + } + samples.push((mono / f32::from(channels)).clamp(-1.0, 1.0)); + } + let stats = PcmStats::analyze(&samples, sample_rate); + if !stats.is_non_silent() { + return Err("Voice WAV is silent or too quiet to clone".to_string()); + } + Ok(DecodedAudio { + sample_rate, + samples, + }) +} + +fn decode_media(source: &Path, extension: &str) -> Result { + let supported = ["m4a", "mp3", "flac", "ogg", "oga", "aif", "aiff"]; + if !supported.contains(&extension) { + return Err(format!( + "Unsupported voice audio format .{extension}. Choose WAV, M4A, MP3, FLAC, OGG, or AIFF" + )); + } + + let file = fs::File::open(source) + .map_err(|error| format!("could not read selected audio: {error}"))?; + let media = MediaSourceStream::new(Box::new(file), Default::default()); + let mut hint = Hint::new(); + hint.with_extension(extension); + let probed = symphonia::default::get_probe() + .format( + &hint, + media, + &FormatOptions::default(), + &MetadataOptions::default(), + ) + .map_err(|error| format!("could not recognize selected audio: {error}"))?; + let mut format = probed.format; + let track = format + .default_track() + .ok_or_else(|| "Selected audio has no decodable track".to_string())?; + let track_id = track.id; + let mut decoder = symphonia::default::get_codecs() + .make(&track.codec_params, &DecoderOptions::default()) + .map_err(|error| format!("could not initialize audio decoder: {error}"))?; + let mut sample_rate = None; + let mut samples = Vec::new(); + + loop { + let packet = match format.next_packet() { + Ok(packet) => packet, + Err(SymphoniaError::ResetRequired) => { + return Err("Selected audio changes format mid-stream".to_string()); + } + Err(SymphoniaError::IoError(error)) + if error.kind() == std::io::ErrorKind::UnexpectedEof => + { + break; + } + Err(error) => return Err(format!("could not read selected audio: {error}")), + }; + if packet.track_id() != track_id { + continue; + } + let decoded = match decoder.decode(&packet) { + Ok(decoded) => decoded, + Err(SymphoniaError::DecodeError(_)) => continue, + Err(error) => return Err(format!("could not decode selected audio: {error}")), + }; + let spec = *decoded.spec(); + if !(MIN_SAMPLE_RATE..=MAX_SAMPLE_RATE).contains(&spec.rate) { + return Err("Voice audio sample rate must be between 8 and 96 kHz".to_string()); + } + if sample_rate.is_some_and(|rate| rate != spec.rate) { + return Err("Selected audio changes sample rate mid-stream".to_string()); + } + sample_rate = Some(spec.rate); + let channels = spec.channels.count(); + if channels == 0 || channels > 8 { + return Err("Voice audio must contain between 1 and 8 channels".to_string()); + } + let mut buffer = SampleBuffer::::new(decoded.capacity() as u64, spec); + buffer.copy_interleaved_ref(decoded); + for frame in buffer.samples().chunks_exact(channels) { + let mono = frame.iter().copied().sum::() / channels as f32; + if !mono.is_finite() { + return Err("Voice audio contains non-finite samples".to_string()); + } + samples.push(mono.clamp(-1.0, 1.0)); + } + if samples.len() as f64 > MAX_DURATION_SECONDS * f64::from(spec.rate) { + return Err("Voice audio must be between 2 and 30 seconds long".to_string()); + } + } + + let sample_rate = + sample_rate.ok_or_else(|| "Selected audio contains no samples".to_string())?; + validate_decoded_audio(&samples, sample_rate)?; + Ok(DecodedAudio { + sample_rate, + samples, + }) +} + +fn validate_decoded_audio(samples: &[f32], sample_rate: u32) -> Result<(), String> { + let stats = PcmStats::analyze(samples, sample_rate); + if !(MIN_DURATION_SECONDS..=MAX_DURATION_SECONDS).contains(&stats.duration_seconds) { + return Err("Voice audio must be between 2 and 30 seconds long".to_string()); + } + if !stats.is_non_silent() { + return Err("Voice audio is silent or too quiet to clone".to_string()); + } + Ok(()) +} + +fn resample_linear(samples: &[f32], source_rate: u32) -> Vec { + if source_rate == CANONICAL_SAMPLE_RATE { + return samples.to_vec(); + } + let output_len = ((samples.len() as u64 * u64::from(CANONICAL_SAMPLE_RATE) + + u64::from(source_rate) / 2) + / u64::from(source_rate)) as usize; + (0..output_len) + .map(|index| { + let source = index as f64 * f64::from(source_rate) / f64::from(CANONICAL_SAMPLE_RATE); + let left = source.floor() as usize; + let fraction = (source - left as f64) as f32; + let a = samples[left.min(samples.len() - 1)]; + let b = samples[(left + 1).min(samples.len() - 1)]; + a + (b - a) * fraction + }) + .collect() +} + +fn encode_pcm16_wav(samples: &[f32], sample_rate: u32) -> Vec { + let data_len = (samples.len() * 2) as u32; + let mut bytes = Vec::with_capacity(44 + data_len as usize); + bytes.extend_from_slice(b"RIFF"); + bytes.extend_from_slice(&(36 + data_len).to_le_bytes()); + bytes.extend_from_slice(b"WAVEfmt "); + bytes.extend_from_slice(&16_u32.to_le_bytes()); + bytes.extend_from_slice(&1_u16.to_le_bytes()); + bytes.extend_from_slice(&1_u16.to_le_bytes()); + bytes.extend_from_slice(&sample_rate.to_le_bytes()); + bytes.extend_from_slice(&(sample_rate * 2).to_le_bytes()); + bytes.extend_from_slice(&2_u16.to_le_bytes()); + bytes.extend_from_slice(&16_u16.to_le_bytes()); + bytes.extend_from_slice(b"data"); + bytes.extend_from_slice(&data_len.to_le_bytes()); + for sample in samples { + let value = (sample.clamp(-1.0, 1.0) * f32::from(i16::MAX)).round() as i16; + bytes.extend_from_slice(&value.to_le_bytes()); + } + bytes +} + +#[cfg(test)] +mod tests { + use super::*; + + fn fixture(sample_rate: u32, seconds: usize, amplitude: f32) -> Vec { + let samples = (0..sample_rate as usize * seconds) + .map(|index| { + amplitude + * (std::f32::consts::TAU * 220.0 * index as f32 / sample_rate as f32).sin() + }) + .collect::>(); + encode_pcm16_wav(&samples, sample_rate) + } + + fn stereo_fixture(sample_rate: u32, seconds: usize, amplitude: f32) -> Vec { + let mono = fixture(sample_rate, seconds, amplitude); + let mono_data = &mono[44..]; + let mut stereo_data = Vec::with_capacity(mono_data.len() * 2); + for sample in mono_data.chunks_exact(2) { + stereo_data.extend_from_slice(sample); + stereo_data.extend_from_slice(sample); + } + let mut stereo = mono[..44].to_vec(); + stereo[4..8].copy_from_slice(&(36 + stereo_data.len() as u32).to_le_bytes()); + stereo[22..24].copy_from_slice(&2_u16.to_le_bytes()); + stereo[28..32].copy_from_slice(&(sample_rate * 4).to_le_bytes()); + stereo[32..34].copy_from_slice(&4_u16.to_le_bytes()); + stereo[40..44].copy_from_slice(&(stereo_data.len() as u32).to_le_bytes()); + stereo.extend_from_slice(&stereo_data); + stereo + } + + #[test] + fn imports_persists_reloads_and_deletes_canonical_voice() { + let temp = tempfile::tempdir().expect("temp voice workspace"); + let source = temp.path().join("My voice.wav"); + fs::write(&source, fixture(44_100, 2, 0.5)).expect("write source"); + let library = PocketVoiceLibrary::new(temp.path().join("library")); + + let imported = library.import_path(&source).expect("import voice"); + assert!(imported.key.starts_with("pocket:imported:")); + assert_eq!(imported.display_name, "My voice"); + + let relaunched = PocketVoiceLibrary::new(library.root()); + assert_eq!( + relaunched.load().expect("reload registry"), + vec![imported.clone()] + ); + let stored = relaunched + .resolve_file(&imported) + .expect("resolve stored voice"); + let decoded = decode_wav(&fs::read(&stored).expect("read stored voice")) + .expect("decode canonical voice"); + assert_eq!(decoded.sample_rate, CANONICAL_SAMPLE_RATE); + assert_eq!(decoded.samples.len(), CANONICAL_SAMPLE_RATE as usize * 2); + + assert_eq!( + relaunched.import_path(&source).expect("idempotent import"), + imported + ); + assert_eq!(relaunched.load().expect("deduplicated registry").len(), 1); + + relaunched.delete(&imported.key).expect("delete voice"); + assert!(relaunched.load().expect("empty registry").is_empty()); + assert!(!stored.exists()); + } + + #[test] + fn common_stereo_audio_is_downmixed_to_canonical_mono() { + let temp = tempfile::tempdir().expect("temp voice workspace"); + let source = temp.path().join("stereo.wav"); + fs::write(&source, stereo_fixture(44_100, 2, 0.5)).expect("write stereo"); + let library = PocketVoiceLibrary::new(temp.path().join("library")); + + let imported = library.import_path(&source).expect("import stereo"); + let stored = library + .resolve_file(&imported) + .expect("resolve stored voice"); + let decoded = decode_wav(&fs::read(stored).expect("read stored voice")) + .expect("decode canonical voice"); + assert_eq!(decoded.sample_rate, CANONICAL_SAMPLE_RATE); + assert_eq!(decoded.samples.len(), CANONICAL_SAMPLE_RATE as usize * 2); + } + + #[test] + #[ignore = "requires BUZZ_VOICE_IMPORT_TEST_DIR with common-format fixtures"] + fn imports_common_audio_format_fixtures() { + let fixtures = + PathBuf::from(std::env::var("BUZZ_VOICE_IMPORT_TEST_DIR").expect("fixture directory")); + let temp = tempfile::tempdir().expect("temp voice workspace"); + let library = PocketVoiceLibrary::new(temp.path().join("library")); + + for file_name in [ + "voice.wav", + "voice.m4a", + "voice.mp3", + "voice.flac", + "voice.ogg", + "voice.aiff", + ] { + let imported = library + .import_path(&fixtures.join(file_name)) + .unwrap_or_else(|error| panic!("import {file_name}: {error}")); + let stored = library + .resolve_file(&imported) + .unwrap_or_else(|error| panic!("resolve {file_name}: {error}")); + let decoded = decode_wav(&fs::read(stored).expect("read canonical voice")) + .expect("decode canonical voice"); + assert_eq!(decoded.sample_rate, CANONICAL_SAMPLE_RATE); + assert!(decoded.samples.len() >= CANONICAL_SAMPLE_RATE as usize * 2); + } + } + + #[test] + fn invalid_unsupported_and_silent_files_do_not_mutate_registry() { + let temp = tempfile::tempdir().expect("temp voice workspace"); + let library = PocketVoiceLibrary::new(temp.path().join("library")); + + let garbage = temp.path().join("garbage.wav"); + fs::write(&garbage, b"not a wave").expect("write garbage"); + assert!(library + .import_path(&garbage) + .expect_err("garbage rejected") + .contains("RIFF/WAVE")); + + let silent = temp.path().join("silent.wav"); + fs::write(&silent, fixture(32_000, 2, 0.0)).expect("write silence"); + assert!(library + .import_path(&silent) + .expect_err("silence rejected") + .contains("silent")); + + let unsupported_container = temp.path().join("voice.txt"); + fs::write(&unsupported_container, b"not audio").expect("write unsupported container"); + assert!(library + .import_path(&unsupported_container) + .expect_err("container rejected") + .contains("Unsupported voice audio format")); + + let mut unsupported = fixture(32_000, 2, 0.5); + unsupported[20..22].copy_from_slice(&6_u16.to_le_bytes()); + let unsupported_path = temp.path().join("unsupported.wav"); + fs::write(&unsupported_path, unsupported).expect("write unsupported"); + assert!(library + .import_path(&unsupported_path) + .expect_err("unsupported rejected") + .contains("PCM or 32-bit float")); + + assert!(library.load().expect("unchanged registry").is_empty()); + } + + #[test] + fn pcm_analysis_distinguishes_signal_from_silence() { + let signal = (0..24_000) + .map(|index| (std::f32::consts::TAU * 440.0 * index as f32 / 24_000.0).sin() * 0.5) + .collect::>(); + let signal_stats = PcmStats::analyze(&signal, 24_000); + assert!(signal_stats.is_non_silent()); + assert_eq!(signal_stats.duration_seconds, 1.0); + assert!(signal_stats.peak > 0.49); + assert!(signal_stats.rms > 0.3); + + let silence = vec![0.0; 24_000]; + assert!(!PcmStats::analyze(&silence, 24_000).is_non_silent()); + } +} diff --git a/crates/buzz-voice/src/lib.rs b/crates/buzz-voice/src/lib.rs index a47f9149c..e4b4ebfed 100644 --- a/crates/buzz-voice/src/lib.rs +++ b/crates/buzz-voice/src/lib.rs @@ -1,5 +1,6 @@ //! Reusable local voice primitives for Buzz. +pub mod imported; pub mod pocket; pub use pocket::{ diff --git a/crates/buzz-voice/tests/pocket_import_audio.rs b/crates/buzz-voice/tests/pocket_import_audio.rs new file mode 100644 index 000000000..8578c368d --- /dev/null +++ b/crates/buzz-voice/tests/pocket_import_audio.rs @@ -0,0 +1,133 @@ +use std::{ + fs, + path::{Path, PathBuf}, +}; + +use buzz_voice::{ + imported::{write_pcm16_wav, PcmStats, PocketVoiceLibrary}, + pocket::{load_text_to_speech, load_voice_style, DEFAULT_VOICE, SAMPLE_RATE, VOICE_FILE_EXT}, +}; + +const PREVIEW_TEXT: &str = "This is an objective Pocket voice preview."; + +fn required_path(name: &str) -> PathBuf { + std::env::var_os(name) + .map(PathBuf::from) + .unwrap_or_else(|| panic!("{name} must point to the required local test path")) +} + +fn checked_in_voice() -> PathBuf { + Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../desktop/src-tauri/resources/pocket-voices/eve.wav") +} + +fn evidence_dir() -> PathBuf { + std::env::var_os("BUZZ_VOICE_EVIDENCE_DIR") + .map(PathBuf::from) + .unwrap_or_else(|| { + Path::new(env!("CARGO_MANIFEST_DIR")).join("../../target/buzz-voice-evidence") + }) +} + +fn synthesize(model_dir: &Path, voice_path: &Path, text: &str) -> (Vec, PcmStats) { + let engine = load_text_to_speech( + model_dir + .to_str() + .expect("Pocket model path must be valid UTF-8"), + ) + .expect("load Pocket model"); + let style = load_voice_style(voice_path).expect("load selected voice"); + let samples = engine + .synth_chunk(text, "en", &style, 1) + .expect("synthesize preview"); + let stats = PcmStats::analyze(&samples, SAMPLE_RATE); + assert!( + stats.is_non_silent(), + "generated PCM must be non-silent: {stats:?}" + ); + assert!( + stats.duration_seconds > 0.2, + "generated PCM is unexpectedly short: {stats:?}" + ); + (samples, stats) +} + +#[test] +#[ignore = "requires BUZZ_POCKET_MODEL_DIR and runs the installed Pocket ONNX model"] +fn objective_import_synthesis_delete_and_mary_fallback() { + let model_dir = required_path("BUZZ_POCKET_MODEL_DIR"); + let temp = tempfile::tempdir().expect("temporary voice workspace"); + let source = temp.path().join("Imported Eve.wav"); + fs::copy(checked_in_voice(), &source).expect("copy checked-in voice fixture"); + + let library_root = temp.path().join("library"); + let library = PocketVoiceLibrary::new(&library_root); + let imported = library.import_path(&source).expect("import valid WAV"); + assert_eq!( + library.find(&imported.key).expect("read selection"), + Some(imported.clone()) + ); + + drop(library); + let relaunched = PocketVoiceLibrary::new(&library_root); + let selected = relaunched + .find(&imported.key) + .expect("reload persisted selection") + .expect("selected imported voice survived relaunch"); + let imported_path = relaunched + .resolve_file(&selected) + .expect("resolve persisted imported voice"); + let (imported_pcm, imported_stats) = synthesize(&model_dir, &imported_path, PREVIEW_TEXT); + + let evidence = evidence_dir(); + fs::create_dir_all(&evidence).expect("create evidence directory"); + let imported_wav = evidence.join("imported-preview.wav"); + write_pcm16_wav(&imported_wav, &imported_pcm, SAMPLE_RATE) + .expect("write imported preview evidence"); + + relaunched + .delete(&imported.key) + .expect("delete imported voice"); + assert_eq!( + relaunched.find(&imported.key).expect("reload after delete"), + None + ); + + let mary_path = model_dir.join(format!("{DEFAULT_VOICE}.{VOICE_FILE_EXT}")); + assert_eq!( + mary_path.file_name().and_then(|name| name.to_str()), + Some("reference_sample.wav"), + "fallback must remain the deterministic Mary reference" + ); + let (mary_pcm, mary_stats) = synthesize(&model_dir, &mary_path, PREVIEW_TEXT); + let mary_wav = evidence.join("mary-fallback-preview.wav"); + write_pcm16_wav(&mary_wav, &mary_pcm, SAMPLE_RATE) + .expect("write Mary fallback preview evidence"); + + println!( + "{}", + serde_json::json!({ + "importedKey": imported.key, + "persistence": "reloaded", + "afterDelete": "pocket:mary", + "importedPreview": { + "path": imported_wav, + "samples": imported_stats.sample_count, + "sampleRate": imported_stats.sample_rate, + "durationSeconds": imported_stats.duration_seconds, + "peak": imported_stats.peak, + "rms": imported_stats.rms, + "nonSilentSamples": imported_stats.non_silent_samples, + }, + "maryFallbackPreview": { + "path": mary_wav, + "samples": mary_stats.sample_count, + "sampleRate": mary_stats.sample_rate, + "durationSeconds": mary_stats.duration_seconds, + "peak": mary_stats.peak, + "rms": mary_stats.rms, + "nonSilentSamples": mary_stats.non_silent_samples, + } + }) + ); +} diff --git a/desktop/src-tauri/Cargo.lock b/desktop/src-tauri/Cargo.lock index e079b85bf..254b7070a 100644 --- a/desktop/src-tauri/Cargo.lock +++ b/desktop/src-tauri/Cargo.lock @@ -1178,13 +1178,17 @@ dependencies = [ name = "buzz-voice" version = "0.1.0" dependencies = [ + "atomic-write-file", + "hex", "ort", "ort-sys", "rand 0.10.2", "sentencepiece-model", "serde", "serde_json", + "sha2 0.11.0", "sherpa-onnx", + "symphonia", "tokenizers", ] @@ -9778,6 +9782,7 @@ dependencies = [ "symphonia-bundle-flac", "symphonia-bundle-mp3", "symphonia-codec-aac", + "symphonia-codec-alac", "symphonia-codec-pcm", "symphonia-codec-vorbis", "symphonia-core", @@ -9822,6 +9827,16 @@ dependencies = [ "symphonia-core", ] +[[package]] +name = "symphonia-codec-alac" +version = "0.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8413fa754942ac16a73634c9dfd1500ed5c61430956b33728567f667fdd393ab" +dependencies = [ + "log", + "symphonia-core", +] + [[package]] name = "symphonia-codec-pcm" version = "0.5.5" diff --git a/desktop/src-tauri/src/huddle/mod.rs b/desktop/src-tauri/src/huddle/mod.rs index 7d889328c..03264f80f 100644 --- a/desktop/src-tauri/src/huddle/mod.rs +++ b/desktop/src-tauri/src/huddle/mod.rs @@ -39,6 +39,7 @@ pub mod stt; pub mod transcription; pub mod tts; pub mod tts_settings; +mod tts_voice_import; mod tts_voice_registry; pub mod wire; diff --git a/desktop/src-tauri/src/huddle/pipeline.rs b/desktop/src-tauri/src/huddle/pipeline.rs index c4749d0c1..fba5464a6 100644 --- a/desktop/src-tauri/src/huddle/pipeline.rs +++ b/desktop/src-tauri/src/huddle/pipeline.rs @@ -392,6 +392,40 @@ pub(crate) async fn maybe_start_tts_pipeline(state: &AppState) -> Result return Ok(false), }; + // Avoid resolving and hashing imported voice files on every hot-start poll + // when TTS is already disabled or running. The guarded claim below repeats + // these checks after the fallible work to close the race. + { + let huddle = state.huddle()?; + if huddle.tts_pipeline.is_some() || !huddle.tts_enabled { + return Ok(false); + } + } + + // Resolve all fallible construction inputs before claiming the sentinel so + // an unreadable optional voice registry cannot wedge future start attempts. + let output_device = state + .huddle_audio + .output_device + .lock() + .unwrap_or_else(|e| e.into_inner()) + .clone(); + let app = state + .app_handle + .lock() + .map_err(|error| format!("app handle lock poisoned: {error}"))? + .clone(); + let voice_preferences = state + .huddle_audio + .tts + .lock() + .map_err(|error| format!("text-to-speech settings lock poisoned: {error}")) + .map(|settings| settings.voice_preferences.clone())?; + let initial_voice = match app { + Some(app) => super::tts_settings::pocket_voice_reference(&app, &voice_preferences)?, + None => super::tts_settings::bundled_pocket_voice_reference(&voice_preferences), + }; + // Atomically check preconditions and claim the construction slot. // The sentinel prevents a second caller from starting construction // while we're building outside the lock. @@ -416,20 +450,6 @@ pub(crate) async fn maybe_start_tts_pipeline(state: &AppState) -> Result super::tts_settings::pocket_voice_reference(&app, &preferences)?, + None => super::tts_settings::bundled_pocket_voice_reference(&preferences), + }; publish(&voice, &mut huddle); Ok(true) } diff --git a/desktop/src-tauri/src/huddle/tts_settings.rs b/desktop/src-tauri/src/huddle/tts_settings.rs index a3e931dff..1b378af82 100644 --- a/desktop/src-tauri/src/huddle/tts_settings.rs +++ b/desktop/src-tauri/src/huddle/tts_settings.rs @@ -81,12 +81,8 @@ pub struct VoiceProvenance { /// may have that backend installed. Resolution is always local. pub type VoicePreferences = Vec; -/// Cross-backend registry for voices known to this client. -/// -/// V1 contains Pocket entries only. Siri, Kokoro, imported voices, and -/// per-agent assignment can add entries or reuse the preference type without -/// changing the registry/settings boundary. -pub fn voice_registry() -> Vec { +/// Bundled Pocket voices available without local imports. +pub fn bundled_voice_registry() -> Vec { POCKET_VOICES .iter() .map(|voice| VoiceRegistryEntry { @@ -107,6 +103,34 @@ pub fn voice_registry() -> Vec { .collect() } +/// Cross-backend registry of bundled and locally installed voices. +pub fn voice_registry(app: &AppHandle) -> Vec { + let mut registry = bundled_voice_registry(); + match super::tts_voice_import::load_registry(app) { + Ok(imported) => registry.extend(imported.into_iter().map(|voice| VoiceRegistryEntry { + key: voice.key, + display_name: voice.display_name, + backend: POCKET_BACKEND_ID.to_string(), + backend_name: "Pocket TTS".to_string(), + availability: VOICE_AVAILABILITY_INSTALLED.to_string(), + fallback_key: Some(MARY_VOICE_KEY.to_string()), + reference_file: Some(voice.file_name), + provenance: VoiceProvenance { + source: "local import".to_string(), + content_hash: Some(voice.content_hash), + license: None, + source_url: None, + }, + })), + Err(error) => { + eprintln!( + "buzz-desktop: {error}; imported Pocket voices are unavailable for this session" + ); + } + } + registry +} + #[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] #[serde(rename_all = "camelCase")] pub struct TtsSettings { @@ -125,8 +149,10 @@ impl Default for TtsSettings { } } -pub fn voice_by_key(key: &str) -> Option { - voice_registry().into_iter().find(|voice| voice.key == key) +pub fn voice_by_key(app: &AppHandle, key: &str) -> Option { + voice_registry(app) + .into_iter() + .find(|voice| voice.key == key) } fn is_qualified_voice_key(key: &str) -> bool { @@ -141,11 +167,19 @@ fn is_locally_available(availability: &str) -> bool { ) } +#[cfg(test)] pub fn resolve_voice_for_backend( preferences: &[String], backend: &str, ) -> Result { - let registry = voice_registry(); + resolve_voice_for_backend_in_registry(preferences, backend, &bundled_voice_registry()) +} + +fn resolve_voice_for_backend_in_registry( + preferences: &[String], + backend: &str, + registry: &[VoiceRegistryEntry], +) -> Result { preferences .iter() .filter_map(|key| registry.iter().find(|voice| voice.key == *key)) @@ -161,14 +195,31 @@ pub fn resolve_voice_for_backend( .ok_or_else(|| format!("No locally available fallback voice for backend {backend}")) } -pub fn pocket_voice_name(preferences: &[String]) -> String { - resolve_voice_for_backend(preferences, POCKET_BACKEND_ID) +pub fn bundled_pocket_voice_reference(preferences: &[String]) -> String { + resolve_voice_for_backend_in_registry(preferences, POCKET_BACKEND_ID, &bundled_voice_registry()) .ok() .and_then(|voice| voice.reference_file) .and_then(|file| file.strip_suffix(".wav").map(str::to_string)) .unwrap_or_else(|| DEFAULT_VOICE.to_string()) } +pub fn pocket_voice_reference(app: &AppHandle, preferences: &[String]) -> Result { + let registry = voice_registry(app); + let voice = resolve_voice_for_backend_in_registry(preferences, POCKET_BACKEND_ID, ®istry)?; + if voice.key.starts_with("pocket:imported:") { + let imported = super::tts_voice_import::load_registry(app)? + .into_iter() + .find(|candidate| candidate.key == voice.key) + .ok_or_else(|| format!("Imported voice {} is unavailable", voice.display_name))?; + return super::tts_voice_import::resolve_file(app, &imported) + .map(|path| path.to_string_lossy().into_owned()); + } + Ok(voice + .reference_file + .and_then(|file| file.strip_suffix(".wav").map(str::to_string)) + .unwrap_or_else(|| DEFAULT_VOICE.to_string())) +} + pub(crate) fn settings_path(app: &AppHandle) -> Result { app.path() .app_data_dir() @@ -290,8 +341,8 @@ pub fn get_tts_settings(state: State<'_, AppState>) -> Result Vec { - voice_registry() +pub fn list_voice_registry(app: AppHandle) -> Vec { + voice_registry(&app) } fn ensure_settings_writable(state: &AppState) -> Result<(), String> { @@ -394,8 +445,8 @@ async fn apply_tts_settings( if settings.agent_text_to_speech { let (active, voice_change_ack) = { let mut huddle = state.huddle()?; - let voice_change_ack = - enable_tts_runtime(&mut huddle, &pocket_voice_name(&settings.voice_preferences)); + let voice_reference = pocket_voice_reference(app, &settings.voice_preferences)?; + let voice_change_ack = enable_tts_runtime(&mut huddle, &voice_reference); ( matches!(huddle.phase, HuddlePhase::Connected | HuddlePhase::Active), voice_change_ack, @@ -431,6 +482,14 @@ async fn finish_voice_change(voice_change: Option) -> Result<() .await } +async fn finish_durable_voice_change(voice_change: Option) { + if let Err(error) = finish_voice_change(voice_change).await { + eprintln!( + "buzz-desktop: tts stage=voice_switch status=delayed reason=ack_timeout error={error}" + ); + } +} + async fn wait_for_voice_change_ack( mut acknowledged: tokio::sync::oneshot::Receiver<()>, timeout: Duration, @@ -479,10 +538,22 @@ pub async fn set_tts_enabled( } fn settings_with_pocket_voice( + settings: TtsSettings, + voice_key: &str, + app: &AppHandle, +) -> Result { + settings_with_pocket_voice_from_registry(settings, voice_key, &voice_registry(app)) +} + +fn settings_with_pocket_voice_from_registry( mut settings: TtsSettings, voice_key: &str, + registry: &[VoiceRegistryEntry], ) -> Result { - let voice = voice_by_key(voice_key).ok_or_else(|| format!("Unknown voice: {voice_key}"))?; + let voice = registry + .iter() + .find(|voice| voice.key == voice_key) + .ok_or_else(|| format!("Unknown voice: {voice_key}"))?; if voice.backend != POCKET_BACKEND_ID || !is_locally_available(&voice.availability) { return Err("The selected Pocket voice is not available on this device".to_string()); } @@ -515,26 +586,21 @@ pub async fn set_pocket_voice( .lock() .map_err(|error| format!("text-to-speech settings lock poisoned: {error}"))? .clone(); - let settings = settings_with_pocket_voice(settings, &voice_key)?; + let settings = settings_with_pocket_voice(settings, &voice_key, &app)?; let voice_change = apply_tts_settings(settings, &app, &state).await?; drop(transition); - if let Err(error) = finish_voice_change(voice_change).await { - // The preference is already durable. Report the delayed live - // transition diagnostically without telling the UI that saving failed; - // the next pipeline start resolves the persisted voice normally. - eprintln!( - "buzz-desktop: tts stage=voice_switch status=delayed reason=ack_timeout error={error}" - ); - } + finish_durable_voice_change(voice_change).await; current_settings(&state) } #[tauri::command] pub async fn preview_pocket_voice( voice_key: String, + app: AppHandle, state: State<'_, AppState>, ) -> Result<(), String> { - let voice = voice_by_key(&voice_key).ok_or_else(|| format!("Unknown voice: {voice_key}"))?; + let voice = + voice_by_key(&app, &voice_key).ok_or_else(|| format!("Unknown voice: {voice_key}"))?; if voice.backend != POCKET_BACKEND_ID { return Err("Only Pocket voices can be previewed in this build".to_string()); } @@ -548,10 +614,7 @@ pub async fn preview_pocket_voice( .lock() .unwrap_or_else(|error| error.into_inner()) .clone(); - let voice_name = voice - .reference_file - .and_then(|file| file.strip_suffix(".wav").map(str::to_string)) - .ok_or_else(|| format!("Voice {voice_key} has no local Pocket reference file"))?; + let voice_name = pocket_voice_reference(&app, std::slice::from_ref(&voice_key))?; tokio::task::spawn_blocking(move || { let active = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false)); let cancel = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false)); @@ -579,6 +642,69 @@ pub async fn preview_pocket_voice( .map_err(|error| format!("Voice preview task failed: {error}"))? } +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "camelCase")] +pub struct TtsVoiceMutation { + pub settings: TtsSettings, + pub registry: Vec, +} + +#[tauri::command] +pub async fn import_pocket_voice( + app: AppHandle, + state: State<'_, AppState>, +) -> Result, String> { + let Some(imported) = super::tts_voice_import::pick_and_import(&app).await? else { + return Ok(None); + }; + let transition = state.huddle_audio.tts_transition.lock().await; + let settings = current_settings(&state)?; + let settings = settings_with_pocket_voice(settings, &imported.key, &app)?; + let voice_change = apply_tts_settings(settings, &app, &state).await?; + drop(transition); + finish_durable_voice_change(voice_change).await; + Ok(Some(TtsVoiceMutation { + settings: current_settings(&state)?, + registry: voice_registry(&app), + })) +} + +#[tauri::command] +pub async fn delete_pocket_voice( + voice_key: String, + app: AppHandle, + state: State<'_, AppState>, +) -> Result { + if !voice_key.starts_with("pocket:imported:") { + return Err("Bundled voices cannot be deleted".to_string()); + } + if voice_by_key(&app, &voice_key).is_none() { + return Err(format!("Unknown imported voice: {voice_key}")); + } + + let transition = state.huddle_audio.tts_transition.lock().await; + let current = current_settings(&state)?; + let selected = resolve_voice_for_backend_in_registry( + ¤t.voice_preferences, + POCKET_BACKEND_ID, + &voice_registry(&app), + ) + .is_ok_and(|voice| voice.key == voice_key); + let voice_change = if selected { + let fallback = settings_with_pocket_voice(current, MARY_VOICE_KEY, &app)?; + apply_tts_settings(fallback, &app, &state).await? + } else { + None + }; + drop(transition); + finish_durable_voice_change(voice_change).await; + super::tts_voice_import::delete(&app, &voice_key)?; + Ok(TtsVoiceMutation { + settings: current_settings(&state)?, + registry: voice_registry(&app), + }) +} + #[cfg(test)] mod tests { use super::*; @@ -619,7 +745,7 @@ mod tests { #[test] fn registry_has_all_official_english_vctk_presets() { assert_eq!( - voice_registry() + bundled_voice_registry() .iter() .map(|voice| { ( @@ -680,7 +806,7 @@ mod tests { fn identity_is_qualified_key_not_display_label() { assert!(is_qualified_voice_key("pocket:imported:audio-content-hash")); assert_ne!(MARY_VOICE_KEY, EVE_VOICE_KEY); - let mut registry = voice_registry(); + let mut registry = bundled_voice_registry(); registry[0].display_name = "Jim".to_string(); registry[1].display_name = "Jim".to_string(); assert_eq!(registry[0].display_name, registry[1].display_name); @@ -792,7 +918,12 @@ mod tests { voice_preferences: vec!["siri:aaron".to_string(), MARY_VOICE_KEY.to_string()], ..TtsSettings::default() }; - let updated = settings_with_pocket_voice(current, EVE_VOICE_KEY).expect("available voice"); + let updated = settings_with_pocket_voice_from_registry( + current, + EVE_VOICE_KEY, + &bundled_voice_registry(), + ) + .expect("available voice"); assert!(!updated.agent_text_to_speech); assert_eq!(updated.voice_preferences, vec!["siri:aaron", EVE_VOICE_KEY]); } @@ -805,8 +936,12 @@ mod tests { // This models the next command after the OFF save fails: it must merge // from effective memory state, not the stale last-persisted ON value. let current = state.huddle_audio.tts.lock().expect("settings").clone(); - let voice_update = - settings_with_pocket_voice(current, EVE_VOICE_KEY).expect("available voice"); + let voice_update = settings_with_pocket_voice_from_registry( + current, + EVE_VOICE_KEY, + &bundled_voice_registry(), + ) + .expect("available voice"); assert!(!voice_update.agent_text_to_speech); } @@ -820,7 +955,12 @@ mod tests { .expect("settings") .agent_text_to_speech = false; let current = state.huddle_audio.tts.lock().expect("settings").clone(); - let unsaved = settings_with_pocket_voice(current, EVE_VOICE_KEY).expect("available voice"); + let unsaved = settings_with_pocket_voice_from_registry( + current, + EVE_VOICE_KEY, + &bundled_voice_registry(), + ) + .expect("available voice"); // This is the only pre-persistence mutation for an OFF candidate. commit_effective_off(&state).expect("commit effective OFF state"); diff --git a/desktop/src-tauri/src/huddle/tts_voice_import.rs b/desktop/src-tauri/src/huddle/tts_voice_import.rs new file mode 100644 index 000000000..cdcc7761e --- /dev/null +++ b/desktop/src-tauri/src/huddle/tts_voice_import.rs @@ -0,0 +1,59 @@ +//! Tauri native-picker adapter for the reusable local Pocket voice library. + +use std::path::PathBuf; + +use buzz_voice_pkg::imported::{ImportedVoice, PocketVoiceLibrary}; +use tauri::{AppHandle, Manager}; + +pub fn voices_dir(app: &AppHandle) -> Result { + app.path() + .app_data_dir() + .map(|path| path.join("tts").join("pocket-voices")) + .map_err(|error| format!("could not locate local voice storage: {error}")) +} + +fn library(app: &AppHandle) -> Result { + voices_dir(app).map(PocketVoiceLibrary::new) +} + +pub fn load_registry(app: &AppHandle) -> Result, String> { + library(app)?.load() +} + +pub fn resolve_file(app: &AppHandle, voice: &ImportedVoice) -> Result { + library(app)?.resolve_file(voice) +} + +pub async fn pick_and_import(app: &AppHandle) -> Result, String> { + use tauri_plugin_dialog::DialogExt; + + let (sender, receiver) = tokio::sync::oneshot::channel(); + app.dialog() + .file() + .add_filter( + "Audio", + &["wav", "m4a", "mp3", "flac", "ogg", "oga", "aif", "aiff"], + ) + .pick_file(move |path| { + let _ = sender.send(path); + }); + let Some(file_path) = receiver + .await + .map_err(|_| "voice picker closed unexpectedly".to_string())? + else { + return Ok(None); + }; + let path = file_path + .as_path() + .ok_or("the selected voice path is invalid")? + .to_path_buf(); + let voice_library = library(app)?; + tokio::task::spawn_blocking(move || voice_library.import_path(&path)) + .await + .map_err(|error| format!("voice import task failed: {error}"))? + .map(Some) +} + +pub fn delete(app: &AppHandle, key: &str) -> Result<(), String> { + library(app)?.delete(key) +} diff --git a/desktop/src-tauri/src/huddle/tts_voice_transition.rs b/desktop/src-tauri/src/huddle/tts_voice_transition.rs index 833d3b1c4..81b33672d 100644 --- a/desktop/src-tauri/src/huddle/tts_voice_transition.rs +++ b/desktop/src-tauri/src/huddle/tts_voice_transition.rs @@ -120,7 +120,7 @@ pub(super) fn reconcile_selected_voice( return true; } - let requested_path = model_dir.join(format!("{requested_voice}.{VOICE_FILE_EXT}")); + let requested_path = voice_path(model_dir, &requested_voice); match load_voice_style(&requested_path) { Ok(requested_style) => { *style = requested_style; @@ -151,6 +151,15 @@ pub(super) fn reconcile_selected_voice( } } +pub(super) fn voice_path(model_dir: &Path, voice: &str) -> std::path::PathBuf { + let path = Path::new(voice); + if path.is_absolute() { + path.to_path_buf() + } else { + model_dir.join(format!("{voice}.{VOICE_FILE_EXT}")) + } +} + pub(super) fn retain_cancelled_text( deferred_text: &mut VecDeque, current_text: &mut Option, diff --git a/desktop/src-tauri/src/lib.rs b/desktop/src-tauri/src/lib.rs index 316acec2a..6814008f0 100644 --- a/desktop/src-tauri/src/lib.rs +++ b/desktop/src-tauri/src/lib.rs @@ -893,6 +893,8 @@ pub fn run() { huddle::tts_settings::list_voice_registry, huddle::tts_settings::set_pocket_voice, huddle::tts_settings::preview_pocket_voice, + huddle::tts_settings::import_pocket_voice, + huddle::tts_settings::delete_pocket_voice, speak_agent_message, add_agent_to_huddle, check_pipeline_hotstart, diff --git a/desktop/src/features/settings/ui/VoiceSettingsCard.tsx b/desktop/src/features/settings/ui/VoiceSettingsCard.tsx index 0854b099e..30d259fec 100644 --- a/desktop/src/features/settings/ui/VoiceSettingsCard.tsx +++ b/desktop/src/features/settings/ui/VoiceSettingsCard.tsx @@ -1,9 +1,19 @@ import * as React from "react"; -import { ChevronDown, Play, Volume2 } from "lucide-react"; +import { ChevronDown, Play, Trash2, Upload, Volume2 } from "lucide-react"; import { invokeTauri } from "@/shared/api/tauri"; import { cn } from "@/shared/lib/cn"; import { Button } from "@/shared/ui/button"; +import { + AlertDialog, + AlertDialogAction, + AlertDialogCancel, + AlertDialogContent, + AlertDialogDescription, + AlertDialogFooter, + AlertDialogHeader, + AlertDialogTitle, +} from "@/shared/ui/alert-dialog"; import { DropdownMenu, DropdownMenuContent, @@ -27,11 +37,18 @@ export type TtsSettings = { voicePreferences: string[]; }; +type TtsVoiceMutation = { + settings: TtsSettings; + registry: VoiceRegistryEntry[]; +}; + export function VoiceSettingsCard() { const [settings, setSettings] = React.useState(null); const [registry, setRegistry] = React.useState([]); const [busy, setBusy] = React.useState(false); const [previewing, setPreviewing] = React.useState(false); + const [deleteCandidate, setDeleteCandidate] = + React.useState(null); const [error, setError] = React.useState(null); React.useEffect(() => { @@ -111,6 +128,50 @@ export function VoiceSettingsCard() { } }, []); + const importPocketVoice = React.useCallback(async () => { + setBusy(true); + setError(null); + try { + const result = await invokeTauri( + "import_pocket_voice", + ); + if (result) { + setSettings(result.settings); + setRegistry(result.registry); + } + } catch (importError) { + setError( + importError instanceof Error + ? importError.message + : "Voice could not be imported.", + ); + } finally { + setBusy(false); + } + }, []); + + const deletePocketVoice = React.useCallback(async (voiceKey: string) => { + setBusy(true); + setError(null); + try { + const result = await invokeTauri( + "delete_pocket_voice", + { voiceKey }, + ); + setSettings(result.settings); + setRegistry(result.registry); + setDeleteCandidate(null); + } catch (deleteError) { + setError( + deleteError instanceof Error + ? deleteError.message + : "Voice could not be deleted.", + ); + } finally { + setBusy(false); + } + }, []); + const voices = voicesForBackend(registry, "pocket"); const selectedVoice = selectedVoiceForBackend( settings?.voicePreferences ?? [], @@ -236,6 +297,28 @@ export function VoiceSettingsCard() { )} Preview + + {selectedVoice?.key.startsWith("pocket:imported:") && ( + + )} @@ -251,6 +334,41 @@ export function VoiceSettingsCard() {

)} + { + if (!open) setDeleteCandidate(null); + }} + open={deleteCandidate !== null} + > + + + Delete imported voice? + + {deleteCandidate + ? `${deleteCandidate.displayName} and its local audio file will be removed.` + : "This imported voice and its local audio file will be removed."} + {selectedVoice?.key === deleteCandidate?.key && + " Mary will be selected instead."} + + + + Cancel + { + event.preventDefault(); + if (deleteCandidate) { + void deletePocketVoice(deleteCandidate.key); + } + }} + > + Delete voice + + + + ); } diff --git a/desktop/src/testing/e2eBridge.ts b/desktop/src/testing/e2eBridge.ts index 97051b68e..818144415 100644 --- a/desktop/src/testing/e2eBridge.ts +++ b/desktop/src/testing/e2eBridge.ts @@ -166,6 +166,8 @@ type E2eConfig = { agentTextToSpeech: boolean; voicePreferences: string[]; }; + /** Native picker boundary result for Pocket voice import tests. */ + pocketVoiceImportResult?: "success" | "cancel" | "invalid"; /** Advertised HEAD for the first mock project without adding that branch. */ projectHeadBranch?: string; /** Builderlab account returned by hosted-community onboarding. Null/omitted = signed out. */ @@ -9911,7 +9913,25 @@ export function maybeInstallE2eTauriMocks() { deviceId: state === "running" ? "mock-endpoint-id" : null, deviceName: state === "running" ? "Mock desktop" : null, }); - const handleMockCommand = async (command: string, payload: unknown) => { + let mockImportedVoices: Array<{ + key: string; + displayName: string; + backend: string; + backendName: string; + availability: "installed"; + fallbackKey: string; + referenceFile: string; + provenance: { + source: string; + contentHash: string; + license: null; + sourceUrl: null; + }; + }> = []; + const handleMockCommand = async ( + command: string, + payload: unknown, + ): Promise => { const activeConfig = getConfig(); const identity = getActiveIdentity(activeConfig); window.__BUZZ_E2E_COMMANDS__?.push(command); @@ -9969,107 +9989,110 @@ export function maybeInstallE2eTauriMocks() { ); case "list_voice_registry": return [ - [ - "anna", - "Anna", - "anna.wav", - "p228_023_enhanced.wav", - "0a6de25cf12bf1540beb85979f306a92be81fecc051c547c5395e7e5237a3856", - ], - [ - "vera", - "Vera", - "vera.wav", - "p229_023_enhanced.wav", - "309cf91a895830f15842b398f69a4962cb1f7e0bfab10e25dd27838e826c204b", - ], - [ - "fantine", - "Fantine", - "fantine.wav", - "p244_023_enhanced.wav", - "5f07d4e2a3f20a15572aae885156b43ef3fc12ef3812996fd135680d9956448b", - ], - [ - "charles", - "Charles", - "charles.wav", - "p254_023_enhanced.wav", - "6b681a429198f16e378d53bccb08d06939da7b00144a7696111d4f8f76be7756", - ], - [ - "paul", - "Paul", - "paul.wav", - "p259_023_enhanced.wav", - "7aba504fe0b3b16478b69ed27ce6007e3cb42b0c1915b5f1c6a6024ae37d679b", - ], - [ - "eponine", - "Eponine", - "eponine.wav", - "p262_023_enhanced.wav", - "a13c27fb47627b05223691a0ef2974358a18c886e6c2f9d2762ff1d02c20926b", - ], - [ - "azelma", - "Azelma", - "azelma.wav", - "p303_023_enhanced.wav", - "60e3d26cdf2efdec5df712152c839928f4d5522821e6554ae11fd96c57ab1026", - ], - [ - "george", - "George", - "george.wav", - "p315_023_enhanced.wav", - "29a41f93bf5236e5b21501091d7774c255d5f3d4e62fa4f9fdf0a92a793c84ae", - ], - [ - "mary", - "Mary", - "reference_sample.wav", - "p333_023_enhanced.wav", - "a35b0468382218e9f37a9a7494d1e4b74deaf18d7ced22265b4e325bb55c183f", - ], - [ - "jane", - "Jane", - "jane.wav", - "p339_023_enhanced.wav", - "2f12e7f155eb3118f55425394f1b049e5b1b67bdc9b3932c8ba4521420aeb84a", - ], - [ - "michael", - "Michael", - "michael.wav", - "p360_023_enhanced.wav", - "b6743e9195e5e3fd34fe9d1633ae93f7ffab787b249e45f6467d7d6f7a6ee6ad", - ], - [ - "eve", - "Eve", - "eve.wav", - "p361_023_enhanced.wav", - "396e7cbd066b0f3fb6d67fa26e7904076958239d736d4390f15b5fe88feb14cd", - ], - ].map( - ([id, displayName, referenceFile, upstreamFile, contentHash]) => ({ - key: `pocket:${id}`, - displayName, - backend: "pocket", - backendName: "Pocket TTS", - availability: "bundled", - fallbackKey: id === "mary" ? null : "pocket:mary", - referenceFile, - provenance: { - source: "bundled", - contentHash, - license: "CC-BY-4.0", - sourceUrl: `https://huggingface.co/kyutai/tts-voices/blob/323332d33f997de8394f24a193e1a76df720e01a/vctk/${upstreamFile}`, - }, - }), - ); + ...[ + [ + "anna", + "Anna", + "anna.wav", + "p228_023_enhanced.wav", + "0a6de25cf12bf1540beb85979f306a92be81fecc051c547c5395e7e5237a3856", + ], + [ + "vera", + "Vera", + "vera.wav", + "p229_023_enhanced.wav", + "309cf91a895830f15842b398f69a4962cb1f7e0bfab10e25dd27838e826c204b", + ], + [ + "fantine", + "Fantine", + "fantine.wav", + "p244_023_enhanced.wav", + "5f07d4e2a3f20a15572aae885156b43ef3fc12ef3812996fd135680d9956448b", + ], + [ + "charles", + "Charles", + "charles.wav", + "p254_023_enhanced.wav", + "6b681a429198f16e378d53bccb08d06939da7b00144a7696111d4f8f76be7756", + ], + [ + "paul", + "Paul", + "paul.wav", + "p259_023_enhanced.wav", + "7aba504fe0b3b16478b69ed27ce6007e3cb42b0c1915b5f1c6a6024ae37d679b", + ], + [ + "eponine", + "Eponine", + "eponine.wav", + "p262_023_enhanced.wav", + "a13c27fb47627b05223691a0ef2974358a18c886e6c2f9d2762ff1d02c20926b", + ], + [ + "azelma", + "Azelma", + "azelma.wav", + "p303_023_enhanced.wav", + "60e3d26cdf2efdec5df712152c839928f4d5522821e6554ae11fd96c57ab1026", + ], + [ + "george", + "George", + "george.wav", + "p315_023_enhanced.wav", + "29a41f93bf5236e5b21501091d7774c255d5f3d4e62fa4f9fdf0a92a793c84ae", + ], + [ + "mary", + "Mary", + "reference_sample.wav", + "p333_023_enhanced.wav", + "a35b0468382218e9f37a9a7494d1e4b74deaf18d7ced22265b4e325bb55c183f", + ], + [ + "jane", + "Jane", + "jane.wav", + "p339_023_enhanced.wav", + "2f12e7f155eb3118f55425394f1b049e5b1b67bdc9b3932c8ba4521420aeb84a", + ], + [ + "michael", + "Michael", + "michael.wav", + "p360_023_enhanced.wav", + "b6743e9195e5e3fd34fe9d1633ae93f7ffab787b249e45f6467d7d6f7a6ee6ad", + ], + [ + "eve", + "Eve", + "eve.wav", + "p361_023_enhanced.wav", + "396e7cbd066b0f3fb6d67fa26e7904076958239d736d4390f15b5fe88feb14cd", + ], + ].map( + ([id, displayName, referenceFile, upstreamFile, contentHash]) => ({ + key: `pocket:${id}`, + displayName, + backend: "pocket", + backendName: "Pocket TTS", + availability: "bundled", + fallbackKey: id === "mary" ? null : "pocket:mary", + referenceFile, + provenance: { + source: "bundled", + contentHash, + license: "CC-BY-4.0", + sourceUrl: `https://huggingface.co/kyutai/tts-voices/blob/323332d33f997de8394f24a193e1a76df720e01a/vctk/${upstreamFile}`, + }, + }), + ), + ...mockImportedVoices, + ]; case "set_tts_enabled": { const enabled = (payload as { enabled?: boolean })?.enabled; if (typeof enabled !== "boolean") @@ -10114,6 +10137,75 @@ export function maybeInstallE2eTauriMocks() { } case "preview_pocket_voice": return null; + case "import_pocket_voice": { + const importResult = + activeConfig?.mock?.pocketVoiceImportResult ?? "success"; + if (importResult === "cancel") return null; + if (importResult === "invalid") { + throw new Error("Voice WAV must contain PCM or 32-bit float audio"); + } + const contentHash = "1".repeat(64); + const imported = { + key: `pocket:imported:${contentHash}`, + displayName: "My voice", + backend: "pocket", + backendName: "Pocket TTS", + availability: "installed" as const, + fallbackKey: "pocket:mary", + referenceFile: `${contentHash}.wav`, + provenance: { + source: "local import", + contentHash, + license: null, + sourceUrl: null, + }, + }; + mockImportedVoices = [imported]; + const current = activeConfig?.mock?.ttsSettings ?? { + version: 1, + agentTextToSpeech: true, + voicePreferences: ["pocket:mary"], + }; + const settings = { + ...current, + voicePreferences: [imported.key], + }; + if (activeConfig) { + activeConfig.mock ??= {}; + activeConfig.mock.ttsSettings = settings; + } + return { + settings, + registry: await handleMockCommand("list_voice_registry", null), + }; + } + case "delete_pocket_voice": { + const voiceKey = (payload as { voiceKey?: string })?.voiceKey; + if (!voiceKey?.startsWith("pocket:imported:")) + throw new Error("Missing imported Pocket voice key"); + mockImportedVoices = mockImportedVoices.filter( + (voice) => voice.key !== voiceKey, + ); + const current = activeConfig?.mock?.ttsSettings ?? { + version: 1, + agentTextToSpeech: true, + voicePreferences: ["pocket:mary"], + }; + const settings = { + ...current, + voicePreferences: current.voicePreferences.includes(voiceKey) + ? ["pocket:mary"] + : current.voicePreferences, + }; + if (activeConfig) { + activeConfig.mock ??= {}; + activeConfig.mock.ttsSettings = settings; + } + return { + settings, + registry: await handleMockCommand("list_voice_registry", null), + }; + } case "get_builderlab_auth": return activeConfig?.mock?.builderlabAuth ?? null; case "start_builderlab_login": { diff --git a/desktop/tests/e2e/voice-settings.spec.ts b/desktop/tests/e2e/voice-settings.spec.ts index 490bee342..9bd8742ed 100644 --- a/desktop/tests/e2e/voice-settings.spec.ts +++ b/desktop/tests/e2e/voice-settings.spec.ts @@ -116,4 +116,97 @@ test.describe("Pocket voice settings", () => { }, }); }); + + test("imports, selects, and safely deletes a local voice", async ({ + page, + }) => { + await installMockBridge(page); + await page.goto("/", { waitUntil: "domcontentloaded" }); + await openSettings(page, "voice"); + + await page.getByTestId("pocket-voice-import").click(); + await expect(page.getByTestId("pocket-voice-selector")).toContainText( + "My voice", + ); + await expect(page.getByTestId("pocket-voice-delete")).toBeVisible(); + await page.getByRole("button", { name: "Preview" }).click(); + + await page.getByTestId("pocket-voice-delete").click(); + await expect(page.getByText("Delete imported voice?")).toBeVisible(); + await page.getByTestId("confirm-pocket-voice-delete").click(); + await expect(page.getByTestId("pocket-voice-selector")).toContainText( + "Mary", + ); + await expect(page.getByTestId("pocket-voice-delete")).toBeHidden(); + await page.getByRole("button", { name: "Preview" }).click(); + + const mutations = await page.evaluate(() => + (window.__BUZZ_E2E_COMMAND_LOG__ ?? []) + .filter((entry) => + [ + "import_pocket_voice", + "preview_pocket_voice", + "delete_pocket_voice", + ].includes(entry.command), + ) + .map((entry) => ({ command: entry.command, payload: entry.payload })), + ); + expect(mutations).toEqual([ + { command: "import_pocket_voice", payload: {} }, + { + command: "preview_pocket_voice", + payload: { voiceKey: `pocket:imported:${"1".repeat(64)}` }, + }, + { + command: "delete_pocket_voice", + payload: { voiceKey: `pocket:imported:${"1".repeat(64)}` }, + }, + { + command: "preview_pocket_voice", + payload: { voiceKey: "pocket:mary" }, + }, + ]); + }); + + test("keeps the selected voice unchanged when the native picker is cancelled", async ({ + page, + }) => { + await installMockBridge(page, { pocketVoiceImportResult: "cancel" }); + await page.goto("/", { waitUntil: "domcontentloaded" }); + await openSettings(page, "voice"); + + await expect(page.getByTestId("pocket-voice-selector")).toContainText( + "Mary", + ); + await page.getByTestId("pocket-voice-import").click(); + await expect(page.getByTestId("pocket-voice-selector")).toContainText( + "Mary", + ); + await expect(page.getByTestId("pocket-voice-delete")).toBeHidden(); + await expect(page.getByTestId("voice-settings-error")).toBeHidden(); + + const audioCommands = await page.evaluate(() => + (window.__BUZZ_E2E_COMMAND_LOG__ ?? []).filter((entry) => + ["preview_pocket_voice", "delete_pocket_voice"].includes(entry.command), + ), + ); + expect(audioCommands).toEqual([]); + }); + + test("surfaces invalid or unsupported WAV errors without changing selection", async ({ + page, + }) => { + await installMockBridge(page, { pocketVoiceImportResult: "invalid" }); + await page.goto("/", { waitUntil: "domcontentloaded" }); + await openSettings(page, "voice"); + + await page.getByTestId("pocket-voice-import").click(); + await expect(page.getByTestId("voice-settings-error")).toContainText( + "Voice WAV must contain PCM or 32-bit float audio", + ); + await expect(page.getByTestId("pocket-voice-selector")).toContainText( + "Mary", + ); + await expect(page.getByTestId("pocket-voice-delete")).toBeHidden(); + }); }); diff --git a/desktop/tests/helpers/bridge.ts b/desktop/tests/helpers/bridge.ts index 2d41d27a5..8a9ab2be1 100644 --- a/desktop/tests/helpers/bridge.ts +++ b/desktop/tests/helpers/bridge.ts @@ -164,6 +164,8 @@ type MockBridgeOptions = { agentTextToSpeech: boolean; voicePreferences: string[]; }; + /** Native picker boundary result for Pocket voice import tests. */ + pocketVoiceImportResult?: "success" | "cancel" | "invalid"; /** Advertised HEAD for the first mock project without adding that branch. */ projectHeadBranch?: string; /** Relay NIP-11 identity used to sign authoritative repository state. */