feat(desktop): import local Pocket voices (#3259)

## Context

Pocket TTS currently offers bundled reference voices. People also need a
local, private way to add a voice without sending audio to a cloud
service.

## Summary

Add a Pocket voice import flow to Voice settings. Buzz opens the native
file picker, decodes common audio formats in the reusable `buzz-voice`
crate, canonicalizes the selected audio, stores it under a
content-derived identity in app data, selects it, and lets the user
delete it later.

## Changes

- Accept WAV, M4A, MP3, FLAC, OGG, and AIFF files between 2 and 30
seconds, including multichannel sources.
- Decode and downmix accepted audio to canonical mono 32 kHz PCM16 WAV
before hashing and storage.
- Store imported voices behind stable `pocket:imported:<sha256>`
identities and content-addressed files.
- Keep absolute file paths inside the native process and expose only
voice metadata to React.
- Include imported voices in Pocket preview and live huddle playback.
- Add Add voice and delete controls while preserving the bundled Pocket
voice catalog.
- Fall back to Mary when the selected imported voice is deleted.
- Keep durable import, selection, and deletion successful when a live
TTS worker acknowledgement is delayed.
- Preserve bundled voices when optional import metadata is unreadable
and keep failed deletion retryable.

## Related issue

None found.

## Testing

Production decoding was exercised with WAV, M4A with AAC, MP3, FLAC, OGG
Vorbis, and AIFF fixtures. Each format canonicalized to mono 32 kHz
PCM16 WAV. Manual validation in the combined daily-driver build covered
native-picker import, Preview, live-huddle playback, deletion, and Mary
fallback.

## Screenshots

The Voice settings card preserves the bundled Pocket catalog and adds
the local Add voice action.

![Pocket TTS voice
import](https://raw.githubusercontent.com/block/buzz/c03ba29060ca544c5ac3394c212f376651b386a3/pr-3259--pocket-voices.png)

## Reviewer-reproducible examples

Create common-format fixtures and run them through the production
importer:

```bash
. ./bin/activate-hermit
fixtures="$(mktemp -d)"
ffmpeg -hide_banner -loglevel error -f lavfi -i "sine=frequency=220:duration=3" -ac 2 -ar 44100 "$fixtures/voice.wav"
ffmpeg -hide_banner -loglevel error -i "$fixtures/voice.wav" -c:a aac "$fixtures/voice.m4a"
ffmpeg -hide_banner -loglevel error -i "$fixtures/voice.wav" "$fixtures/voice.mp3"
ffmpeg -hide_banner -loglevel error -i "$fixtures/voice.wav" "$fixtures/voice.flac"
ffmpeg -hide_banner -loglevel error -i "$fixtures/voice.wav" -c:a libvorbis "$fixtures/voice.ogg"
ffmpeg -hide_banner -loglevel error -i "$fixtures/voice.wav" -c:a pcm_s16be "$fixtures/voice.aiff"
BUZZ_VOICE_IMPORT_TEST_DIR="$fixtures" \
  cargo test -p buzz-voice imports_common_audio_format_fixtures -- --ignored --nocapture
```

Exercise import persistence, synthesis, deletion, and bundled-voice
fallback with an installed Pocket model:

```bash
BUZZ_POCKET_MODEL_DIR=/path/to/pocket-model-bundle \
  cargo test -p buzz-voice --test pocket_import_audio \
  objective_import_synthesis_delete_and_mary_fallback \
  -- --ignored --nocapture
```

Exercise the native-picker boundary, selection, preview dispatch,
deletion, cancellation, and invalid-file states:

```bash
cd desktop
pnpm build:e2e
pnpm exec playwright test tests/e2e/voice-settings.spec.ts --project=smoke
```

---------

Signed-off-by: John Tennant <jtennant@block.xyz>
Signed-off-by: John Tennant <johnmatthewtennant@gmail.com>
Signed-off-by: John Tennant <jtennant@squareup.com>
Signed-off-by: npub1qyvc0c5kl4gqv2fd97fsk46tu378sqgy35vc83rvgfwne90sel7s0ed67d <011987e296fd5006292d2f930b574be47c7801048d1983c46c425d3c95f0cffd@buzz.block.builderlab.xyz>
Signed-off-by: npub12gtutshhh76rx0jx697f32f9tffd4hhp3hx58fp4x6u4uemkm7sqf8f757 <5217c5c2f7bfb4333e46d17c98a9255a52dadee18dcd43a43536b95e6776dfa0@buzz.block.builderlab.xyz>
Co-authored-by: John Tennant <jtennant@block.xyz>
Co-authored-by: npub1qyvc0c5kl4gqv2fd97fsk46tu378sqgy35vc83rvgfwne90sel7s0ed67d <011987e296fd5006292d2f930b574be47c7801048d1983c46c425d3c95f0cffd@buzz.block.builderlab.xyz>
Co-authored-by: npub12gtutshhh76rx0jx697f32f9tffd4hhp3hx58fp4x6u4uemkm7sqf8f757 <5217c5c2f7bfb4333e46d17c98a9255a52dadee18dcd43a43536b95e6776dfa0@buzz.block.builderlab.xyz>
This commit is contained in:
John Matthew Tennant
2026-07-31 11:15:09 -04:00
committed by GitHub
co-authored by John Tennant npub1qyvc0c5kl4gqv2fd97fsk46tu378sqgy35vc83rvgfwne90sel7s0ed67d npub12gtutshhh76rx0jx697f32f9tffd4hhp3hx58fp4x6u4uemkm7sqf8f757
parent 39ce3dfc3c
commit c104eecfb3
16 changed files with 1778 additions and 158 deletions
Generated
+191
View File
@@ -410,6 +410,16 @@ version = "1.1.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1505bd5d3d116872e7271a6d4e16d81d0c8570876c8de68093a09ac269d8aac0"
[[package]]
name = "atomic-write-file"
version = "0.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "84790c55b5704b0d35130bf16a4ce22a8e70eb0ea773522557524d9a4852663d"
dependencies = [
"nix 0.30.1",
"rand 0.9.4",
]
[[package]]
name = "attohttpc"
version = "0.30.1"
@@ -1295,13 +1305,18 @@ dependencies = [
name = "buzz-voice"
version = "0.1.0"
dependencies = [
"atomic-write-file",
"hex",
"ort",
"ort-sys",
"rand 0.10.1",
"sentencepiece-model",
"serde",
"serde_json",
"sha2 0.11.0",
"sherpa-onnx",
"symphonia",
"tempfile",
"tokenizers",
]
@@ -2762,6 +2777,12 @@ dependencies = [
"smallvec",
]
[[package]]
name = "extended"
version = "0.1.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "af9673d8203fcb076b19dfd17e38b3d4ae9f44959416ea532ce72415a6020365"
[[package]]
name = "fancy-regex"
version = "0.11.0"
@@ -5580,6 +5601,18 @@ dependencies = [
"memoffset",
]
[[package]]
name = "nix"
version = "0.30.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "74523f3a35e05aba87a1d978330aef40f67b0304ac79c1c00b294c9830543db6"
dependencies = [
"bitflags 2.13.0",
"cfg-if 1.0.4",
"cfg_aliases",
"libc",
]
[[package]]
name = "nix"
version = "0.31.3"
@@ -9016,6 +9049,164 @@ version = "0.1.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a7973cce6668464ea31f176d85b13c7ab3bba2cb3b77a2ed26abd7801688010a"
[[package]]
name = "symphonia"
version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5773a4c030a19d9bfaa090f49746ff35c75dfddfa700df7a5939d5e076a57039"
dependencies = [
"lazy_static",
"symphonia-bundle-flac",
"symphonia-bundle-mp3",
"symphonia-codec-aac",
"symphonia-codec-alac",
"symphonia-codec-pcm",
"symphonia-codec-vorbis",
"symphonia-core",
"symphonia-format-isomp4",
"symphonia-format-ogg",
"symphonia-format-riff",
"symphonia-metadata",
]
[[package]]
name = "symphonia-bundle-flac"
version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c91565e180aea25d9b80a910c546802526ffd0072d0b8974e3ebe59b686c9976"
dependencies = [
"log",
"symphonia-core",
"symphonia-metadata",
"symphonia-utils-xiph",
]
[[package]]
name = "symphonia-bundle-mp3"
version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4872dd6bb56bf5eac799e3e957aa1981086c3e613b27e0ac23b176054f7c57ed"
dependencies = [
"lazy_static",
"log",
"symphonia-core",
"symphonia-metadata",
]
[[package]]
name = "symphonia-codec-aac"
version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4c263845aa86881416849c1729a54c7f55164f8b96111dba59de46849e73a790"
dependencies = [
"lazy_static",
"log",
"symphonia-core",
]
[[package]]
name = "symphonia-codec-alac"
version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8413fa754942ac16a73634c9dfd1500ed5c61430956b33728567f667fdd393ab"
dependencies = [
"log",
"symphonia-core",
]
[[package]]
name = "symphonia-codec-pcm"
version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4e89d716c01541ad3ebe7c91ce4c8d38a7cf266a3f7b2f090b108fb0cb031d95"
dependencies = [
"log",
"symphonia-core",
]
[[package]]
name = "symphonia-codec-vorbis"
version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f025837c309cd69ffef572750b4a2257b59552c5399a5e49707cc5b1b85d1c73"
dependencies = [
"log",
"symphonia-core",
"symphonia-utils-xiph",
]
[[package]]
name = "symphonia-core"
version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ea00cc4f79b7f6bb7ff87eddc065a1066f3a43fe1875979056672c9ef948c2af"
dependencies = [
"arrayvec",
"bitflags 1.3.2",
"bytemuck",
"lazy_static",
"log",
]
[[package]]
name = "symphonia-format-isomp4"
version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "243739585d11f81daf8dac8d9f3d18cc7898f6c09a259675fc364b382c30e0a5"
dependencies = [
"encoding_rs",
"log",
"symphonia-core",
"symphonia-metadata",
"symphonia-utils-xiph",
]
[[package]]
name = "symphonia-format-ogg"
version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2b4955c67c1ed3aa8ae8428d04ca8397fbef6a19b2b051e73b5da8b1435639cb"
dependencies = [
"log",
"symphonia-core",
"symphonia-metadata",
"symphonia-utils-xiph",
]
[[package]]
name = "symphonia-format-riff"
version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c2d7c3df0e7d94efb68401d81906eae73c02b40d5ec1a141962c592d0f11a96f"
dependencies = [
"extended",
"log",
"symphonia-core",
"symphonia-metadata",
]
[[package]]
name = "symphonia-metadata"
version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "36306ff42b9ffe6e5afc99d49e121e0bd62fe79b9db7b9681d48e29fa19e6b16"
dependencies = [
"encoding_rs",
"lazy_static",
"log",
"symphonia-core",
]
[[package]]
name = "symphonia-utils-xiph"
version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ee27c85ab799a338446b68eec77abf42e1a6f1bb490656e121c6e27bfbab9f16"
dependencies = [
"symphonia-core",
"symphonia-metadata",
]
[[package]]
name = "syn"
version = "1.0.109"
+7
View File
@@ -8,11 +8,18 @@ repository.workspace = true
description = "Reusable local voice primitives for Buzz"
[dependencies]
atomic-write-file = "0.3"
hex = { workspace = true }
ort = { version = "=2.0.0-rc.12", default-features = false, features = ["api-24", "ndarray", "std"] }
ort-sys = { version = "=2.0.0-rc.12", features = ["disable-linking"] }
rand = "0.10"
sentencepiece-model = "0.1"
serde = { version = "1", features = ["derive"] }
serde_json = "1"
sha2 = { workspace = true }
sherpa-onnx = "1.12"
symphonia = { version = "0.5", default-features = false, features = ["aac", "aiff", "alac", "flac", "isomp4", "mp3", "ogg", "pcm", "vorbis", "wav"] }
tokenizers = { version = "0.22", default-features = false, features = ["fancy-regex"] }
[dev-dependencies]
tempfile = "3"
+730
View File
@@ -0,0 +1,730 @@
//! Device-local Pocket reference voice validation, canonicalization, and storage.
use std::{
fs,
io::Write,
path::{Path, PathBuf},
};
use atomic_write_file::AtomicWriteFile;
use serde::{Deserialize, Serialize};
use sha2::{Digest, Sha256};
use symphonia::core::{
audio::SampleBuffer, codecs::DecoderOptions, errors::Error as SymphoniaError,
formats::FormatOptions, io::MediaSourceStream, meta::MetadataOptions, probe::Hint,
};
const MAX_SOURCE_BYTES: u64 = 25 * 1024 * 1024;
const MIN_SAMPLE_RATE: u32 = 8_000;
const MAX_SAMPLE_RATE: u32 = 96_000;
const MIN_DURATION_SECONDS: f64 = 2.0;
const MAX_DURATION_SECONDS: f64 = 30.0;
pub const CANONICAL_SAMPLE_RATE: u32 = 32_000;
const REGISTRY_VERSION: u32 = 1;
const REGISTRY_FILE: &str = "registry.json";
#[derive(Clone, Debug, Deserialize, Serialize, PartialEq, Eq)]
#[serde(rename_all = "camelCase")]
pub struct ImportedVoice {
pub key: String,
pub display_name: String,
pub content_hash: String,
pub file_name: String,
}
#[derive(Default, Deserialize, Serialize)]
#[serde(rename_all = "camelCase")]
struct ImportedVoiceRegistry {
version: u32,
voices: Vec<ImportedVoice>,
}
#[derive(Clone, Debug)]
pub struct PocketVoiceLibrary {
root: PathBuf,
}
impl PocketVoiceLibrary {
pub fn new(root: impl Into<PathBuf>) -> Self {
Self { root: root.into() }
}
pub fn root(&self) -> &Path {
&self.root
}
fn registry_path(&self) -> PathBuf {
self.root.join(REGISTRY_FILE)
}
pub fn load(&self) -> Result<Vec<ImportedVoice>, String> {
let path = self.registry_path();
if !path.exists() {
return Ok(Vec::new());
}
let bytes =
fs::read(&path).map_err(|error| format!("could not read imported voices: {error}"))?;
let registry: ImportedVoiceRegistry = serde_json::from_slice(&bytes)
.map_err(|error| format!("imported voice registry is invalid: {error}"))?;
if registry.version > REGISTRY_VERSION {
return Err(format!(
"imported voice registry version {} is newer than this Buzz build supports",
registry.version
));
}
Ok(registry
.voices
.into_iter()
.filter(valid_identity)
.filter(|voice| self.resolve_file(voice).is_ok())
.collect())
}
fn save(&self, voices: &[ImportedVoice]) -> Result<(), String> {
ensure_storage_dir(&self.root)?;
let payload = serde_json::to_vec_pretty(&ImportedVoiceRegistry {
version: REGISTRY_VERSION,
voices: voices.to_vec(),
})
.map_err(|error| format!("could not encode imported voice registry: {error}"))?;
atomic_write_restricted(&self.registry_path(), &payload)
.map_err(|error| format!("could not save imported voice registry: {error}"))
}
pub fn resolve_file(&self, voice: &ImportedVoice) -> Result<PathBuf, String> {
if !valid_identity(voice) {
return Err("Imported voice registry contains an invalid file identity".to_string());
}
let path = self.root.join(&voice.file_name);
if !is_regular_file_without_symlink(&path) {
return Err(format!("Imported voice {} is missing", voice.display_name));
}
let bytes =
fs::read(&path).map_err(|error| format!("could not verify imported voice: {error}"))?;
if hex::encode(Sha256::digest(bytes)) != voice.content_hash {
return Err(format!(
"Imported voice {} does not match its content identity",
voice.display_name
));
}
Ok(path)
}
pub fn find(&self, key: &str) -> Result<Option<ImportedVoice>, String> {
Ok(self.load()?.into_iter().find(|voice| voice.key == key))
}
pub fn import_path(&self, source: &Path) -> Result<ImportedVoice, String> {
let metadata = fs::metadata(source)
.map_err(|error| format!("could not inspect selected audio: {error}"))?;
if metadata.len() > MAX_SOURCE_BYTES {
return Err("Voice audio must be 25 MB or smaller".to_string());
}
let extension = source
.extension()
.and_then(|extension| extension.to_str())
.map(str::to_ascii_lowercase)
.ok_or_else(|| "Voice audio must have a supported file extension".to_string())?;
let samples = if extension == "wav" {
let source_bytes = fs::read(source)
.map_err(|error| format!("could not read selected audio: {error}"))?;
decode_wav(&source_bytes)?
} else {
decode_media(source, &extension)?
};
let canonical_samples = resample_linear(&samples.samples, samples.sample_rate);
let canonical = encode_pcm16_wav(&canonical_samples, CANONICAL_SAMPLE_RATE);
let hash = hex::encode(Sha256::digest(&canonical));
let key = format!("pocket:imported:{hash}");
let file_name = format!("{hash}.wav");
let display_name = source
.file_stem()
.and_then(|name| name.to_str())
.map(str::trim)
.filter(|name| !name.is_empty())
.unwrap_or("Imported voice")
.chars()
.take(80)
.collect::<String>();
ensure_storage_dir(&self.root)?;
let file_path = self.root.join(&file_name);
let file_created = !file_path.exists();
if file_created {
atomic_write_restricted(&file_path, &canonical)
.map_err(|error| format!("could not save imported voice audio: {error}"))?;
} else {
if !is_regular_file_without_symlink(&file_path) {
return Err("Imported voice storage contains an unsafe file entry".to_string());
}
let existing = fs::read(&file_path)
.map_err(|error| format!("could not verify imported voice audio: {error}"))?;
if hex::encode(Sha256::digest(&existing)) != hash {
return Err("Imported voice storage contains mismatched audio data".to_string());
}
}
let mut imported = ImportedVoice {
key,
display_name,
content_hash: hash,
file_name,
};
let mut voices = self.load()?;
if let Some(existing) = voices
.iter()
.find(|voice| voice.content_hash == imported.content_hash)
{
imported = existing.clone();
} else {
voices.push(imported.clone());
}
if let Err(error) = self.save(&voices) {
if file_created {
let _ = fs::remove_file(&file_path);
}
return Err(error);
}
Ok(imported)
}
pub fn delete(&self, key: &str) -> Result<(), String> {
let mut voices = self.load()?;
let index = voices
.iter()
.position(|voice| voice.key == key)
.ok_or_else(|| format!("Unknown imported voice: {key}"))?;
let previous_voices = voices.clone();
let removed = voices.remove(index);
self.save(&voices)?;
let path = self.root.join(removed.file_name);
match fs::remove_file(path) {
Ok(()) => Ok(()),
Err(error) if error.kind() == std::io::ErrorKind::NotFound => Ok(()),
Err(error) => {
self.save(&previous_voices).map_err(|rollback_error| {
format!(
"Imported voice audio could not be deleted ({error}), and its registry \
entry could not be restored ({rollback_error})"
)
})?;
Err(format!(
"Imported voice audio could not be deleted: {error}"
))
}
}
}
}
#[derive(Clone, Copy, Debug, PartialEq)]
pub struct PcmStats {
pub sample_count: usize,
pub sample_rate: u32,
pub duration_seconds: f64,
pub peak: f32,
pub rms: f32,
pub non_silent_samples: usize,
}
impl PcmStats {
pub fn analyze(samples: &[f32], sample_rate: u32) -> Self {
let peak = samples
.iter()
.filter(|sample| sample.is_finite())
.fold(0.0_f32, |peak, sample| peak.max(sample.abs()));
let square_sum = samples
.iter()
.filter(|sample| sample.is_finite())
.map(|sample| sample * sample)
.sum::<f32>();
let rms = if samples.is_empty() {
0.0
} else {
(square_sum / samples.len() as f32).sqrt()
};
Self {
sample_count: samples.len(),
sample_rate,
duration_seconds: if sample_rate == 0 {
0.0
} else {
samples.len() as f64 / f64::from(sample_rate)
},
peak,
rms,
non_silent_samples: samples
.iter()
.filter(|sample| sample.is_finite() && sample.abs() >= 0.001)
.count(),
}
}
pub fn is_non_silent(self) -> bool {
self.peak >= 0.001 && self.rms >= 0.0001 && self.non_silent_samples > 0
}
}
pub fn write_pcm16_wav(path: &Path, samples: &[f32], sample_rate: u32) -> Result<(), String> {
let bytes = encode_pcm16_wav(samples, sample_rate);
fs::write(path, bytes).map_err(|error| format!("could not write PCM evidence: {error}"))
}
fn ensure_storage_dir(path: &Path) -> Result<(), String> {
fs::create_dir_all(path)
.map_err(|error| format!("could not create local voice storage: {error}"))?;
#[cfg(unix)]
{
use std::os::unix::fs::PermissionsExt;
fs::set_permissions(path, fs::Permissions::from_mode(0o700))
.map_err(|error| format!("could not restrict local voice storage: {error}"))?;
}
Ok(())
}
fn atomic_write_restricted(path: &Path, payload: &[u8]) -> Result<(), String> {
let resolved = fs::canonicalize(path).unwrap_or_else(|_| path.to_path_buf());
let mut file = AtomicWriteFile::open(&resolved)
.map_err(|error| format!("open {} for atomic write: {error}", resolved.display()))?;
#[cfg(unix)]
{
use std::os::unix::fs::PermissionsExt;
file.set_permissions(fs::Permissions::from_mode(0o600))
.map_err(|error| format!("set {} permissions: {error}", resolved.display()))?;
}
file.write_all(payload)
.map_err(|error| format!("write {}: {error}", resolved.display()))?;
file.commit()
.map_err(|error| format!("commit {}: {error}", resolved.display()))
}
fn valid_hash(hash: &str) -> bool {
hash.len() == 64 && hash.bytes().all(|byte| byte.is_ascii_hexdigit())
}
fn valid_identity(voice: &ImportedVoice) -> bool {
valid_hash(&voice.content_hash)
&& voice.key == format!("pocket:imported:{}", voice.content_hash)
&& voice.file_name == format!("{}.wav", voice.content_hash)
}
fn is_regular_file_without_symlink(path: &Path) -> bool {
fs::symlink_metadata(path)
.is_ok_and(|metadata| metadata.file_type().is_file() && !metadata.file_type().is_symlink())
}
#[derive(Debug)]
struct DecodedAudio {
sample_rate: u32,
samples: Vec<f32>,
}
fn decode_wav(bytes: &[u8]) -> Result<DecodedAudio, String> {
if bytes.len() < 12 || &bytes[..4] != b"RIFF" || &bytes[8..12] != b"WAVE" {
return Err("Selected file is not a valid RIFF/WAVE file".to_string());
}
let mut offset = 12usize;
let mut format = None;
let mut data = None;
while offset.checked_add(8).is_some_and(|end| end <= bytes.len()) {
let id = &bytes[offset..offset + 4];
let size =
u32::from_le_bytes(bytes[offset + 4..offset + 8].try_into().unwrap_or([0; 4])) as usize;
let start = offset + 8;
let end = start.checked_add(size).ok_or("WAV chunk size overflow")?;
if end > bytes.len() {
return Err("Selected WAV contains a truncated chunk".to_string());
}
if id == b"fmt " {
format = Some(&bytes[start..end]);
} else if id == b"data" {
data = Some(&bytes[start..end]);
}
offset = end + (size & 1);
}
let format = format.ok_or("Selected WAV has no format chunk")?;
let data = data.ok_or("Selected WAV has no audio data")?;
if format.len() < 16 {
return Err("Selected WAV has an invalid format chunk".to_string());
}
let encoding = u16::from_le_bytes(format[0..2].try_into().unwrap_or([0; 2]));
let encoding = if encoding == 0xfffe && format.len() >= 40 {
u16::from_le_bytes(format[24..26].try_into().unwrap_or([0; 2]))
} else {
encoding
};
let channels = u16::from_le_bytes(format[2..4].try_into().unwrap_or([0; 2]));
let sample_rate = u32::from_le_bytes(format[4..8].try_into().unwrap_or([0; 4]));
let block_align = u16::from_le_bytes(format[12..14].try_into().unwrap_or([0; 2])) as usize;
let bits = u16::from_le_bytes(format[14..16].try_into().unwrap_or([0; 2]));
if channels == 0 || channels > 8 {
return Err("Voice WAV must contain between 1 and 8 channels".to_string());
}
if !(MIN_SAMPLE_RATE..=MAX_SAMPLE_RATE).contains(&sample_rate) {
return Err("Voice WAV sample rate must be between 8 and 96 kHz".to_string());
}
let bytes_per_sample = usize::from(bits.div_ceil(8));
if block_align != bytes_per_sample * usize::from(channels)
|| block_align == 0
|| data.len() % block_align != 0
{
return Err("Voice WAV has invalid sample alignment".to_string());
}
if !matches!((encoding, bits), (1, 8 | 16 | 24 | 32) | (3, 32)) {
return Err("Voice WAV must contain PCM or 32-bit float audio".to_string());
}
let frames = data.len() / block_align;
let duration = frames as f64 / f64::from(sample_rate);
if !(MIN_DURATION_SECONDS..=MAX_DURATION_SECONDS).contains(&duration) {
return Err("Voice WAV must be between 2 and 30 seconds long".to_string());
}
let mut samples = Vec::with_capacity(frames);
for frame in data.chunks_exact(block_align) {
let mut mono = 0.0_f32;
for chunk in frame.chunks_exact(bytes_per_sample) {
let sample = match (encoding, bits) {
(1, 8) => (f32::from(chunk[0]) - 128.0) / 128.0,
(1, 16) => f32::from(i16::from_le_bytes([chunk[0], chunk[1]])) / 32768.0,
(1, 24) => {
let raw = i32::from_le_bytes([
chunk[0],
chunk[1],
chunk[2],
if chunk[2] & 0x80 == 0 { 0 } else { 0xff },
]);
raw as f32 / 8_388_608.0
}
(1, 32) => {
i32::from_le_bytes(chunk.try_into().map_err(|_| "invalid PCM sample")?) as f32
/ 2_147_483_648.0
}
(3, 32) => f32::from_le_bytes(
chunk
.try_into()
.map_err(|_| "invalid floating-point sample")?,
),
_ => unreachable!(),
};
if !sample.is_finite() {
return Err("Voice WAV contains non-finite samples".to_string());
}
mono += sample;
}
samples.push((mono / f32::from(channels)).clamp(-1.0, 1.0));
}
let stats = PcmStats::analyze(&samples, sample_rate);
if !stats.is_non_silent() {
return Err("Voice WAV is silent or too quiet to clone".to_string());
}
Ok(DecodedAudio {
sample_rate,
samples,
})
}
fn decode_media(source: &Path, extension: &str) -> Result<DecodedAudio, String> {
let supported = ["m4a", "mp3", "flac", "ogg", "oga", "aif", "aiff"];
if !supported.contains(&extension) {
return Err(format!(
"Unsupported voice audio format .{extension}. Choose WAV, M4A, MP3, FLAC, OGG, or AIFF"
));
}
let file = fs::File::open(source)
.map_err(|error| format!("could not read selected audio: {error}"))?;
let media = MediaSourceStream::new(Box::new(file), Default::default());
let mut hint = Hint::new();
hint.with_extension(extension);
let probed = symphonia::default::get_probe()
.format(
&hint,
media,
&FormatOptions::default(),
&MetadataOptions::default(),
)
.map_err(|error| format!("could not recognize selected audio: {error}"))?;
let mut format = probed.format;
let track = format
.default_track()
.ok_or_else(|| "Selected audio has no decodable track".to_string())?;
let track_id = track.id;
let mut decoder = symphonia::default::get_codecs()
.make(&track.codec_params, &DecoderOptions::default())
.map_err(|error| format!("could not initialize audio decoder: {error}"))?;
let mut sample_rate = None;
let mut samples = Vec::new();
loop {
let packet = match format.next_packet() {
Ok(packet) => packet,
Err(SymphoniaError::ResetRequired) => {
return Err("Selected audio changes format mid-stream".to_string());
}
Err(SymphoniaError::IoError(error))
if error.kind() == std::io::ErrorKind::UnexpectedEof =>
{
break;
}
Err(error) => return Err(format!("could not read selected audio: {error}")),
};
if packet.track_id() != track_id {
continue;
}
let decoded = match decoder.decode(&packet) {
Ok(decoded) => decoded,
Err(SymphoniaError::DecodeError(_)) => continue,
Err(error) => return Err(format!("could not decode selected audio: {error}")),
};
let spec = *decoded.spec();
if !(MIN_SAMPLE_RATE..=MAX_SAMPLE_RATE).contains(&spec.rate) {
return Err("Voice audio sample rate must be between 8 and 96 kHz".to_string());
}
if sample_rate.is_some_and(|rate| rate != spec.rate) {
return Err("Selected audio changes sample rate mid-stream".to_string());
}
sample_rate = Some(spec.rate);
let channels = spec.channels.count();
if channels == 0 || channels > 8 {
return Err("Voice audio must contain between 1 and 8 channels".to_string());
}
let mut buffer = SampleBuffer::<f32>::new(decoded.capacity() as u64, spec);
buffer.copy_interleaved_ref(decoded);
for frame in buffer.samples().chunks_exact(channels) {
let mono = frame.iter().copied().sum::<f32>() / channels as f32;
if !mono.is_finite() {
return Err("Voice audio contains non-finite samples".to_string());
}
samples.push(mono.clamp(-1.0, 1.0));
}
if samples.len() as f64 > MAX_DURATION_SECONDS * f64::from(spec.rate) {
return Err("Voice audio must be between 2 and 30 seconds long".to_string());
}
}
let sample_rate =
sample_rate.ok_or_else(|| "Selected audio contains no samples".to_string())?;
validate_decoded_audio(&samples, sample_rate)?;
Ok(DecodedAudio {
sample_rate,
samples,
})
}
fn validate_decoded_audio(samples: &[f32], sample_rate: u32) -> Result<(), String> {
let stats = PcmStats::analyze(samples, sample_rate);
if !(MIN_DURATION_SECONDS..=MAX_DURATION_SECONDS).contains(&stats.duration_seconds) {
return Err("Voice audio must be between 2 and 30 seconds long".to_string());
}
if !stats.is_non_silent() {
return Err("Voice audio is silent or too quiet to clone".to_string());
}
Ok(())
}
fn resample_linear(samples: &[f32], source_rate: u32) -> Vec<f32> {
if source_rate == CANONICAL_SAMPLE_RATE {
return samples.to_vec();
}
let output_len = ((samples.len() as u64 * u64::from(CANONICAL_SAMPLE_RATE)
+ u64::from(source_rate) / 2)
/ u64::from(source_rate)) as usize;
(0..output_len)
.map(|index| {
let source = index as f64 * f64::from(source_rate) / f64::from(CANONICAL_SAMPLE_RATE);
let left = source.floor() as usize;
let fraction = (source - left as f64) as f32;
let a = samples[left.min(samples.len() - 1)];
let b = samples[(left + 1).min(samples.len() - 1)];
a + (b - a) * fraction
})
.collect()
}
fn encode_pcm16_wav(samples: &[f32], sample_rate: u32) -> Vec<u8> {
let data_len = (samples.len() * 2) as u32;
let mut bytes = Vec::with_capacity(44 + data_len as usize);
bytes.extend_from_slice(b"RIFF");
bytes.extend_from_slice(&(36 + data_len).to_le_bytes());
bytes.extend_from_slice(b"WAVEfmt ");
bytes.extend_from_slice(&16_u32.to_le_bytes());
bytes.extend_from_slice(&1_u16.to_le_bytes());
bytes.extend_from_slice(&1_u16.to_le_bytes());
bytes.extend_from_slice(&sample_rate.to_le_bytes());
bytes.extend_from_slice(&(sample_rate * 2).to_le_bytes());
bytes.extend_from_slice(&2_u16.to_le_bytes());
bytes.extend_from_slice(&16_u16.to_le_bytes());
bytes.extend_from_slice(b"data");
bytes.extend_from_slice(&data_len.to_le_bytes());
for sample in samples {
let value = (sample.clamp(-1.0, 1.0) * f32::from(i16::MAX)).round() as i16;
bytes.extend_from_slice(&value.to_le_bytes());
}
bytes
}
#[cfg(test)]
mod tests {
use super::*;
fn fixture(sample_rate: u32, seconds: usize, amplitude: f32) -> Vec<u8> {
let samples = (0..sample_rate as usize * seconds)
.map(|index| {
amplitude
* (std::f32::consts::TAU * 220.0 * index as f32 / sample_rate as f32).sin()
})
.collect::<Vec<_>>();
encode_pcm16_wav(&samples, sample_rate)
}
fn stereo_fixture(sample_rate: u32, seconds: usize, amplitude: f32) -> Vec<u8> {
let mono = fixture(sample_rate, seconds, amplitude);
let mono_data = &mono[44..];
let mut stereo_data = Vec::with_capacity(mono_data.len() * 2);
for sample in mono_data.chunks_exact(2) {
stereo_data.extend_from_slice(sample);
stereo_data.extend_from_slice(sample);
}
let mut stereo = mono[..44].to_vec();
stereo[4..8].copy_from_slice(&(36 + stereo_data.len() as u32).to_le_bytes());
stereo[22..24].copy_from_slice(&2_u16.to_le_bytes());
stereo[28..32].copy_from_slice(&(sample_rate * 4).to_le_bytes());
stereo[32..34].copy_from_slice(&4_u16.to_le_bytes());
stereo[40..44].copy_from_slice(&(stereo_data.len() as u32).to_le_bytes());
stereo.extend_from_slice(&stereo_data);
stereo
}
#[test]
fn imports_persists_reloads_and_deletes_canonical_voice() {
let temp = tempfile::tempdir().expect("temp voice workspace");
let source = temp.path().join("My voice.wav");
fs::write(&source, fixture(44_100, 2, 0.5)).expect("write source");
let library = PocketVoiceLibrary::new(temp.path().join("library"));
let imported = library.import_path(&source).expect("import voice");
assert!(imported.key.starts_with("pocket:imported:"));
assert_eq!(imported.display_name, "My voice");
let relaunched = PocketVoiceLibrary::new(library.root());
assert_eq!(
relaunched.load().expect("reload registry"),
vec![imported.clone()]
);
let stored = relaunched
.resolve_file(&imported)
.expect("resolve stored voice");
let decoded = decode_wav(&fs::read(&stored).expect("read stored voice"))
.expect("decode canonical voice");
assert_eq!(decoded.sample_rate, CANONICAL_SAMPLE_RATE);
assert_eq!(decoded.samples.len(), CANONICAL_SAMPLE_RATE as usize * 2);
assert_eq!(
relaunched.import_path(&source).expect("idempotent import"),
imported
);
assert_eq!(relaunched.load().expect("deduplicated registry").len(), 1);
relaunched.delete(&imported.key).expect("delete voice");
assert!(relaunched.load().expect("empty registry").is_empty());
assert!(!stored.exists());
}
#[test]
fn common_stereo_audio_is_downmixed_to_canonical_mono() {
let temp = tempfile::tempdir().expect("temp voice workspace");
let source = temp.path().join("stereo.wav");
fs::write(&source, stereo_fixture(44_100, 2, 0.5)).expect("write stereo");
let library = PocketVoiceLibrary::new(temp.path().join("library"));
let imported = library.import_path(&source).expect("import stereo");
let stored = library
.resolve_file(&imported)
.expect("resolve stored voice");
let decoded = decode_wav(&fs::read(stored).expect("read stored voice"))
.expect("decode canonical voice");
assert_eq!(decoded.sample_rate, CANONICAL_SAMPLE_RATE);
assert_eq!(decoded.samples.len(), CANONICAL_SAMPLE_RATE as usize * 2);
}
#[test]
#[ignore = "requires BUZZ_VOICE_IMPORT_TEST_DIR with common-format fixtures"]
fn imports_common_audio_format_fixtures() {
let fixtures =
PathBuf::from(std::env::var("BUZZ_VOICE_IMPORT_TEST_DIR").expect("fixture directory"));
let temp = tempfile::tempdir().expect("temp voice workspace");
let library = PocketVoiceLibrary::new(temp.path().join("library"));
for file_name in [
"voice.wav",
"voice.m4a",
"voice.mp3",
"voice.flac",
"voice.ogg",
"voice.aiff",
] {
let imported = library
.import_path(&fixtures.join(file_name))
.unwrap_or_else(|error| panic!("import {file_name}: {error}"));
let stored = library
.resolve_file(&imported)
.unwrap_or_else(|error| panic!("resolve {file_name}: {error}"));
let decoded = decode_wav(&fs::read(stored).expect("read canonical voice"))
.expect("decode canonical voice");
assert_eq!(decoded.sample_rate, CANONICAL_SAMPLE_RATE);
assert!(decoded.samples.len() >= CANONICAL_SAMPLE_RATE as usize * 2);
}
}
#[test]
fn invalid_unsupported_and_silent_files_do_not_mutate_registry() {
let temp = tempfile::tempdir().expect("temp voice workspace");
let library = PocketVoiceLibrary::new(temp.path().join("library"));
let garbage = temp.path().join("garbage.wav");
fs::write(&garbage, b"not a wave").expect("write garbage");
assert!(library
.import_path(&garbage)
.expect_err("garbage rejected")
.contains("RIFF/WAVE"));
let silent = temp.path().join("silent.wav");
fs::write(&silent, fixture(32_000, 2, 0.0)).expect("write silence");
assert!(library
.import_path(&silent)
.expect_err("silence rejected")
.contains("silent"));
let unsupported_container = temp.path().join("voice.txt");
fs::write(&unsupported_container, b"not audio").expect("write unsupported container");
assert!(library
.import_path(&unsupported_container)
.expect_err("container rejected")
.contains("Unsupported voice audio format"));
let mut unsupported = fixture(32_000, 2, 0.5);
unsupported[20..22].copy_from_slice(&6_u16.to_le_bytes());
let unsupported_path = temp.path().join("unsupported.wav");
fs::write(&unsupported_path, unsupported).expect("write unsupported");
assert!(library
.import_path(&unsupported_path)
.expect_err("unsupported rejected")
.contains("PCM or 32-bit float"));
assert!(library.load().expect("unchanged registry").is_empty());
}
#[test]
fn pcm_analysis_distinguishes_signal_from_silence() {
let signal = (0..24_000)
.map(|index| (std::f32::consts::TAU * 440.0 * index as f32 / 24_000.0).sin() * 0.5)
.collect::<Vec<_>>();
let signal_stats = PcmStats::analyze(&signal, 24_000);
assert!(signal_stats.is_non_silent());
assert_eq!(signal_stats.duration_seconds, 1.0);
assert!(signal_stats.peak > 0.49);
assert!(signal_stats.rms > 0.3);
let silence = vec![0.0; 24_000];
assert!(!PcmStats::analyze(&silence, 24_000).is_non_silent());
}
}
+1
View File
@@ -1,5 +1,6 @@
//! Reusable local voice primitives for Buzz.
pub mod imported;
pub mod pocket;
pub use pocket::{
@@ -0,0 +1,133 @@
use std::{
fs,
path::{Path, PathBuf},
};
use buzz_voice::{
imported::{write_pcm16_wav, PcmStats, PocketVoiceLibrary},
pocket::{load_text_to_speech, load_voice_style, DEFAULT_VOICE, SAMPLE_RATE, VOICE_FILE_EXT},
};
const PREVIEW_TEXT: &str = "This is an objective Pocket voice preview.";
fn required_path(name: &str) -> PathBuf {
std::env::var_os(name)
.map(PathBuf::from)
.unwrap_or_else(|| panic!("{name} must point to the required local test path"))
}
fn checked_in_voice() -> PathBuf {
Path::new(env!("CARGO_MANIFEST_DIR"))
.join("../../desktop/src-tauri/resources/pocket-voices/eve.wav")
}
fn evidence_dir() -> PathBuf {
std::env::var_os("BUZZ_VOICE_EVIDENCE_DIR")
.map(PathBuf::from)
.unwrap_or_else(|| {
Path::new(env!("CARGO_MANIFEST_DIR")).join("../../target/buzz-voice-evidence")
})
}
fn synthesize(model_dir: &Path, voice_path: &Path, text: &str) -> (Vec<f32>, PcmStats) {
let engine = load_text_to_speech(
model_dir
.to_str()
.expect("Pocket model path must be valid UTF-8"),
)
.expect("load Pocket model");
let style = load_voice_style(voice_path).expect("load selected voice");
let samples = engine
.synth_chunk(text, "en", &style, 1)
.expect("synthesize preview");
let stats = PcmStats::analyze(&samples, SAMPLE_RATE);
assert!(
stats.is_non_silent(),
"generated PCM must be non-silent: {stats:?}"
);
assert!(
stats.duration_seconds > 0.2,
"generated PCM is unexpectedly short: {stats:?}"
);
(samples, stats)
}
#[test]
#[ignore = "requires BUZZ_POCKET_MODEL_DIR and runs the installed Pocket ONNX model"]
fn objective_import_synthesis_delete_and_mary_fallback() {
let model_dir = required_path("BUZZ_POCKET_MODEL_DIR");
let temp = tempfile::tempdir().expect("temporary voice workspace");
let source = temp.path().join("Imported Eve.wav");
fs::copy(checked_in_voice(), &source).expect("copy checked-in voice fixture");
let library_root = temp.path().join("library");
let library = PocketVoiceLibrary::new(&library_root);
let imported = library.import_path(&source).expect("import valid WAV");
assert_eq!(
library.find(&imported.key).expect("read selection"),
Some(imported.clone())
);
drop(library);
let relaunched = PocketVoiceLibrary::new(&library_root);
let selected = relaunched
.find(&imported.key)
.expect("reload persisted selection")
.expect("selected imported voice survived relaunch");
let imported_path = relaunched
.resolve_file(&selected)
.expect("resolve persisted imported voice");
let (imported_pcm, imported_stats) = synthesize(&model_dir, &imported_path, PREVIEW_TEXT);
let evidence = evidence_dir();
fs::create_dir_all(&evidence).expect("create evidence directory");
let imported_wav = evidence.join("imported-preview.wav");
write_pcm16_wav(&imported_wav, &imported_pcm, SAMPLE_RATE)
.expect("write imported preview evidence");
relaunched
.delete(&imported.key)
.expect("delete imported voice");
assert_eq!(
relaunched.find(&imported.key).expect("reload after delete"),
None
);
let mary_path = model_dir.join(format!("{DEFAULT_VOICE}.{VOICE_FILE_EXT}"));
assert_eq!(
mary_path.file_name().and_then(|name| name.to_str()),
Some("reference_sample.wav"),
"fallback must remain the deterministic Mary reference"
);
let (mary_pcm, mary_stats) = synthesize(&model_dir, &mary_path, PREVIEW_TEXT);
let mary_wav = evidence.join("mary-fallback-preview.wav");
write_pcm16_wav(&mary_wav, &mary_pcm, SAMPLE_RATE)
.expect("write Mary fallback preview evidence");
println!(
"{}",
serde_json::json!({
"importedKey": imported.key,
"persistence": "reloaded",
"afterDelete": "pocket:mary",
"importedPreview": {
"path": imported_wav,
"samples": imported_stats.sample_count,
"sampleRate": imported_stats.sample_rate,
"durationSeconds": imported_stats.duration_seconds,
"peak": imported_stats.peak,
"rms": imported_stats.rms,
"nonSilentSamples": imported_stats.non_silent_samples,
},
"maryFallbackPreview": {
"path": mary_wav,
"samples": mary_stats.sample_count,
"sampleRate": mary_stats.sample_rate,
"durationSeconds": mary_stats.duration_seconds,
"peak": mary_stats.peak,
"rms": mary_stats.rms,
"nonSilentSamples": mary_stats.non_silent_samples,
}
})
);
}
+15
View File
@@ -1178,13 +1178,17 @@ dependencies = [
name = "buzz-voice"
version = "0.1.0"
dependencies = [
"atomic-write-file",
"hex",
"ort",
"ort-sys",
"rand 0.10.2",
"sentencepiece-model",
"serde",
"serde_json",
"sha2 0.11.0",
"sherpa-onnx",
"symphonia",
"tokenizers",
]
@@ -9778,6 +9782,7 @@ dependencies = [
"symphonia-bundle-flac",
"symphonia-bundle-mp3",
"symphonia-codec-aac",
"symphonia-codec-alac",
"symphonia-codec-pcm",
"symphonia-codec-vorbis",
"symphonia-core",
@@ -9822,6 +9827,16 @@ dependencies = [
"symphonia-core",
]
[[package]]
name = "symphonia-codec-alac"
version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8413fa754942ac16a73634c9dfd1500ed5c61430956b33728567f667fdd393ab"
dependencies = [
"log",
"symphonia-core",
]
[[package]]
name = "symphonia-codec-pcm"
version = "0.5.5"
+1
View File
@@ -39,6 +39,7 @@ pub mod stt;
pub mod transcription;
pub mod tts;
pub mod tts_settings;
mod tts_voice_import;
mod tts_voice_registry;
pub mod wire;
+45 -18
View File
@@ -392,6 +392,40 @@ pub(crate) async fn maybe_start_tts_pipeline(state: &AppState) -> Result<bool, S
None => return Ok(false),
};
// Avoid resolving and hashing imported voice files on every hot-start poll
// when TTS is already disabled or running. The guarded claim below repeats
// these checks after the fallible work to close the race.
{
let huddle = state.huddle()?;
if huddle.tts_pipeline.is_some() || !huddle.tts_enabled {
return Ok(false);
}
}
// Resolve all fallible construction inputs before claiming the sentinel so
// an unreadable optional voice registry cannot wedge future start attempts.
let output_device = state
.huddle_audio
.output_device
.lock()
.unwrap_or_else(|e| e.into_inner())
.clone();
let app = state
.app_handle
.lock()
.map_err(|error| format!("app handle lock poisoned: {error}"))?
.clone();
let voice_preferences = state
.huddle_audio
.tts
.lock()
.map_err(|error| format!("text-to-speech settings lock poisoned: {error}"))
.map(|settings| settings.voice_preferences.clone())?;
let initial_voice = match app {
Some(app) => super::tts_settings::pocket_voice_reference(&app, &voice_preferences)?,
None => super::tts_settings::bundled_pocket_voice_reference(&voice_preferences),
};
// Atomically check preconditions and claim the construction slot.
// The sentinel prevents a second caller from starting construction
// while we're building outside the lock.
@@ -416,20 +450,6 @@ pub(crate) async fn maybe_start_tts_pipeline(state: &AppState) -> Result<bool, S
// Construct outside the lock — this spawns the TTS worker thread and
// loads ONNX sessions (~200ms). If this fails, clear the sentinel.
let output_device = state
.huddle_audio
.output_device
.lock()
.unwrap_or_else(|e| e.into_inner())
.clone();
let initial_voice = state
.huddle_audio
.tts
.lock()
.map_err(|error| format!("text-to-speech settings lock poisoned: {error}"))
.map(|settings| {
super::tts_settings::pocket_voice_name(&settings.voice_preferences).to_string()
})?;
let constructed_voice = initial_voice.clone();
let constructed = tokio::task::spawn_blocking(move || {
tts::TtsPipeline::new_with_voice(
@@ -513,14 +533,21 @@ fn finalize_tts_pipeline_start(
{
return Ok(false);
}
let voice = state
let app = state
.app_handle
.lock()
.map_err(|error| format!("app handle lock poisoned: {error}"))?
.clone();
let preferences = state
.huddle_audio
.tts
.lock()
.map_err(|error| format!("text-to-speech settings lock poisoned: {error}"))
.map(|settings| {
super::tts_settings::pocket_voice_name(&settings.voice_preferences).to_string()
})?;
.map(|settings| settings.voice_preferences.clone())?;
let voice = match app {
Some(app) => super::tts_settings::pocket_voice_reference(&app, &preferences)?,
None => super::tts_settings::bundled_pocket_voice_reference(&preferences),
};
publish(&voice, &mut huddle);
Ok(true)
}
+176 -36
View File
@@ -81,12 +81,8 @@ pub struct VoiceProvenance {
/// may have that backend installed. Resolution is always local.
pub type VoicePreferences = Vec<String>;
/// Cross-backend registry for voices known to this client.
///
/// V1 contains Pocket entries only. Siri, Kokoro, imported voices, and
/// per-agent assignment can add entries or reuse the preference type without
/// changing the registry/settings boundary.
pub fn voice_registry() -> Vec<VoiceRegistryEntry> {
/// Bundled Pocket voices available without local imports.
pub fn bundled_voice_registry() -> Vec<VoiceRegistryEntry> {
POCKET_VOICES
.iter()
.map(|voice| VoiceRegistryEntry {
@@ -107,6 +103,34 @@ pub fn voice_registry() -> Vec<VoiceRegistryEntry> {
.collect()
}
/// Cross-backend registry of bundled and locally installed voices.
pub fn voice_registry(app: &AppHandle) -> Vec<VoiceRegistryEntry> {
let mut registry = bundled_voice_registry();
match super::tts_voice_import::load_registry(app) {
Ok(imported) => registry.extend(imported.into_iter().map(|voice| VoiceRegistryEntry {
key: voice.key,
display_name: voice.display_name,
backend: POCKET_BACKEND_ID.to_string(),
backend_name: "Pocket TTS".to_string(),
availability: VOICE_AVAILABILITY_INSTALLED.to_string(),
fallback_key: Some(MARY_VOICE_KEY.to_string()),
reference_file: Some(voice.file_name),
provenance: VoiceProvenance {
source: "local import".to_string(),
content_hash: Some(voice.content_hash),
license: None,
source_url: None,
},
})),
Err(error) => {
eprintln!(
"buzz-desktop: {error}; imported Pocket voices are unavailable for this session"
);
}
}
registry
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
#[serde(rename_all = "camelCase")]
pub struct TtsSettings {
@@ -125,8 +149,10 @@ impl Default for TtsSettings {
}
}
pub fn voice_by_key(key: &str) -> Option<VoiceRegistryEntry> {
voice_registry().into_iter().find(|voice| voice.key == key)
pub fn voice_by_key(app: &AppHandle, key: &str) -> Option<VoiceRegistryEntry> {
voice_registry(app)
.into_iter()
.find(|voice| voice.key == key)
}
fn is_qualified_voice_key(key: &str) -> bool {
@@ -141,11 +167,19 @@ fn is_locally_available(availability: &str) -> bool {
)
}
#[cfg(test)]
pub fn resolve_voice_for_backend(
preferences: &[String],
backend: &str,
) -> Result<VoiceRegistryEntry, String> {
let registry = voice_registry();
resolve_voice_for_backend_in_registry(preferences, backend, &bundled_voice_registry())
}
fn resolve_voice_for_backend_in_registry(
preferences: &[String],
backend: &str,
registry: &[VoiceRegistryEntry],
) -> Result<VoiceRegistryEntry, String> {
preferences
.iter()
.filter_map(|key| registry.iter().find(|voice| voice.key == *key))
@@ -161,14 +195,31 @@ pub fn resolve_voice_for_backend(
.ok_or_else(|| format!("No locally available fallback voice for backend {backend}"))
}
pub fn pocket_voice_name(preferences: &[String]) -> String {
resolve_voice_for_backend(preferences, POCKET_BACKEND_ID)
pub fn bundled_pocket_voice_reference(preferences: &[String]) -> String {
resolve_voice_for_backend_in_registry(preferences, POCKET_BACKEND_ID, &bundled_voice_registry())
.ok()
.and_then(|voice| voice.reference_file)
.and_then(|file| file.strip_suffix(".wav").map(str::to_string))
.unwrap_or_else(|| DEFAULT_VOICE.to_string())
}
pub fn pocket_voice_reference(app: &AppHandle, preferences: &[String]) -> Result<String, String> {
let registry = voice_registry(app);
let voice = resolve_voice_for_backend_in_registry(preferences, POCKET_BACKEND_ID, &registry)?;
if voice.key.starts_with("pocket:imported:") {
let imported = super::tts_voice_import::load_registry(app)?
.into_iter()
.find(|candidate| candidate.key == voice.key)
.ok_or_else(|| format!("Imported voice {} is unavailable", voice.display_name))?;
return super::tts_voice_import::resolve_file(app, &imported)
.map(|path| path.to_string_lossy().into_owned());
}
Ok(voice
.reference_file
.and_then(|file| file.strip_suffix(".wav").map(str::to_string))
.unwrap_or_else(|| DEFAULT_VOICE.to_string()))
}
pub(crate) fn settings_path(app: &AppHandle) -> Result<PathBuf, String> {
app.path()
.app_data_dir()
@@ -290,8 +341,8 @@ pub fn get_tts_settings(state: State<'_, AppState>) -> Result<TtsSettings, Strin
}
#[tauri::command]
pub fn list_voice_registry() -> Vec<VoiceRegistryEntry> {
voice_registry()
pub fn list_voice_registry(app: AppHandle) -> Vec<VoiceRegistryEntry> {
voice_registry(&app)
}
fn ensure_settings_writable(state: &AppState) -> Result<(), String> {
@@ -394,8 +445,8 @@ async fn apply_tts_settings(
if settings.agent_text_to_speech {
let (active, voice_change_ack) = {
let mut huddle = state.huddle()?;
let voice_change_ack =
enable_tts_runtime(&mut huddle, &pocket_voice_name(&settings.voice_preferences));
let voice_reference = pocket_voice_reference(app, &settings.voice_preferences)?;
let voice_change_ack = enable_tts_runtime(&mut huddle, &voice_reference);
(
matches!(huddle.phase, HuddlePhase::Connected | HuddlePhase::Active),
voice_change_ack,
@@ -431,6 +482,14 @@ async fn finish_voice_change(voice_change: Option<VoiceChangeWait>) -> Result<()
.await
}
async fn finish_durable_voice_change(voice_change: Option<VoiceChangeWait>) {
if let Err(error) = finish_voice_change(voice_change).await {
eprintln!(
"buzz-desktop: tts stage=voice_switch status=delayed reason=ack_timeout error={error}"
);
}
}
async fn wait_for_voice_change_ack(
mut acknowledged: tokio::sync::oneshot::Receiver<()>,
timeout: Duration,
@@ -479,10 +538,22 @@ pub async fn set_tts_enabled(
}
fn settings_with_pocket_voice(
settings: TtsSettings,
voice_key: &str,
app: &AppHandle,
) -> Result<TtsSettings, String> {
settings_with_pocket_voice_from_registry(settings, voice_key, &voice_registry(app))
}
fn settings_with_pocket_voice_from_registry(
mut settings: TtsSettings,
voice_key: &str,
registry: &[VoiceRegistryEntry],
) -> Result<TtsSettings, String> {
let voice = voice_by_key(voice_key).ok_or_else(|| format!("Unknown voice: {voice_key}"))?;
let voice = registry
.iter()
.find(|voice| voice.key == voice_key)
.ok_or_else(|| format!("Unknown voice: {voice_key}"))?;
if voice.backend != POCKET_BACKEND_ID || !is_locally_available(&voice.availability) {
return Err("The selected Pocket voice is not available on this device".to_string());
}
@@ -515,26 +586,21 @@ pub async fn set_pocket_voice(
.lock()
.map_err(|error| format!("text-to-speech settings lock poisoned: {error}"))?
.clone();
let settings = settings_with_pocket_voice(settings, &voice_key)?;
let settings = settings_with_pocket_voice(settings, &voice_key, &app)?;
let voice_change = apply_tts_settings(settings, &app, &state).await?;
drop(transition);
if let Err(error) = finish_voice_change(voice_change).await {
// The preference is already durable. Report the delayed live
// transition diagnostically without telling the UI that saving failed;
// the next pipeline start resolves the persisted voice normally.
eprintln!(
"buzz-desktop: tts stage=voice_switch status=delayed reason=ack_timeout error={error}"
);
}
finish_durable_voice_change(voice_change).await;
current_settings(&state)
}
#[tauri::command]
pub async fn preview_pocket_voice(
voice_key: String,
app: AppHandle,
state: State<'_, AppState>,
) -> Result<(), String> {
let voice = voice_by_key(&voice_key).ok_or_else(|| format!("Unknown voice: {voice_key}"))?;
let voice =
voice_by_key(&app, &voice_key).ok_or_else(|| format!("Unknown voice: {voice_key}"))?;
if voice.backend != POCKET_BACKEND_ID {
return Err("Only Pocket voices can be previewed in this build".to_string());
}
@@ -548,10 +614,7 @@ pub async fn preview_pocket_voice(
.lock()
.unwrap_or_else(|error| error.into_inner())
.clone();
let voice_name = voice
.reference_file
.and_then(|file| file.strip_suffix(".wav").map(str::to_string))
.ok_or_else(|| format!("Voice {voice_key} has no local Pocket reference file"))?;
let voice_name = pocket_voice_reference(&app, std::slice::from_ref(&voice_key))?;
tokio::task::spawn_blocking(move || {
let active = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false));
let cancel = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false));
@@ -579,6 +642,69 @@ pub async fn preview_pocket_voice(
.map_err(|error| format!("Voice preview task failed: {error}"))?
}
#[derive(Debug, Clone, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct TtsVoiceMutation {
pub settings: TtsSettings,
pub registry: Vec<VoiceRegistryEntry>,
}
#[tauri::command]
pub async fn import_pocket_voice(
app: AppHandle,
state: State<'_, AppState>,
) -> Result<Option<TtsVoiceMutation>, String> {
let Some(imported) = super::tts_voice_import::pick_and_import(&app).await? else {
return Ok(None);
};
let transition = state.huddle_audio.tts_transition.lock().await;
let settings = current_settings(&state)?;
let settings = settings_with_pocket_voice(settings, &imported.key, &app)?;
let voice_change = apply_tts_settings(settings, &app, &state).await?;
drop(transition);
finish_durable_voice_change(voice_change).await;
Ok(Some(TtsVoiceMutation {
settings: current_settings(&state)?,
registry: voice_registry(&app),
}))
}
#[tauri::command]
pub async fn delete_pocket_voice(
voice_key: String,
app: AppHandle,
state: State<'_, AppState>,
) -> Result<TtsVoiceMutation, String> {
if !voice_key.starts_with("pocket:imported:") {
return Err("Bundled voices cannot be deleted".to_string());
}
if voice_by_key(&app, &voice_key).is_none() {
return Err(format!("Unknown imported voice: {voice_key}"));
}
let transition = state.huddle_audio.tts_transition.lock().await;
let current = current_settings(&state)?;
let selected = resolve_voice_for_backend_in_registry(
&current.voice_preferences,
POCKET_BACKEND_ID,
&voice_registry(&app),
)
.is_ok_and(|voice| voice.key == voice_key);
let voice_change = if selected {
let fallback = settings_with_pocket_voice(current, MARY_VOICE_KEY, &app)?;
apply_tts_settings(fallback, &app, &state).await?
} else {
None
};
drop(transition);
finish_durable_voice_change(voice_change).await;
super::tts_voice_import::delete(&app, &voice_key)?;
Ok(TtsVoiceMutation {
settings: current_settings(&state)?,
registry: voice_registry(&app),
})
}
#[cfg(test)]
mod tests {
use super::*;
@@ -619,7 +745,7 @@ mod tests {
#[test]
fn registry_has_all_official_english_vctk_presets() {
assert_eq!(
voice_registry()
bundled_voice_registry()
.iter()
.map(|voice| {
(
@@ -680,7 +806,7 @@ mod tests {
fn identity_is_qualified_key_not_display_label() {
assert!(is_qualified_voice_key("pocket:imported:audio-content-hash"));
assert_ne!(MARY_VOICE_KEY, EVE_VOICE_KEY);
let mut registry = voice_registry();
let mut registry = bundled_voice_registry();
registry[0].display_name = "Jim".to_string();
registry[1].display_name = "Jim".to_string();
assert_eq!(registry[0].display_name, registry[1].display_name);
@@ -792,7 +918,12 @@ mod tests {
voice_preferences: vec!["siri:aaron".to_string(), MARY_VOICE_KEY.to_string()],
..TtsSettings::default()
};
let updated = settings_with_pocket_voice(current, EVE_VOICE_KEY).expect("available voice");
let updated = settings_with_pocket_voice_from_registry(
current,
EVE_VOICE_KEY,
&bundled_voice_registry(),
)
.expect("available voice");
assert!(!updated.agent_text_to_speech);
assert_eq!(updated.voice_preferences, vec!["siri:aaron", EVE_VOICE_KEY]);
}
@@ -805,8 +936,12 @@ mod tests {
// This models the next command after the OFF save fails: it must merge
// from effective memory state, not the stale last-persisted ON value.
let current = state.huddle_audio.tts.lock().expect("settings").clone();
let voice_update =
settings_with_pocket_voice(current, EVE_VOICE_KEY).expect("available voice");
let voice_update = settings_with_pocket_voice_from_registry(
current,
EVE_VOICE_KEY,
&bundled_voice_registry(),
)
.expect("available voice");
assert!(!voice_update.agent_text_to_speech);
}
@@ -820,7 +955,12 @@ mod tests {
.expect("settings")
.agent_text_to_speech = false;
let current = state.huddle_audio.tts.lock().expect("settings").clone();
let unsaved = settings_with_pocket_voice(current, EVE_VOICE_KEY).expect("available voice");
let unsaved = settings_with_pocket_voice_from_registry(
current,
EVE_VOICE_KEY,
&bundled_voice_registry(),
)
.expect("available voice");
// This is the only pre-persistence mutation for an OFF candidate.
commit_effective_off(&state).expect("commit effective OFF state");
@@ -0,0 +1,59 @@
//! Tauri native-picker adapter for the reusable local Pocket voice library.
use std::path::PathBuf;
use buzz_voice_pkg::imported::{ImportedVoice, PocketVoiceLibrary};
use tauri::{AppHandle, Manager};
pub fn voices_dir(app: &AppHandle) -> Result<PathBuf, String> {
app.path()
.app_data_dir()
.map(|path| path.join("tts").join("pocket-voices"))
.map_err(|error| format!("could not locate local voice storage: {error}"))
}
fn library(app: &AppHandle) -> Result<PocketVoiceLibrary, String> {
voices_dir(app).map(PocketVoiceLibrary::new)
}
pub fn load_registry(app: &AppHandle) -> Result<Vec<ImportedVoice>, String> {
library(app)?.load()
}
pub fn resolve_file(app: &AppHandle, voice: &ImportedVoice) -> Result<PathBuf, String> {
library(app)?.resolve_file(voice)
}
pub async fn pick_and_import(app: &AppHandle) -> Result<Option<ImportedVoice>, String> {
use tauri_plugin_dialog::DialogExt;
let (sender, receiver) = tokio::sync::oneshot::channel();
app.dialog()
.file()
.add_filter(
"Audio",
&["wav", "m4a", "mp3", "flac", "ogg", "oga", "aif", "aiff"],
)
.pick_file(move |path| {
let _ = sender.send(path);
});
let Some(file_path) = receiver
.await
.map_err(|_| "voice picker closed unexpectedly".to_string())?
else {
return Ok(None);
};
let path = file_path
.as_path()
.ok_or("the selected voice path is invalid")?
.to_path_buf();
let voice_library = library(app)?;
tokio::task::spawn_blocking(move || voice_library.import_path(&path))
.await
.map_err(|error| format!("voice import task failed: {error}"))?
.map(Some)
}
pub fn delete(app: &AppHandle, key: &str) -> Result<(), String> {
library(app)?.delete(key)
}
@@ -120,7 +120,7 @@ pub(super) fn reconcile_selected_voice(
return true;
}
let requested_path = model_dir.join(format!("{requested_voice}.{VOICE_FILE_EXT}"));
let requested_path = voice_path(model_dir, &requested_voice);
match load_voice_style(&requested_path) {
Ok(requested_style) => {
*style = requested_style;
@@ -151,6 +151,15 @@ pub(super) fn reconcile_selected_voice(
}
}
pub(super) fn voice_path(model_dir: &Path, voice: &str) -> std::path::PathBuf {
let path = Path::new(voice);
if path.is_absolute() {
path.to_path_buf()
} else {
model_dir.join(format!("{voice}.{VOICE_FILE_EXT}"))
}
}
pub(super) fn retain_cancelled_text(
deferred_text: &mut VecDeque<QueuedText>,
current_text: &mut Option<QueuedText>,
+2
View File
@@ -893,6 +893,8 @@ pub fn run() {
huddle::tts_settings::list_voice_registry,
huddle::tts_settings::set_pocket_voice,
huddle::tts_settings::preview_pocket_voice,
huddle::tts_settings::import_pocket_voice,
huddle::tts_settings::delete_pocket_voice,
speak_agent_message,
add_agent_to_huddle,
check_pipeline_hotstart,
@@ -1,9 +1,19 @@
import * as React from "react";
import { ChevronDown, Play, Volume2 } from "lucide-react";
import { ChevronDown, Play, Trash2, Upload, Volume2 } from "lucide-react";
import { invokeTauri } from "@/shared/api/tauri";
import { cn } from "@/shared/lib/cn";
import { Button } from "@/shared/ui/button";
import {
AlertDialog,
AlertDialogAction,
AlertDialogCancel,
AlertDialogContent,
AlertDialogDescription,
AlertDialogFooter,
AlertDialogHeader,
AlertDialogTitle,
} from "@/shared/ui/alert-dialog";
import {
DropdownMenu,
DropdownMenuContent,
@@ -27,11 +37,18 @@ export type TtsSettings = {
voicePreferences: string[];
};
type TtsVoiceMutation = {
settings: TtsSettings;
registry: VoiceRegistryEntry[];
};
export function VoiceSettingsCard() {
const [settings, setSettings] = React.useState<TtsSettings | null>(null);
const [registry, setRegistry] = React.useState<VoiceRegistryEntry[]>([]);
const [busy, setBusy] = React.useState(false);
const [previewing, setPreviewing] = React.useState(false);
const [deleteCandidate, setDeleteCandidate] =
React.useState<VoiceRegistryEntry | null>(null);
const [error, setError] = React.useState<string | null>(null);
React.useEffect(() => {
@@ -111,6 +128,50 @@ export function VoiceSettingsCard() {
}
}, []);
const importPocketVoice = React.useCallback(async () => {
setBusy(true);
setError(null);
try {
const result = await invokeTauri<TtsVoiceMutation | null>(
"import_pocket_voice",
);
if (result) {
setSettings(result.settings);
setRegistry(result.registry);
}
} catch (importError) {
setError(
importError instanceof Error
? importError.message
: "Voice could not be imported.",
);
} finally {
setBusy(false);
}
}, []);
const deletePocketVoice = React.useCallback(async (voiceKey: string) => {
setBusy(true);
setError(null);
try {
const result = await invokeTauri<TtsVoiceMutation>(
"delete_pocket_voice",
{ voiceKey },
);
setSettings(result.settings);
setRegistry(result.registry);
setDeleteCandidate(null);
} catch (deleteError) {
setError(
deleteError instanceof Error
? deleteError.message
: "Voice could not be deleted.",
);
} finally {
setBusy(false);
}
}, []);
const voices = voicesForBackend(registry, "pocket");
const selectedVoice = selectedVoiceForBackend(
settings?.voicePreferences ?? [],
@@ -236,6 +297,28 @@ export function VoiceSettingsCard() {
)}
Preview
</Button>
<Button
data-testid="pocket-voice-import"
disabled={controlsDisabled}
onClick={() => void importPocketVoice()}
size="sm"
variant="outline"
>
<Upload className="h-4 w-4" />
Add voice
</Button>
{selectedVoice?.key.startsWith("pocket:imported:") && (
<Button
aria-label={`Delete ${selectedVoice.displayName}`}
data-testid="pocket-voice-delete"
disabled={controlsDisabled}
onClick={() => setDeleteCandidate(selectedVoice)}
size="icon"
variant="ghost"
>
<Trash2 className="h-4 w-4" />
</Button>
)}
</div>
</SettingsOptionRow>
</SettingsOptionGroup>
@@ -251,6 +334,41 @@ export function VoiceSettingsCard() {
</p>
)}
</div>
<AlertDialog
onOpenChange={(open) => {
if (!open) setDeleteCandidate(null);
}}
open={deleteCandidate !== null}
>
<AlertDialogContent>
<AlertDialogHeader>
<AlertDialogTitle>Delete imported voice?</AlertDialogTitle>
<AlertDialogDescription>
{deleteCandidate
? `${deleteCandidate.displayName} and its local audio file will be removed.`
: "This imported voice and its local audio file will be removed."}
{selectedVoice?.key === deleteCandidate?.key &&
" Mary will be selected instead."}
</AlertDialogDescription>
</AlertDialogHeader>
<AlertDialogFooter>
<AlertDialogCancel disabled={busy}>Cancel</AlertDialogCancel>
<AlertDialogAction
className="bg-destructive text-destructive-foreground hover:bg-destructive/90"
data-testid="confirm-pocket-voice-delete"
disabled={busy || !deleteCandidate}
onClick={(event) => {
event.preventDefault();
if (deleteCandidate) {
void deletePocketVoice(deleteCandidate.key);
}
}}
>
Delete voice
</AlertDialogAction>
</AlertDialogFooter>
</AlertDialogContent>
</AlertDialog>
</section>
);
}
+194 -102
View File
@@ -166,6 +166,8 @@ type E2eConfig = {
agentTextToSpeech: boolean;
voicePreferences: string[];
};
/** Native picker boundary result for Pocket voice import tests. */
pocketVoiceImportResult?: "success" | "cancel" | "invalid";
/** Advertised HEAD for the first mock project without adding that branch. */
projectHeadBranch?: string;
/** Builderlab account returned by hosted-community onboarding. Null/omitted = signed out. */
@@ -9911,7 +9913,25 @@ export function maybeInstallE2eTauriMocks() {
deviceId: state === "running" ? "mock-endpoint-id" : null,
deviceName: state === "running" ? "Mock desktop" : null,
});
const handleMockCommand = async (command: string, payload: unknown) => {
let mockImportedVoices: Array<{
key: string;
displayName: string;
backend: string;
backendName: string;
availability: "installed";
fallbackKey: string;
referenceFile: string;
provenance: {
source: string;
contentHash: string;
license: null;
sourceUrl: null;
};
}> = [];
const handleMockCommand = async (
command: string,
payload: unknown,
): Promise<unknown> => {
const activeConfig = getConfig();
const identity = getActiveIdentity(activeConfig);
window.__BUZZ_E2E_COMMANDS__?.push(command);
@@ -9969,107 +9989,110 @@ export function maybeInstallE2eTauriMocks() {
);
case "list_voice_registry":
return [
[
"anna",
"Anna",
"anna.wav",
"p228_023_enhanced.wav",
"0a6de25cf12bf1540beb85979f306a92be81fecc051c547c5395e7e5237a3856",
],
[
"vera",
"Vera",
"vera.wav",
"p229_023_enhanced.wav",
"309cf91a895830f15842b398f69a4962cb1f7e0bfab10e25dd27838e826c204b",
],
[
"fantine",
"Fantine",
"fantine.wav",
"p244_023_enhanced.wav",
"5f07d4e2a3f20a15572aae885156b43ef3fc12ef3812996fd135680d9956448b",
],
[
"charles",
"Charles",
"charles.wav",
"p254_023_enhanced.wav",
"6b681a429198f16e378d53bccb08d06939da7b00144a7696111d4f8f76be7756",
],
[
"paul",
"Paul",
"paul.wav",
"p259_023_enhanced.wav",
"7aba504fe0b3b16478b69ed27ce6007e3cb42b0c1915b5f1c6a6024ae37d679b",
],
[
"eponine",
"Eponine",
"eponine.wav",
"p262_023_enhanced.wav",
"a13c27fb47627b05223691a0ef2974358a18c886e6c2f9d2762ff1d02c20926b",
],
[
"azelma",
"Azelma",
"azelma.wav",
"p303_023_enhanced.wav",
"60e3d26cdf2efdec5df712152c839928f4d5522821e6554ae11fd96c57ab1026",
],
[
"george",
"George",
"george.wav",
"p315_023_enhanced.wav",
"29a41f93bf5236e5b21501091d7774c255d5f3d4e62fa4f9fdf0a92a793c84ae",
],
[
"mary",
"Mary",
"reference_sample.wav",
"p333_023_enhanced.wav",
"a35b0468382218e9f37a9a7494d1e4b74deaf18d7ced22265b4e325bb55c183f",
],
[
"jane",
"Jane",
"jane.wav",
"p339_023_enhanced.wav",
"2f12e7f155eb3118f55425394f1b049e5b1b67bdc9b3932c8ba4521420aeb84a",
],
[
"michael",
"Michael",
"michael.wav",
"p360_023_enhanced.wav",
"b6743e9195e5e3fd34fe9d1633ae93f7ffab787b249e45f6467d7d6f7a6ee6ad",
],
[
"eve",
"Eve",
"eve.wav",
"p361_023_enhanced.wav",
"396e7cbd066b0f3fb6d67fa26e7904076958239d736d4390f15b5fe88feb14cd",
],
].map(
([id, displayName, referenceFile, upstreamFile, contentHash]) => ({
key: `pocket:${id}`,
displayName,
backend: "pocket",
backendName: "Pocket TTS",
availability: "bundled",
fallbackKey: id === "mary" ? null : "pocket:mary",
referenceFile,
provenance: {
source: "bundled",
contentHash,
license: "CC-BY-4.0",
sourceUrl: `https://huggingface.co/kyutai/tts-voices/blob/323332d33f997de8394f24a193e1a76df720e01a/vctk/${upstreamFile}`,
},
}),
);
...[
[
"anna",
"Anna",
"anna.wav",
"p228_023_enhanced.wav",
"0a6de25cf12bf1540beb85979f306a92be81fecc051c547c5395e7e5237a3856",
],
[
"vera",
"Vera",
"vera.wav",
"p229_023_enhanced.wav",
"309cf91a895830f15842b398f69a4962cb1f7e0bfab10e25dd27838e826c204b",
],
[
"fantine",
"Fantine",
"fantine.wav",
"p244_023_enhanced.wav",
"5f07d4e2a3f20a15572aae885156b43ef3fc12ef3812996fd135680d9956448b",
],
[
"charles",
"Charles",
"charles.wav",
"p254_023_enhanced.wav",
"6b681a429198f16e378d53bccb08d06939da7b00144a7696111d4f8f76be7756",
],
[
"paul",
"Paul",
"paul.wav",
"p259_023_enhanced.wav",
"7aba504fe0b3b16478b69ed27ce6007e3cb42b0c1915b5f1c6a6024ae37d679b",
],
[
"eponine",
"Eponine",
"eponine.wav",
"p262_023_enhanced.wav",
"a13c27fb47627b05223691a0ef2974358a18c886e6c2f9d2762ff1d02c20926b",
],
[
"azelma",
"Azelma",
"azelma.wav",
"p303_023_enhanced.wav",
"60e3d26cdf2efdec5df712152c839928f4d5522821e6554ae11fd96c57ab1026",
],
[
"george",
"George",
"george.wav",
"p315_023_enhanced.wav",
"29a41f93bf5236e5b21501091d7774c255d5f3d4e62fa4f9fdf0a92a793c84ae",
],
[
"mary",
"Mary",
"reference_sample.wav",
"p333_023_enhanced.wav",
"a35b0468382218e9f37a9a7494d1e4b74deaf18d7ced22265b4e325bb55c183f",
],
[
"jane",
"Jane",
"jane.wav",
"p339_023_enhanced.wav",
"2f12e7f155eb3118f55425394f1b049e5b1b67bdc9b3932c8ba4521420aeb84a",
],
[
"michael",
"Michael",
"michael.wav",
"p360_023_enhanced.wav",
"b6743e9195e5e3fd34fe9d1633ae93f7ffab787b249e45f6467d7d6f7a6ee6ad",
],
[
"eve",
"Eve",
"eve.wav",
"p361_023_enhanced.wav",
"396e7cbd066b0f3fb6d67fa26e7904076958239d736d4390f15b5fe88feb14cd",
],
].map(
([id, displayName, referenceFile, upstreamFile, contentHash]) => ({
key: `pocket:${id}`,
displayName,
backend: "pocket",
backendName: "Pocket TTS",
availability: "bundled",
fallbackKey: id === "mary" ? null : "pocket:mary",
referenceFile,
provenance: {
source: "bundled",
contentHash,
license: "CC-BY-4.0",
sourceUrl: `https://huggingface.co/kyutai/tts-voices/blob/323332d33f997de8394f24a193e1a76df720e01a/vctk/${upstreamFile}`,
},
}),
),
...mockImportedVoices,
];
case "set_tts_enabled": {
const enabled = (payload as { enabled?: boolean })?.enabled;
if (typeof enabled !== "boolean")
@@ -10114,6 +10137,75 @@ export function maybeInstallE2eTauriMocks() {
}
case "preview_pocket_voice":
return null;
case "import_pocket_voice": {
const importResult =
activeConfig?.mock?.pocketVoiceImportResult ?? "success";
if (importResult === "cancel") return null;
if (importResult === "invalid") {
throw new Error("Voice WAV must contain PCM or 32-bit float audio");
}
const contentHash = "1".repeat(64);
const imported = {
key: `pocket:imported:${contentHash}`,
displayName: "My voice",
backend: "pocket",
backendName: "Pocket TTS",
availability: "installed" as const,
fallbackKey: "pocket:mary",
referenceFile: `${contentHash}.wav`,
provenance: {
source: "local import",
contentHash,
license: null,
sourceUrl: null,
},
};
mockImportedVoices = [imported];
const current = activeConfig?.mock?.ttsSettings ?? {
version: 1,
agentTextToSpeech: true,
voicePreferences: ["pocket:mary"],
};
const settings = {
...current,
voicePreferences: [imported.key],
};
if (activeConfig) {
activeConfig.mock ??= {};
activeConfig.mock.ttsSettings = settings;
}
return {
settings,
registry: await handleMockCommand("list_voice_registry", null),
};
}
case "delete_pocket_voice": {
const voiceKey = (payload as { voiceKey?: string })?.voiceKey;
if (!voiceKey?.startsWith("pocket:imported:"))
throw new Error("Missing imported Pocket voice key");
mockImportedVoices = mockImportedVoices.filter(
(voice) => voice.key !== voiceKey,
);
const current = activeConfig?.mock?.ttsSettings ?? {
version: 1,
agentTextToSpeech: true,
voicePreferences: ["pocket:mary"],
};
const settings = {
...current,
voicePreferences: current.voicePreferences.includes(voiceKey)
? ["pocket:mary"]
: current.voicePreferences,
};
if (activeConfig) {
activeConfig.mock ??= {};
activeConfig.mock.ttsSettings = settings;
}
return {
settings,
registry: await handleMockCommand("list_voice_registry", null),
};
}
case "get_builderlab_auth":
return activeConfig?.mock?.builderlabAuth ?? null;
case "start_builderlab_login": {
+93
View File
@@ -116,4 +116,97 @@ test.describe("Pocket voice settings", () => {
},
});
});
test("imports, selects, and safely deletes a local voice", async ({
page,
}) => {
await installMockBridge(page);
await page.goto("/", { waitUntil: "domcontentloaded" });
await openSettings(page, "voice");
await page.getByTestId("pocket-voice-import").click();
await expect(page.getByTestId("pocket-voice-selector")).toContainText(
"My voice",
);
await expect(page.getByTestId("pocket-voice-delete")).toBeVisible();
await page.getByRole("button", { name: "Preview" }).click();
await page.getByTestId("pocket-voice-delete").click();
await expect(page.getByText("Delete imported voice?")).toBeVisible();
await page.getByTestId("confirm-pocket-voice-delete").click();
await expect(page.getByTestId("pocket-voice-selector")).toContainText(
"Mary",
);
await expect(page.getByTestId("pocket-voice-delete")).toBeHidden();
await page.getByRole("button", { name: "Preview" }).click();
const mutations = await page.evaluate(() =>
(window.__BUZZ_E2E_COMMAND_LOG__ ?? [])
.filter((entry) =>
[
"import_pocket_voice",
"preview_pocket_voice",
"delete_pocket_voice",
].includes(entry.command),
)
.map((entry) => ({ command: entry.command, payload: entry.payload })),
);
expect(mutations).toEqual([
{ command: "import_pocket_voice", payload: {} },
{
command: "preview_pocket_voice",
payload: { voiceKey: `pocket:imported:${"1".repeat(64)}` },
},
{
command: "delete_pocket_voice",
payload: { voiceKey: `pocket:imported:${"1".repeat(64)}` },
},
{
command: "preview_pocket_voice",
payload: { voiceKey: "pocket:mary" },
},
]);
});
test("keeps the selected voice unchanged when the native picker is cancelled", async ({
page,
}) => {
await installMockBridge(page, { pocketVoiceImportResult: "cancel" });
await page.goto("/", { waitUntil: "domcontentloaded" });
await openSettings(page, "voice");
await expect(page.getByTestId("pocket-voice-selector")).toContainText(
"Mary",
);
await page.getByTestId("pocket-voice-import").click();
await expect(page.getByTestId("pocket-voice-selector")).toContainText(
"Mary",
);
await expect(page.getByTestId("pocket-voice-delete")).toBeHidden();
await expect(page.getByTestId("voice-settings-error")).toBeHidden();
const audioCommands = await page.evaluate(() =>
(window.__BUZZ_E2E_COMMAND_LOG__ ?? []).filter((entry) =>
["preview_pocket_voice", "delete_pocket_voice"].includes(entry.command),
),
);
expect(audioCommands).toEqual([]);
});
test("surfaces invalid or unsupported WAV errors without changing selection", async ({
page,
}) => {
await installMockBridge(page, { pocketVoiceImportResult: "invalid" });
await page.goto("/", { waitUntil: "domcontentloaded" });
await openSettings(page, "voice");
await page.getByTestId("pocket-voice-import").click();
await expect(page.getByTestId("voice-settings-error")).toContainText(
"Voice WAV must contain PCM or 32-bit float audio",
);
await expect(page.getByTestId("pocket-voice-selector")).toContainText(
"Mary",
);
await expect(page.getByTestId("pocket-voice-delete")).toBeHidden();
});
});
+2
View File
@@ -164,6 +164,8 @@ type MockBridgeOptions = {
agentTextToSpeech: boolean;
voicePreferences: string[];
};
/** Native picker boundary result for Pocket voice import tests. */
pocketVoiceImportResult?: "success" | "cancel" | "invalid";
/** Advertised HEAD for the first mock project without adding that branch. */
projectHeadBranch?: string;
/** Relay NIP-11 identity used to sign authoritative repository state. */