diff --git a/desktop/playwright.config.ts b/desktop/playwright.config.ts index 459fa7574..ffa51e4b0 100644 --- a/desktop/playwright.config.ts +++ b/desktop/playwright.config.ts @@ -48,6 +48,7 @@ export default defineConfig({ "**/activity-scope-label-screenshots.spec.ts", "**/welcome-agent-modal-screenshots.spec.ts", "**/local-archive-screenshots.spec.ts", + "**/voice-settings.spec.ts", "**/agent-readiness-screenshots.spec.ts", "**/agent-error-state-screenshots.spec.ts", "**/edit-agent.spec.ts", diff --git a/desktop/src-tauri/resources/pocket-voices/NOTICE.md b/desktop/src-tauri/resources/pocket-voices/NOTICE.md new file mode 100644 index 000000000..ecbee2691 --- /dev/null +++ b/desktop/src-tauri/resources/pocket-voices/NOTICE.md @@ -0,0 +1,25 @@ +# Marius Pocket TTS reference voice + +`marius.wav` is an audio-identical copy of Kyutai's +`voice-donations/Selfie.wav`; only the filename changed. + +- Display name: Marius +- Source: https://huggingface.co/kyutai/tts-voices/blob/main/voice-donations/Selfie.wav +- Upstream revision: `323332d33f997de8394f24a193e1a76df720e01a` +- SHA-256: `076968c3122520f3412eb7090e8c1c3f75fe57be1e24a2f96465583d84c71e16` +- Format: WAV PCM signed 16-bit little-endian, mono, 24 kHz, 10 seconds +- License: CC0 1.0 Universal + (https://creativecommons.org/publicdomain/zero/1.0/) +- Repository provenance: + https://huggingface.co/kyutai/tts-voices/blob/main/README.md +- Donation terms: + https://kyutai.org/next/legal/Terms%20of%20Use%20-%20Unmute%20Voice%20Donation%20Project%20v1.pdf + +The adult contributor submitted their own natural voice through Kyutai's +Unmute Voice Donation Project. Its terms expressly contemplate public reuse, +text-to-speech and voice-replica use, and synthetic speech generated in the +contributed voice. + +Neither Kyutai nor the contributor endorses Buzz. Buzz does not identify or +attempt to identify the contributor, and synthetic output must not be presented +as a genuine recording by them. diff --git a/desktop/src-tauri/resources/pocket-voices/marius.wav b/desktop/src-tauri/resources/pocket-voices/marius.wav new file mode 100644 index 000000000..11ab46039 Binary files /dev/null and b/desktop/src-tauri/resources/pocket-voices/marius.wav differ diff --git a/desktop/src-tauri/src/app_state.rs b/desktop/src-tauri/src/app_state.rs index 94d162e62..7f7a52114 100644 --- a/desktop/src-tauri/src/app_state.rs +++ b/desktop/src-tauri/src/app_state.rs @@ -30,9 +30,8 @@ pub struct AppState { /// Workspace-provided relay URL override. Set by `apply_workspace` on app /// init and takes priority over env vars and compile-time defaults. pub relay_url_override: Mutex>, - /// Set during backend setup when managed agents are eligible for launch - /// restore. `apply_workspace` consumes it after installing the workspace - /// relay and identity, so agents never start against the fallback relay. + /// Set during setup when managed agents are eligible for launch restore; + /// consumed after workspace identity/relay install to avoid the fallback. pub managed_agent_restore_pending: AtomicBool, /// Whether desktop may repair managed-agent kind:0 profiles from its local /// records. Disabled by the agent-managed profiles experiment so an agent's @@ -40,19 +39,17 @@ pub struct AppState { pub managed_agent_profile_reconcile_enabled: AtomicBool, /// Shared shutdown signal checked by launch-time agent restoration. pub shutdown_started: AtomicBool, - /// Serializes every managed-runtime transition that changes the protected - /// PID set: spawn/register, adoption, stop, shutdown, and sweep snapshots. + /// Serializes managed-runtime transitions that change the protected PID set: + /// spawn/register, adoption, stop, shutdown, and sweep snapshots. /// Never perform network I/O while holding this lock. pub managed_agent_runtime_transition: Mutex<()>, pub managed_agents_store_lock: Mutex<()>, pub channel_templates_store_lock: Mutex<()>, pub managed_agent_processes: Mutex>, pub huddle_state: Mutex, - /// Tauri app handle — stored after setup so huddle commands can emit - /// `huddle-state-changed` events without needing the handle threaded - /// through every call site. - /// - /// Set once during `setup()` in `lib.rs`; never cleared. + pub tts_settings: Mutex, + pub tts_settings_load_error: Mutex>, + pub tts_settings_transition: tokio::sync::Mutex<()>, pub app_handle: Mutex>, /// Selected audio output device name. `None` = system default. /// Used by `connect_audio_relay` and TTS pipeline when opening sinks. @@ -213,6 +210,9 @@ pub fn build_app_state() -> AppState { managed_agent_processes: Mutex::new(HashMap::new()), session_config_cache: Mutex::new(HashMap::new()), huddle_state: Mutex::new(HuddleState::default()), + tts_settings: Mutex::new(Default::default()), + tts_settings_load_error: Mutex::new(None), + tts_settings_transition: tokio::sync::Mutex::new(()), app_handle: Mutex::new(None), audio_output_device: Mutex::new(None), media_proxy_port: AtomicU16::new(0), diff --git a/desktop/src-tauri/src/huddle/mod.rs b/desktop/src-tauri/src/huddle/mod.rs index a815bf2d0..7eaa5add4 100644 --- a/desktop/src-tauri/src/huddle/mod.rs +++ b/desktop/src-tauri/src/huddle/mod.rs @@ -37,6 +37,7 @@ pub mod state; pub mod stt; pub mod transcription; pub mod tts; +pub mod tts_settings; pub mod wire; // ── Shared utilities ────────────────────────────────────────────────────────── @@ -63,6 +64,7 @@ pub(super) fn drain_until_shutdown( pub use state::{HuddleJoinInfo, HuddlePhase, HuddleState, VoiceInputMode}; pub use transcription::{set_huddle_transcription_enabled, start_stt_pipeline}; +pub use tts_settings::set_tts_enabled; // ── Imports ─────────────────────────────────────────────────────────────────── @@ -817,47 +819,6 @@ pub fn get_model_status(_state: State<'_, AppState>) -> Result) -> Result<(), String> { - let old_pipeline = { - let mut hs = state.huddle()?; - hs.tts_enabled = enabled; - if !enabled { - hs.tts_pipeline.take() // Take out of lock. - } else { - None - } - }; - // Shut down outside the lock — thread join happens here. - if let Some(ref pipeline) = old_pipeline { - pipeline.shutdown(); - } - drop(old_pipeline); - - if enabled { - // Re-start TTS pipeline if models are available and huddle is active. - let phase = { - let hs = state.huddle()?; - hs.phase.clone() - }; - if matches!(phase, HuddlePhase::Connected | HuddlePhase::Active) { - if let Err(e) = maybe_start_tts_pipeline(&state).await { - eprintln!("buzz-desktop: TTS pipeline restart failed: {e}"); - } - } - } - - Ok(()) -} - /// Speak an agent message via TTS. /// /// Maximum text length accepted for TTS synthesis. @@ -895,13 +856,26 @@ pub async fn speak_agent_message(text: String, state: State<'_, AppState>) -> Re } } - let hs = state.huddle()?; - if hs.tts_enabled { - if let Some(ref pipeline) = hs.tts_pipeline { - pipeline.speak(text)?; - } - } - Ok(()) + let sender = { + let hs = state.huddle()?; + hs.tts_enabled + .then(|| { + hs.tts_pipeline + .as_ref() + .map(|pipeline| pipeline.text_sender()) + }) + .flatten() + }; + let Some(sender) = sender else { + return Ok(()); + }; + tokio::task::spawn_blocking(move || { + sender + .send(text) + .map_err(|error| format!("TTS queue closed while waiting to enqueue: {error}")) + }) + .await + .map_err(|error| format!("TTS enqueue task failed: {error}"))? } /// Add an agent to the active huddle. diff --git a/desktop/src-tauri/src/huddle/models.rs b/desktop/src-tauri/src/huddle/models.rs index 11c9ee4c7..de37a0538 100644 --- a/desktop/src-tauri/src/huddle/models.rs +++ b/desktop/src-tauri/src/huddle/models.rs @@ -28,6 +28,9 @@ use super::pocket::{ april_model_info, PocketModelArtifact, APRIL_BUNDLE_ID, APRIL_MODEL_ID, APRIL_MODEL_REVISION, }; +#[path = "models_voice_upgrade.rs"] +mod voice_upgrade; + // ── Integrity verification ──────────────────────────────────────────────────── // // All model artifacts are verified against pinned SHA-256 hashes before @@ -186,6 +189,14 @@ https://datashare.ed.ac.uk/handle/10283/3443 (CC-BY-4.0). Recording enhancement (denoise/dereverb) by ai-coustics: https://ai-coustics.com/ +Bundled reference voice (marius.wav): +\"Marius\", an audio-identical copy of Kyutai's +`voice-donations/Selfie.wav`, distributed under CC0 1.0 Universal: +https://huggingface.co/kyutai/tts-voices/blob/main/voice-donations/Selfie.wav +https://creativecommons.org/publicdomain/zero/1.0/ +The contributor submitted their own voice under Kyutai's Unmute Voice Donation +Project terms. Neither Kyutai nor the contributor endorses Buzz. + Buzz ships all ONNX/model artifacts and the reference voice WAV unmodified, renamed only by placement in the local model directory. @@ -205,6 +216,7 @@ const TTS_EXPECTED_FILES: &[&str] = &[ "tokenizer.model", "LICENSE", "reference_sample.wav", + "marius.wav", TTS_LICENSE_FILE_NAME, ]; @@ -705,6 +717,9 @@ impl ModelManager { /// Start a background Pocket TTS download. No-op if already ready or downloading. pub fn start_tts_download(&self, http_client: reqwest::Client) { + if let Err(error) = voice_upgrade::install_marius_into_v3_model(&self.models_dir) { + eprintln!("buzz-desktop: could not upgrade existing Pocket voices in place: {error}"); + } let manager = self.clone(); self.tts.start_download( &self.models_dir, @@ -822,7 +837,7 @@ impl ModelManager { /// - five ONNX sessions selected by the April INT8 bundle /// - bundle metadata, SentencePiece tokenizer, and learned voice BOS /// - upstream `LICENSE` plus Buzz's `MODEL_LICENSE.txt` attribution sidecar - /// - `reference_sample.wav` as the bundled default voice + /// - `reference_sample.wav` and embedded `marius.wav` reference voices /// /// Files are written to a temp directory first, then moved atomically. async fn download_tts_model(&self, http_client: reqwest::Client) -> Result<(), String> { @@ -907,6 +922,12 @@ impl ModelManager { tokio::fs::write(temp_dir.join(TTS_LICENSE_FILE_NAME), TTS_LICENSE_TEXT) .await .map_err(|e| format!("write TTS model license sidecar: {e}"))?; + tokio::fs::write( + temp_dir.join("marius.wav"), + voice_upgrade::POCKET_MARIUS_WAV, + ) + .await + .map_err(|e| format!("install bundled Marius voice: {e}"))?; self.tts.set_status(ModelStatus::Downloading { progress_percent: 90, diff --git a/desktop/src-tauri/src/huddle/models_voice_upgrade.rs b/desktop/src-tauri/src/huddle/models_voice_upgrade.rs new file mode 100644 index 000000000..28c6d26da --- /dev/null +++ b/desktop/src-tauri/src/huddle/models_voice_upgrade.rs @@ -0,0 +1,73 @@ +use super::*; + +pub(super) const POCKET_MARIUS_WAV: &[u8] = + include_bytes!("../../resources/pocket-voices/marius.wav"); +const PRE_MARIUS_TTS_MODEL_VERSION: &str = "3"; + +/// Add Marius to an otherwise-ready v3 install without re-downloading models. +/// +/// The manifest is written last, so interruption leaves v3 intact and the next +/// launch retries. +pub(super) fn install_marius_into_v3_model(models_dir: &Path) -> Result<(), String> { + let model_dir = models_dir.join(TTS_MODEL_DIR_NAME); + let manifest_path = model_dir.join(MANIFEST_FILENAME); + let version = match std::fs::read_to_string(&manifest_path) { + Ok(version) => version, + Err(_) => return Ok(()), + }; + if version.trim() != PRE_MARIUS_TTS_MODEL_VERSION { + return Ok(()); + } + if !TTS_EXPECTED_FILES + .iter() + .filter(|filename| **filename != "marius.wav") + .all(|filename| model_dir.join(filename).is_file()) + { + return Ok(()); + } + + std::fs::write(model_dir.join("marius.wav"), POCKET_MARIUS_WAV) + .map_err(|error| format!("write Marius voice: {error}"))?; + std::fs::write(model_dir.join(TTS_LICENSE_FILE_NAME), TTS_LICENSE_TEXT) + .map_err(|error| format!("update Pocket voice notice: {error}"))?; + std::fs::write(manifest_path, TTS_MODEL_VERSION) + .map_err(|error| format!("update Pocket model manifest: {error}")) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn v3_install_adds_marius_without_redownloading_models() { + let temp = tempfile::tempdir().expect("tempdir"); + let model_dir = temp.path().join(TTS_MODEL_DIR_NAME); + std::fs::create_dir_all(&model_dir).expect("create model dir"); + for file in TTS_EXPECTED_FILES + .iter() + .filter(|filename| **filename != "marius.wav") + { + std::fs::write(model_dir.join(file), b"existing").expect("write prior file"); + } + std::fs::write( + model_dir.join(MANIFEST_FILENAME), + PRE_MARIUS_TTS_MODEL_VERSION, + ) + .expect("write prior manifest"); + + install_marius_into_v3_model(temp.path()).expect("in-place upgrade"); + + assert_eq!( + std::fs::read(model_dir.join("marius.wav")).expect("Marius installed"), + POCKET_MARIUS_WAV + ); + assert_eq!( + std::fs::read_to_string(model_dir.join(MANIFEST_FILENAME)).expect("updated manifest"), + TTS_MODEL_VERSION + ); + assert!( + ModelSlot::new(TTS_MODEL_DIR_NAME, TTS_EXPECTED_FILES, TTS_MODEL_VERSION) + .is_ready(temp.path()) + ); + } +} diff --git a/desktop/src-tauri/src/huddle/pipeline.rs b/desktop/src-tauri/src/huddle/pipeline.rs index ceccedd8b..478771712 100644 --- a/desktop/src-tauri/src/huddle/pipeline.rs +++ b/desktop/src-tauri/src/huddle/pipeline.rs @@ -216,8 +216,15 @@ pub(crate) async fn maybe_start_tts_pipeline(state: &AppState) -> Result, - /// Voice name (e.g. "reference_sample"). Stored for future voice-switching support. - #[allow(dead_code)] - voice: String, + /// Selected manifest voice. The worker reloads only the lightweight style + /// when this changes; the warmed Pocket engine and audio player stay alive. + voice: Arc>, /// Worker thread handle — taken on drop to join cleanly. thread: Option>, } impl TtsPipeline { - /// Spawn the TTS pipeline thread using the default voice. + /// Spawn the TTS pipeline thread with a manifest-backed voice name. /// - /// `model_dir` must contain the Pocket TTS files declared by `huddle::models` - /// (the five ONNX sessions, the two JSON tables, and `.wav`). - /// - /// `tts_active` is set to `true` while audio is playing and `false` when idle. - /// Pass the same `Arc` to the STT pipeline to gate microphone input. - /// - /// `cancel` is the shared barge-in flag from `HuddleState.tts_cancel`. Pass the - /// same `Arc` to the STT pipeline so both sides reference the same flag for the - /// entire huddle session — no stale references after pipeline restarts. - pub fn new( - model_dir: PathBuf, - tts_active: Arc, - cancel: Arc, - output_device: Option, - ) -> Result { - use super::pocket::DEFAULT_VOICE; - Self::new_with_voice(model_dir, tts_active, cancel, DEFAULT_VOICE, output_device) - } - - /// Spawn the TTS pipeline thread with a specific voice name. Today only the - /// bundled default voice (see `pocket::DEFAULT_VOICE`) is shipped; other - /// names will surface a clear error from `load_voice_style`. + /// `cancel` is shared with STT for barge-in. The same handle survives voice + /// changes so the warmed Pocket engine is retained. pub fn new_with_voice( model_dir: PathBuf, tts_active: Arc, @@ -173,7 +153,8 @@ impl TtsPipeline { let shutdown_worker = Arc::clone(&shutdown); let cancel_worker = Arc::clone(&cancel); let tts_active_worker = Arc::clone(&tts_active); - let voice_name = voice.to_string(); + let voice = Arc::new(Mutex::new(voice.to_string())); + let voice_worker = Arc::clone(&voice); let model_dir_worker = model_dir.clone(); let handle = thread::Builder::new() @@ -181,7 +162,7 @@ impl TtsPipeline { .spawn(move || { tts_worker( model_dir_worker, - voice_name, + voice_worker, text_rx, tts_active_worker, shutdown_worker, @@ -196,7 +177,7 @@ impl TtsPipeline { tts_active, shutdown, cancel, - voice: voice.to_string(), + voice, thread: Some(handle), }) } @@ -212,6 +193,23 @@ impl TtsPipeline { }) } + /// Clone the bounded queue sender so callers can apply backpressure without + /// holding the huddle mutex. Disabling TTS drops the receiver and unblocks + /// any waiting sender while the shared cancellation flag stops playback. + pub(crate) fn text_sender(&self) -> SyncSender { + self.text_tx.clone() + } + + /// Select a bundled Pocket voice for subsequent speech. + /// + /// Current playback and queued text are cancelled immediately so content + /// cannot continue in the old voice. The worker keeps its warmed inference + /// engine and reloads only the reference style before the next utterance. + pub fn select_voice(&self, voice: &str) { + *self.voice.lock().unwrap_or_else(|error| error.into_inner()) = voice.to_string(); + self.cancel.store(true, Ordering::Release); + } + /// Signal the worker thread to stop. pub fn shutdown(&self) { self.shutdown.store(true, Ordering::Release); @@ -239,7 +237,7 @@ impl Drop for TtsPipeline { fn tts_worker( model_dir: PathBuf, - voice_name: String, + selected_voice: Arc>, text_rx: mpsc::Receiver, tts_active: Arc, shutdown: Arc, @@ -262,8 +260,12 @@ fn tts_worker( }; // ── 2. Load voice style ─────────────────────────────────────────────────── + let mut voice_name = selected_voice + .lock() + .unwrap_or_else(|error| error.into_inner()) + .clone(); let voice_path = model_dir.join(format!("{voice_name}.{VOICE_FILE_EXT}")); - let style = match load_voice_style(&voice_path) { + let mut style = match load_voice_style(&voice_path) { Ok(s) => s, Err(e) => { eprintln!( @@ -436,6 +438,46 @@ fn tts_worker( continue; } + // Voice changes cancel the old utterance/queue and are observed here, + // before receiving subsequent text. A bad bundled asset falls back to + // Mary without discarding the already-warmed Pocket engine. + let requested_voice = selected_voice + .lock() + .unwrap_or_else(|error| error.into_inner()) + .clone(); + if requested_voice != voice_name { + let requested_path = model_dir.join(format!("{requested_voice}.{VOICE_FILE_EXT}")); + match load_voice_style(&requested_path) { + Ok(requested_style) => { + style = requested_style; + voice_name = requested_voice; + } + Err(error) => { + use super::pocket::DEFAULT_VOICE; + eprintln!( + "buzz-desktop: Pocket voice {requested_voice} is unavailable ({error}); falling back to Mary" + ); + let fallback_path = model_dir.join(format!("{DEFAULT_VOICE}.{VOICE_FILE_EXT}")); + match load_voice_style(&fallback_path) { + Ok(fallback_style) => { + style = fallback_style; + voice_name = DEFAULT_VOICE.to_string(); + *selected_voice + .lock() + .unwrap_or_else(|lock_error| lock_error.into_inner()) = + DEFAULT_VOICE.to_string(); + } + Err(fallback_error) => { + eprintln!( + "buzz-desktop: Mary voice fallback is unavailable: {fallback_error}" + ); + continue; + } + } + } + } + } + let raw_text = match text_rx.recv_timeout(RECV_TIMEOUT) { Ok(t) => t, Err(mpsc::RecvTimeoutError::Timeout) => { @@ -800,3 +842,6 @@ use super::drain_until_shutdown; #[cfg(test)] #[path = "tts_tests.rs"] mod tests; +#[cfg(test)] +#[path = "tts_voice_selection_tests.rs"] +mod voice_selection_tests; diff --git a/desktop/src-tauri/src/huddle/tts_settings.rs b/desktop/src-tauri/src/huddle/tts_settings.rs new file mode 100644 index 000000000..9a949d839 --- /dev/null +++ b/desktop/src-tauri/src/huddle/tts_settings.rs @@ -0,0 +1,715 @@ +//! Installation-global text-to-speech preferences and the local voice registry. +//! +//! Voice keys are backend-qualified (`pocket:mary`, `siri:aaron`) and +//! preferences are ordered. A client resolves the first compatible entry for +//! its one active playback backend. The same [`VoicePreferences`] value can be +//! embedded in installation-global settings or future agent identity without a +//! schema change. Availability is intentionally client-local. + +use std::path::{Path, PathBuf}; + +use serde::{Deserialize, Serialize}; +use tauri::{AppHandle, Manager, State}; + +use crate::{app_state::AppState, managed_agents::storage::atomic_write_json_restricted}; + +use super::{models, pocket::DEFAULT_VOICE, HuddlePhase}; + +const SETTINGS_FILE: &str = "tts-settings.json"; +const CURRENT_VERSION: u32 = 1; +pub const POCKET_BACKEND_ID: &str = "pocket"; +pub const MARY_VOICE_KEY: &str = "pocket:mary"; +pub const MARIUS_VOICE_KEY: &str = "pocket:marius"; + +const VOICE_AVAILABILITY_BUNDLED: &str = "bundled"; +const VOICE_AVAILABILITY_INSTALLED: &str = "installed"; + +#[derive(Debug, Clone, Serialize, PartialEq, Eq)] +#[serde(rename_all = "camelCase")] +pub struct VoiceRegistryEntry { + /// Stable identity, never derived from or merged by the display name. + /// + /// Built-ins use `backend:slug`. Future imports use + /// `pocket:imported:` so two clips with the same + /// editable label remain distinct. + pub key: String, + pub display_name: String, + pub backend: String, + pub backend_name: String, + /// Client-local state: bundled, installed, downloadable, or unavailable. + pub availability: String, + pub fallback_key: Option, + pub reference_file: Option, + pub provenance: VoiceProvenance, +} + +#[derive(Debug, Clone, Serialize, PartialEq, Eq)] +#[serde(rename_all = "camelCase")] +pub struct VoiceProvenance { + pub source: String, + pub content_hash: Option, + pub license: Option, + pub source_url: Option, +} + +/// Ordered, backend-qualified preferences shared by global and agent settings. +/// +/// Unknown but well-formed keys remain persisted because a different client +/// may have that backend installed. Resolution is always local. +pub type VoicePreferences = Vec; + +/// Cross-backend registry for voices known to this client. +/// +/// V1 contains Pocket entries only. Siri, Kokoro, imported voices, and +/// per-agent assignment can add entries or reuse the preference type without +/// changing the registry/settings boundary. +pub fn voice_registry() -> Vec { + vec![ + VoiceRegistryEntry { + key: MARY_VOICE_KEY.to_string(), + display_name: "Mary".to_string(), + backend: POCKET_BACKEND_ID.to_string(), + backend_name: "Pocket TTS".to_string(), + availability: VOICE_AVAILABILITY_BUNDLED.to_string(), + fallback_key: None, + reference_file: Some("reference_sample.wav".to_string()), + provenance: VoiceProvenance { + source: "bundled".to_string(), + content_hash: Some( + "a35b0468382218e9f37a9a7494d1e4b74deaf18d7ced22265b4e325bb55c183f" + .to_string(), + ), + license: Some("CC-BY-4.0".to_string()), + source_url: Some("https://datashare.ed.ac.uk/handle/10283/3443".to_string()), + }, + }, + VoiceRegistryEntry { + key: MARIUS_VOICE_KEY.to_string(), + display_name: "Marius".to_string(), + backend: POCKET_BACKEND_ID.to_string(), + backend_name: "Pocket TTS".to_string(), + availability: VOICE_AVAILABILITY_BUNDLED.to_string(), + fallback_key: Some(MARY_VOICE_KEY.to_string()), + reference_file: Some("marius.wav".to_string()), + provenance: VoiceProvenance { + source: "bundled".to_string(), + content_hash: Some( + "076968c3122520f3412eb7090e8c1c3f75fe57be1e24a2f96465583d84c71e16" + .to_string(), + ), + license: Some("CC0-1.0".to_string()), + source_url: Some( + "https://huggingface.co/kyutai/tts-voices/blob/323332d33f997de8394f24a193e1a76df720e01a/voice-donations/Selfie.wav" + .to_string(), + ), + }, + }, + ] +} + +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] +#[serde(rename_all = "camelCase")] +pub struct TtsSettings { + pub version: u32, + pub agent_text_to_speech: bool, + pub voice_preferences: VoicePreferences, +} + +impl Default for TtsSettings { + fn default() -> Self { + Self { + version: CURRENT_VERSION, + agent_text_to_speech: true, + voice_preferences: vec![MARY_VOICE_KEY.to_string()], + } + } +} + +pub fn voice_by_key(key: &str) -> Option { + voice_registry().into_iter().find(|voice| voice.key == key) +} + +fn is_qualified_voice_key(key: &str) -> bool { + key.split_once(':') + .is_some_and(|(backend, voice)| !backend.is_empty() && !voice.is_empty()) +} + +fn is_locally_available(availability: &str) -> bool { + matches!( + availability, + VOICE_AVAILABILITY_BUNDLED | VOICE_AVAILABILITY_INSTALLED + ) +} + +pub fn resolve_voice_for_backend( + preferences: &[String], + backend: &str, +) -> Result { + let registry = voice_registry(); + preferences + .iter() + .filter_map(|key| registry.iter().find(|voice| voice.key == *key)) + .find(|voice| voice.backend == backend && is_locally_available(voice.availability.as_str())) + .or_else(|| { + registry.iter().find(|voice| { + voice.backend == backend + && voice.fallback_key.is_none() + && is_locally_available(voice.availability.as_str()) + }) + }) + .cloned() + .ok_or_else(|| format!("No locally available fallback voice for backend {backend}")) +} + +pub fn pocket_voice_name(preferences: &[String]) -> String { + resolve_voice_for_backend(preferences, POCKET_BACKEND_ID) + .ok() + .and_then(|voice| voice.reference_file) + .and_then(|file| file.strip_suffix(".wav").map(str::to_string)) + .unwrap_or_else(|| DEFAULT_VOICE.to_string()) +} + +pub(crate) fn settings_path(app: &AppHandle) -> Result { + app.path() + .app_data_dir() + .map(|dir| dir.join(SETTINGS_FILE)) + .map_err(|error| format!("could not locate Buzz settings storage: {error}")) +} + +pub(crate) fn load_from_path(path: &Path) -> Result { + if !path.exists() { + return Ok(TtsSettings::default()); + } + let bytes = std::fs::read(path) + .map_err(|error| format!("could not read text-to-speech settings: {error}"))?; + let value: serde_json::Value = serde_json::from_slice(&bytes) + .map_err(|error| format!("text-to-speech settings are not valid JSON: {error}"))?; + + // Pre-V1 experiment builds stored an unversioned, incompatible shape. + // Migrate it to deterministic V1 defaults instead of carrying it over. + if value.get("version").is_none() { + return Ok(TtsSettings::default()); + } + + let version = value + .get("version") + .and_then(serde_json::Value::as_u64) + .ok_or("text-to-speech settings version is invalid")?; + if version > u64::from(CURRENT_VERSION) { + return Err(format!( + "text-to-speech settings version {version} is newer than this Buzz build supports" + )); + } + + // Early experiments used one bare Pocket `voiceId`. Preserve the toggle + // and qualify that value into the ordered cross-backend preference schema. + if value.get("voicePreferences").is_none() { + let legacy_voice = value + .get("voiceId") + .or_else(|| value.get("voice_id")) + .and_then(serde_json::Value::as_str) + .unwrap_or("mary"); + let voice_key = if is_qualified_voice_key(legacy_voice) { + legacy_voice.to_string() + } else { + format!("{POCKET_BACKEND_ID}:{legacy_voice}") + }; + return Ok(TtsSettings { + version: CURRENT_VERSION, + agent_text_to_speech: value + .get("agentTextToSpeech") + .and_then(serde_json::Value::as_bool) + .unwrap_or(true), + voice_preferences: vec![voice_key], + }); + } + + let mut settings: TtsSettings = serde_json::from_value(value) + .map_err(|error| format!("text-to-speech settings are invalid: {error}"))?; + settings.version = CURRENT_VERSION; + if settings.voice_preferences.is_empty() + || settings + .voice_preferences + .iter() + .any(|key| !is_qualified_voice_key(key)) + { + settings.voice_preferences = TtsSettings::default().voice_preferences; + } + Ok(settings) +} + +pub(crate) fn save_to_path(path: &Path, settings: &TtsSettings) -> Result<(), String> { + if settings.voice_preferences.is_empty() { + return Err("At least one voice preference is required".to_string()); + } + if let Some(key) = settings + .voice_preferences + .iter() + .find(|key| !is_qualified_voice_key(key)) + { + return Err(format!( + "Voice preference keys must be backend-qualified: {key}" + )); + } + let payload = serde_json::to_vec_pretty(settings) + .map_err(|error| format!("could not encode text-to-speech settings: {error}"))?; + atomic_write_json_restricted(path, &payload) + .map_err(|error| format!("could not save text-to-speech settings: {error}")) +} + +pub fn load_for_app(app: &AppHandle) -> (TtsSettings, Option) { + let result = settings_path(app).and_then(|path| load_from_path(&path)); + match result { + Ok(settings) => (settings, None), + Err(error) => { + eprintln!("buzz-desktop: {error}; preserving the file and using Mary for this session"); + (TtsSettings::default(), Some(error)) + } + } +} + +#[tauri::command] +pub fn get_tts_settings(state: State<'_, AppState>) -> Result { + if let Some(error) = state + .tts_settings_load_error + .lock() + .map_err(|lock_error| format!("text-to-speech settings lock poisoned: {lock_error}"))? + .clone() + { + return Err(format!( + "Voice settings could not be loaded and were left unchanged: {error}" + )); + } + state + .tts_settings + .lock() + .map(|settings| settings.clone()) + .map_err(|error| format!("text-to-speech settings lock poisoned: {error}")) +} + +#[tauri::command] +pub fn list_voice_registry() -> Vec { + voice_registry() +} + +fn ensure_settings_writable(state: &AppState) -> Result<(), String> { + if let Some(error) = state + .tts_settings_load_error + .lock() + .map_err(|lock_error| format!("text-to-speech settings lock poisoned: {lock_error}"))? + .as_ref() + { + return Err(format!( + "Voice settings were not saved because the existing file could not be loaded: {error}" + )); + } + Ok(()) +} + +fn cancel_huddle_speech( + huddle: &mut super::HuddleState, +) -> Option> { + huddle.tts_enabled = false; + huddle + .tts_cancel + .store(true, std::sync::atomic::Ordering::Release); + huddle.tts_pipeline.take() +} + +fn disable_tts_runtime(state: &AppState) -> Result<(), String> { + let old_pipeline = { + let mut huddle = state.huddle()?; + cancel_huddle_speech(&mut huddle) + }; + if let Some(ref pipeline) = old_pipeline { + pipeline.shutdown(); + } + drop(old_pipeline); + state.emit_huddle_state_changed(); + Ok(()) +} + +fn commit_effective_off(state: &AppState) -> Result<(), String> { + state + .tts_settings + .lock() + .map_err(|error| format!("text-to-speech settings lock poisoned: {error}"))? + .agent_text_to_speech = false; + Ok(()) +} + +async fn apply_tts_settings( + settings: TtsSettings, + app: &AppHandle, + state: &AppState, +) -> Result { + if settings.version != CURRENT_VERSION { + return Err(format!( + "Unsupported text-to-speech settings version: {}", + settings.version + )); + } + + // OFF is safety-sensitive: stop current and queued speech before any disk + // I/O, and never resume it merely because persistence fails. + if !settings.agent_text_to_speech { + disable_tts_runtime(state)?; + commit_effective_off(state)?; + } + + ensure_settings_writable(state)?; + save_to_path(&settings_path(app)?, &settings)?; + + *state + .tts_settings + .lock() + .map_err(|error| format!("text-to-speech settings lock poisoned: {error}"))? = + settings.clone(); + + if settings.agent_text_to_speech { + let active = { + let mut huddle = state.huddle()?; + huddle.tts_enabled = true; + huddle + .tts_cancel + .store(false, std::sync::atomic::Ordering::Release); + if let Some(pipeline) = huddle.tts_pipeline.as_ref() { + pipeline.select_voice(&pocket_voice_name(&settings.voice_preferences)); + } + matches!(huddle.phase, HuddlePhase::Connected | HuddlePhase::Active) + }; + if active { + if let Err(error) = super::pipeline::maybe_start_tts_pipeline(state).await { + eprintln!("buzz-desktop: could not hot-start text to speech: {error}"); + } + } + state.emit_huddle_state_changed(); + } + Ok(settings) +} + +/// Compatibility command for the huddle speaker button. It updates the same +/// installation-global preference as Settings; there is no per-huddle override. +#[tauri::command] +pub async fn set_tts_enabled( + enabled: bool, + app: AppHandle, + state: State<'_, AppState>, +) -> Result { + let _transition = state.tts_settings_transition.lock().await; + let mut settings = state + .tts_settings + .lock() + .map_err(|error| format!("text-to-speech settings lock poisoned: {error}"))? + .clone(); + settings.agent_text_to_speech = enabled; + apply_tts_settings(settings, &app, &state).await +} + +fn settings_with_pocket_voice( + mut settings: TtsSettings, + voice_key: &str, +) -> Result { + let voice = voice_by_key(voice_key).ok_or_else(|| format!("Unknown voice: {voice_key}"))?; + if voice.backend != POCKET_BACKEND_ID || !is_locally_available(&voice.availability) { + return Err("The selected Pocket voice is not available on this device".to_string()); + } + let first_pocket_index = settings + .voice_preferences + .iter() + .position(|key| key.starts_with("pocket:")); + settings + .voice_preferences + .retain(|key| !key.starts_with("pocket:")); + let insert_at = first_pocket_index + .unwrap_or(settings.voice_preferences.len()) + .min(settings.voice_preferences.len()); + settings + .voice_preferences + .insert(insert_at, voice_key.to_string()); + Ok(settings) +} + +#[tauri::command] +pub async fn set_pocket_voice( + voice_key: String, + app: AppHandle, + state: State<'_, AppState>, +) -> Result { + let _transition = state.tts_settings_transition.lock().await; + let settings = state + .tts_settings + .lock() + .map_err(|error| format!("text-to-speech settings lock poisoned: {error}"))? + .clone(); + let settings = settings_with_pocket_voice(settings, &voice_key)?; + apply_tts_settings(settings, &app, &state).await +} + +#[tauri::command] +pub async fn preview_pocket_voice( + voice_key: String, + state: State<'_, AppState>, +) -> Result<(), String> { + let voice = voice_by_key(&voice_key).ok_or_else(|| format!("Unknown voice: {voice_key}"))?; + if voice.backend != POCKET_BACKEND_ID { + return Err("Only Pocket voices can be previewed in this build".to_string()); + } + if !models::is_tts_ready() { + return Err("Voice files are still downloading. Try preview again shortly.".to_string()); + } + let model_dir = models::tts_model_dir().ok_or("Pocket voice files are unavailable")?; + let output_device = state + .audio_output_device + .lock() + .unwrap_or_else(|error| error.into_inner()) + .clone(); + let voice_name = voice + .reference_file + .and_then(|file| file.strip_suffix(".wav").map(str::to_string)) + .ok_or_else(|| format!("Voice {voice_key} has no local Pocket reference file"))?; + tokio::task::spawn_blocking(move || { + let active = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false)); + let cancel = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false)); + let pipeline = super::tts::TtsPipeline::new_with_voice( + model_dir, + active.clone(), + cancel, + &voice_name, + output_device, + )?; + pipeline.speak("Hello! This is how I’ll read agent responses.".to_string())?; + let started = std::time::Instant::now(); + let mut heard_audio = false; + while started.elapsed() < std::time::Duration::from_secs(30) { + let is_active = active.load(std::sync::atomic::Ordering::Acquire); + heard_audio |= is_active; + if heard_audio && !is_active { + return Ok(()); + } + std::thread::sleep(std::time::Duration::from_millis(25)); + } + Err("Voice preview timed out. Check your audio output and try again.".to_string()) + }) + .await + .map_err(|error| format!("Voice preview task failed: {error}"))? +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn defaults_are_backwards_compatible_and_use_mary() { + assert_eq!( + TtsSettings::default(), + TtsSettings { + version: 1, + agent_text_to_speech: true, + voice_preferences: vec!["pocket:mary".to_string()], + } + ); + } + + #[test] + fn v1_registry_has_exactly_the_two_reviewed_bundled_voices() { + assert_eq!( + voice_registry() + .iter() + .map(|voice| { + ( + voice.key.as_str(), + voice.display_name.as_str(), + voice.reference_file.as_deref(), + ) + }) + .collect::>(), + vec![ + ("pocket:mary", "Mary", Some("reference_sample.wav")), + ("pocket:marius", "Marius", Some("marius.wav")), + ] + ); + } + + #[test] + fn local_backend_resolution_uses_first_compatible_preference() { + let preferences = vec![ + "siri:aaron".to_string(), + MARIUS_VOICE_KEY.to_string(), + MARY_VOICE_KEY.to_string(), + "kokoro:af_heart".to_string(), + ]; + assert_eq!( + resolve_voice_for_backend(&preferences, POCKET_BACKEND_ID) + .expect("Pocket fallback") + .key, + MARIUS_VOICE_KEY + ); + } + + #[test] + fn unsupported_or_missing_preferences_fall_back_to_backend_default() { + let preferences = vec![ + "siri:aaron".to_string(), + "pocket:imported:deadbeef".to_string(), + ]; + assert_eq!( + resolve_voice_for_backend(&preferences, POCKET_BACKEND_ID) + .expect("Pocket fallback") + .key, + MARY_VOICE_KEY + ); + } + + #[test] + fn identity_is_qualified_key_not_display_label() { + assert!(is_qualified_voice_key("pocket:imported:audio-content-hash")); + assert_ne!(MARY_VOICE_KEY, MARIUS_VOICE_KEY); + let mut registry = voice_registry(); + registry[0].display_name = "Jim".to_string(); + registry[1].display_name = "Jim".to_string(); + assert_eq!(registry[0].display_name, registry[1].display_name); + assert_ne!(registry[0].key, registry[1].key); + assert_eq!( + registry + .iter() + .map(|voice| voice.key.as_str()) + .collect::>() + .len(), + registry.len() + ); + } + + #[test] + fn bundled_marius_asset_has_reviewed_size_and_wav_container() { + let bytes = include_bytes!("../../resources/pocket-voices/marius.wav"); + assert_eq!(bytes.len(), 480_044); + assert_eq!(&bytes[0..4], b"RIFF"); + assert_eq!(&bytes[8..12], b"WAVE"); + assert_eq!( + hex::encode(::digest(bytes)), + "076968c3122520f3412eb7090e8c1c3f75fe57be1e24a2f96465583d84c71e16" + ); + } + + #[test] + fn migrates_unversioned_experiment_settings_to_v1_defaults() { + let dir = tempfile::tempdir().expect("temp dir"); + let path = dir.path().join(SETTINGS_FILE); + std::fs::write(&path, r#"{"voice":"legacy-experiment"}"#).expect("fixture write"); + assert_eq!( + load_from_path(&path).expect("migration"), + TtsSettings::default() + ); + } + + #[test] + fn migrates_bare_pocket_voice_id_to_qualified_preferences() { + let dir = tempfile::tempdir().expect("temp dir"); + let path = dir.path().join(SETTINGS_FILE); + std::fs::write( + &path, + r#"{"version":1,"agentTextToSpeech":false,"voiceId":"marius"}"#, + ) + .expect("fixture write"); + assert_eq!( + load_from_path(&path).expect("migration"), + TtsSettings { + version: 1, + agent_text_to_speech: false, + voice_preferences: vec![MARIUS_VOICE_KEY.to_string()], + } + ); + } + + #[test] + fn unknown_qualified_preferences_are_preserved_for_other_clients() { + let dir = tempfile::tempdir().expect("temp dir"); + let path = dir.path().join(SETTINGS_FILE); + std::fs::write( + &path, + r#"{"version":1,"agentTextToSpeech":false,"voicePreferences":["siri:aaron","pocket:imported:abc123"]}"#, + ) + .expect("fixture write"); + let settings = load_from_path(&path).expect("load"); + assert!(!settings.agent_text_to_speech); + assert_eq!( + settings.voice_preferences, + vec!["siri:aaron", "pocket:imported:abc123"] + ); + } + + #[test] + fn rejects_future_schema_versions_clearly() { + let dir = tempfile::tempdir().expect("temp dir"); + let path = dir.path().join(SETTINGS_FILE); + std::fs::write( + &path, + r#"{"version":99,"agentTextToSpeech":true,"voicePreferences":["pocket:mary"]}"#, + ) + .expect("fixture write"); + assert!(load_from_path(&path) + .expect_err("future version should fail") + .contains("newer than this Buzz build supports")); + } + + #[test] + fn disabling_cancels_runtime_before_persistence_can_fail() { + let mut huddle = super::super::HuddleState { + tts_enabled: true, + ..super::super::HuddleState::default() + }; + assert!(!huddle.tts_cancel.load(std::sync::atomic::Ordering::Acquire)); + assert!(cancel_huddle_speech(&mut huddle).is_none()); + assert!(!huddle.tts_enabled); + assert!(huddle.tts_cancel.load(std::sync::atomic::Ordering::Acquire)); + } + + #[test] + fn pocket_voice_update_preserves_the_latest_toggle_and_other_backends() { + let current = TtsSettings { + agent_text_to_speech: false, + voice_preferences: vec!["siri:aaron".to_string(), MARY_VOICE_KEY.to_string()], + ..TtsSettings::default() + }; + let updated = + settings_with_pocket_voice(current, MARIUS_VOICE_KEY).expect("available voice"); + assert!(!updated.agent_text_to_speech); + assert_eq!( + updated.voice_preferences, + vec!["siri:aaron", MARIUS_VOICE_KEY] + ); + } + + #[test] + fn failed_off_persistence_cannot_be_undone_by_a_later_voice_update() { + let state = crate::app_state::build_app_state(); + commit_effective_off(&state).expect("commit effective OFF state"); + + // This models the next command after the OFF save fails: it must merge + // from effective memory state, not the stale last-persisted ON value. + let current = state.tts_settings.lock().expect("settings").clone(); + let voice_update = + settings_with_pocket_voice(current, MARIUS_VOICE_KEY).expect("available voice"); + assert!(!voice_update.agent_text_to_speech); + } + + #[test] + fn failed_disabled_voice_save_does_not_change_the_remembered_voice() { + let state = crate::app_state::build_app_state(); + state + .tts_settings + .lock() + .expect("settings") + .agent_text_to_speech = false; + let current = state.tts_settings.lock().expect("settings").clone(); + let unsaved = + settings_with_pocket_voice(current, MARIUS_VOICE_KEY).expect("available voice"); + + // This is the only pre-persistence mutation for an OFF candidate. + commit_effective_off(&state).expect("commit effective OFF state"); + let remembered = state.tts_settings.lock().expect("settings").clone(); + assert_eq!(remembered.voice_preferences, vec![MARY_VOICE_KEY]); + assert_eq!(unsaved.voice_preferences, vec![MARIUS_VOICE_KEY]); + } +} diff --git a/desktop/src-tauri/src/huddle/tts_voice_selection_tests.rs b/desktop/src-tauri/src/huddle/tts_voice_selection_tests.rs new file mode 100644 index 000000000..a2a8410bb --- /dev/null +++ b/desktop/src-tauri/src/huddle/tts_voice_selection_tests.rs @@ -0,0 +1,30 @@ +use super::*; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::Arc; + +#[test] +fn selecting_a_voice_immediately_raises_cancel_and_retains_the_engine_handle() { + let model_dir = tempfile::tempdir().expect("temp model dir"); + let active = Arc::new(AtomicBool::new(false)); + let cancel = Arc::new(AtomicBool::new(false)); + let pipeline = TtsPipeline::new_with_voice( + model_dir.path().to_path_buf(), + active, + Arc::clone(&cancel), + "reference_sample", + None, + ) + .expect("pipeline handle"); + + pipeline.select_voice("marius"); + + assert!(cancel.load(Ordering::Acquire)); + assert_eq!( + pipeline + .voice + .lock() + .unwrap_or_else(|error| error.into_inner()) + .as_str(), + "marius" + ); +} diff --git a/desktop/src-tauri/src/lib.rs b/desktop/src-tauri/src/lib.rs index 35f4eae86..303463a2f 100644 --- a/desktop/src-tauri/src/lib.rs +++ b/desktop/src-tauri/src/lib.rs @@ -452,6 +452,18 @@ pub fn run() { *guard = Some(app_handle.clone()); } + let (tts_settings, tts_settings_load_error) = + huddle::tts_settings::load_for_app(&app_handle); + if let Ok(mut guard) = state.tts_settings.lock() { + *guard = tts_settings.clone(); + } + if let Ok(mut guard) = state.tts_settings_load_error.lock() { + *guard = tts_settings_load_error; + } + if let Ok(mut huddle) = state.huddle_state.lock() { + huddle.tts_enabled = tts_settings.agent_text_to_speech; + } + // Bring up the runtime-owned shared-compute coordinator before // saved agents are restored. Its lifetime is tied to the app, not // a UI mount; it publishes discovery and reconciles membership for @@ -858,6 +870,10 @@ pub fn run() { download_voice_models, get_model_status, set_tts_enabled, + huddle::tts_settings::get_tts_settings, + huddle::tts_settings::list_voice_registry, + huddle::tts_settings::set_pocket_voice, + huddle::tts_settings::preview_pocket_voice, speak_agent_message, add_agent_to_huddle, check_pipeline_hotstart, diff --git a/desktop/src/features/huddle/lib/ttsLiveMessages.test.mjs b/desktop/src/features/huddle/lib/ttsLiveMessages.test.mjs new file mode 100644 index 000000000..088a6d6eb --- /dev/null +++ b/desktop/src/features/huddle/lib/ttsLiveMessages.test.mjs @@ -0,0 +1,154 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { + createInitialMembershipGate, + createLatestStateGate, + createOrderedSpeaker, + speakableAgentText, +} from "./ttsLiveMessages.ts"; + +const agents = new Set(["agent"]); +const base = { + id: "1", + kind: 9, + pubkey: "agent", + content: "Hello there", + tags: [], +}; + +test("speaks only new agent-authored text message events", () => { + assert.equal(speakableAgentText(base, agents, "human"), "Hello there"); + assert.equal( + speakableAgentText({ ...base, kind: 7 }, agents, "human"), + null, + "reactions and other event kinds are excluded", + ); + assert.equal( + speakableAgentText({ ...base, kind: 10 }, agents, "human"), + null, + "edits and status events are excluded", + ); + assert.equal( + speakableAgentText({ ...base, pubkey: "human" }, agents, "human"), + null, + "human-authored messages are excluded", + ); + assert.equal( + speakableAgentText({ ...base, content: " " }, agents, "human"), + null, + "empty and non-text content are excluded", + ); + assert.equal( + speakableAgentText( + { ...base, content: "[System] tool started" }, + agents, + "human", + ), + null, + "legacy system rows are excluded", + ); +}); + +test("strips attachment markup and skips attachment-only events", () => { + const url = "https://cdn.example/voice.png"; + const tags = [["imeta", `url ${url}`, "m image/png"]]; + assert.equal( + speakableAgentText( + { ...base, content: `![image](${url})`, tags }, + agents, + "human", + ), + null, + ); + assert.equal( + speakableAgentText( + { ...base, content: `Here is the diagram.\n\n![image](${url})`, tags }, + agents, + "human", + ), + "Here is the diagram.", + ); + assert.equal( + speakableAgentText( + { ...base, content: `||\n![image](${url})\n||`, tags }, + agents, + "human", + ), + null, + ); +}); + +test("queues agent messages in live thread arrival order", async () => { + const spoken = []; + let releaseFirst; + const firstBlocked = new Promise((resolve) => { + releaseFirst = resolve; + }); + const speaker = createOrderedSpeaker(async (text) => { + if (text === "first") await firstBlocked; + spoken.push(text); + }, assert.fail); + + speaker.enqueue("first"); + speaker.enqueue("second"); + await Promise.resolve(); + assert.deepEqual(spoken, []); + releaseFirst(); + await new Promise((resolve) => setTimeout(resolve, 0)); + assert.deepEqual(spoken, ["first", "second"]); +}); + +test("disabling cancels queued speech and rejects new messages until enabled", async () => { + const invoked = []; + let releaseFirst; + const firstBlocked = new Promise((resolve) => { + releaseFirst = resolve; + }); + const speaker = createOrderedSpeaker(async (text) => { + invoked.push(text); + if (text === "first") await firstBlocked; + }, assert.fail); + + speaker.enqueue("first"); + speaker.enqueue("queued-before-off"); + await Promise.resolve(); + speaker.setEnabled(false); + speaker.enqueue("while-off"); + releaseFirst(); + await new Promise((resolve) => setTimeout(resolve, 0)); + speaker.setEnabled(true); + speaker.enqueue("after-on"); + await new Promise((resolve) => setTimeout(resolve, 0)); + assert.deepEqual(invoked, ["first", "after-on"]); +}); + +test("a live TTS state event supersedes a delayed bootstrap result", () => { + const applied = []; + const gate = createLatestStateGate((enabled) => applied.push(enabled)); + const applyBootstrap = gate.beginSnapshot(); + + gate.applyEvent(false); + applyBootstrap(true); + + assert.deepEqual(applied, [false]); +}); + +test("buffers initial live events until membership resolves in order", () => { + const delivered = []; + const gate = createInitialMembershipGate((event) => delivered.push(event)); + gate.push("first"); + gate.push("second"); + assert.deepEqual(delivered, []); + gate.succeed(); + gate.push("third"); + assert.deepEqual(delivered, ["first", "second", "third"]); +}); + +test("drops the initial buffer fail-closed when membership lookup fails", () => { + const delivered = []; + const gate = createInitialMembershipGate((event) => delivered.push(event)); + gate.push("unverified"); + gate.fail(); + assert.deepEqual(delivered, []); +}); diff --git a/desktop/src/features/huddle/lib/ttsLiveMessages.ts b/desktop/src/features/huddle/lib/ttsLiveMessages.ts new file mode 100644 index 000000000..1e1aeb8b3 --- /dev/null +++ b/desktop/src/features/huddle/lib/ttsLiveMessages.ts @@ -0,0 +1,124 @@ +export type LiveTtsEvent = { + id: string; + kind: number; + pubkey: string; + content: string; + tags: string[][]; +}; + +function textWithoutAttachments(event: LiveTtsEvent): string { + const urls = new Set( + event.tags + .filter((tag) => tag[0] === "imeta") + .flatMap((tag) => + tag + .slice(1) + .filter((field) => field.startsWith("url ")) + .map((field) => field.slice(4)), + ), + ); + if (urls.size === 0) return event.content; + const withoutMedia = event.content + .split("\n") + .filter( + (line) => !Array.from(urls).some((url) => line.includes(`](${url})`)), + ) + .join("\n"); + return withoutMedia.replace( + /(^|\n)\s*\|\|\s*\n(?:\s*\n)*\s*\|\|\s*(?=\n|$)/gu, + "$1", + ); +} + +export function speakableAgentText( + event: LiveTtsEvent, + agentPubkeys: ReadonlySet, + selfPubkey: string | null, +): string | null { + if (event.kind !== 9) return null; + if (!agentPubkeys.has(event.pubkey)) return null; + if (event.pubkey === selfPubkey) return null; + const content = textWithoutAttachments(event).trim(); + if (content.length <= 1) return null; + if (content.startsWith("[System]")) return null; + return content; +} + +/** + * Serialize native speak calls so live messages enter the bounded Pocket queue + * in thread arrival order even when the bridge resolves calls asynchronously. + */ +export function createOrderedSpeaker( + speak: (text: string) => Promise, + onError: (error: unknown) => void, +): { + enqueue: (text: string) => void; + setEnabled: (enabled: boolean) => void; +} { + let tail = Promise.resolve(); + let enabled = true; + let generation = 0; + return { + enqueue(text) { + if (!enabled) return; + const queuedGeneration = generation; + tail = tail + .then(() => { + if (!enabled || generation !== queuedGeneration) return; + return speak(text); + }) + .catch(onError); + }, + setEnabled(nextEnabled) { + if (!nextEnabled) generation += 1; + enabled = nextEnabled; + }, + }; +} + +/** Ensure a delayed bootstrap snapshot cannot overwrite a newer live event. */ +export function createLatestStateGate(apply: (value: T) => void): { + applyEvent: (value: T) => void; + beginSnapshot: () => (value: T) => void; +} { + let revision = 0; + return { + applyEvent(value) { + revision += 1; + apply(value); + }, + beginSnapshot() { + const snapshotRevision = revision; + return (value) => { + if (revision === snapshotRevision) apply(value); + }; + }, + }; +} + +/** Hold live events until the first authoritative agent-membership lookup. */ +export function createInitialMembershipGate(deliver: (event: T) => void): { + push: (event: T) => void; + succeed: () => void; + fail: () => void; +} { + let settled = false; + let pending: T[] = []; + return { + push(event) { + if (settled) deliver(event); + else pending.push(event); + }, + succeed() { + if (settled) return; + settled = true; + const buffered = pending; + pending = []; + for (const event of buffered) deliver(event); + }, + fail() { + settled = true; + pending = []; + }, + }; +} diff --git a/desktop/src/features/huddle/lib/useTtsSubscription.ts b/desktop/src/features/huddle/lib/useTtsSubscription.ts index d534fc016..251d7e99b 100644 --- a/desktop/src/features/huddle/lib/useTtsSubscription.ts +++ b/desktop/src/features/huddle/lib/useTtsSubscription.ts @@ -1,7 +1,15 @@ import { invoke } from "@tauri-apps/api/core"; +import { listen } from "@tauri-apps/api/event"; import * as React from "react"; +import { buildHuddleTtsLiveFilter } from "@/shared/api/relayChannelFilters"; import { relayClient } from "@/shared/api/relayClient"; +import { + createInitialMembershipGate, + createLatestStateGate, + createOrderedSpeaker, + speakableAgentText, +} from "./ttsLiveMessages"; const AGENT_PUBKEY_REFRESH_INTERVAL_MS = 30_000; @@ -20,6 +28,7 @@ export function useTtsSubscription( let disposed = false; let cleanup: (() => void) | null = null; + let unlistenHuddleState: (() => void) | null = null; // ── Agent identity (authoritative, fail-closed) ─────────────────────── // @@ -33,42 +42,98 @@ export function useTtsSubscription( let agentsLoaded = false; const agentPubkeys = new Set(); - async function loadAgentPubkeys() { + const speakInOrder = createOrderedSpeaker( + async (text) => { + if (!disposed) { + await invoke("speak_agent_message", { text }); + } + }, + (err) => { + console.warn("[huddle] TTS speak failed:", err); + }, + ); + + const deliver = (event: Parameters[0]) => { + if (!agentsLoaded || disposed) return; + const text = speakableAgentText( + event, + agentPubkeys, + selfPubkeyRef.current, + ); + if (text) speakInOrder.enqueue(text); + }; + const initialMembershipGate = createInitialMembershipGate(deliver); + + async function loadAgentPubkeys(initial = false) { try { const pubkeys = await invoke("get_huddle_agent_pubkeys"); + if (disposed) return; agentPubkeys.clear(); for (const pk of pubkeys) agentPubkeys.add(pk); agentsLoaded = true; + if (initial) { + initialMembershipGate.succeed(); + } } catch (e) { // Fail-closed on ALL failures, including refresh after prior success. // Clear the set and mark as not loaded — TTS goes mute until the // next successful refresh. Stale membership must never authorize speech. agentPubkeys.clear(); agentsLoaded = false; + if (initial) { + initialMembershipGate.fail(); + } console.error("[huddle] Failed to load agent pubkeys:", e); } } // Initial load + periodic refresh (catches mid-huddle agent additions). - void loadAgentPubkeys(); + void loadAgentPubkeys(true); const agentRefreshId = window.setInterval(() => { void loadAgentPubkeys(); }, AGENT_PUBKEY_REFRESH_INTERVAL_MS); + // Install the state listener before requesting a snapshot. If a newer + // event arrives while IPC is pending, it supersedes the stale snapshot. + const ttsStateGate = createLatestStateGate<{ tts_enabled: boolean }>( + (state) => { + if (!disposed) speakInOrder.setEnabled(state.tts_enabled); + }, + ); + void listen<{ tts_enabled: boolean }>("huddle-state-changed", (event) => { + if (!disposed) ttsStateGate.applyEvent(event.payload); + }) + .then((unlisten) => { + if (disposed) { + unlisten(); + return; + } + unlistenHuddleState = unlisten; + const applyBootstrap = ttsStateGate.beginSnapshot(); + void invoke<{ tts_enabled: boolean }>("get_huddle_state") + .then((state) => { + if (!disposed) applyBootstrap(state); + }) + .catch((err) => { + if (!disposed) applyBootstrap({ tts_enabled: false }); + console.warn("[huddle] Failed to load TTS state:", err); + }); + }) + .catch((err) => { + speakInOrder.setEnabled(false); + console.warn("[huddle] Failed to listen for TTS state:", err); + }); + // ── Live-only subscription ─────────────────────────────────────────── - // subscribeToChannelLive uses `since: now` — the relay never sends - // historical backlog. Every event delivered is a live message. + // A kind:9, limit:0 subscription receives future fan-out while the relay + // returns no stored rows, including pre-join rows from the current second. // Event-ID dedup handles reconnect replay (same event arriving twice). const seenEventIds = new Set(); const seenOrder: string[] = []; const MAX_SEEN_EVENTS = 5000; - relayClient - .subscribeToChannelLive(ephemeralChannelId, (event) => { + .subscribeLive(buildHuddleTtsLiveFilter(ephemeralChannelId), (event) => { if (disposed) return; - // Defense-in-depth: subscription already filters to kind:9 only. - if (event.kind !== 9) return; - // Dedup by event ID (covers reconnect replay). if (seenEventIds.has(event.id)) return; seenEventIds.add(event.id); @@ -78,20 +143,9 @@ export function useTtsSubscription( if (oldest !== undefined) seenEventIds.delete(oldest); } - // Fail-closed: don't speak until agent list is loaded. - if (!agentsLoaded) return; - // Only speak agent messages — skip human STT transcripts. - if (!agentPubkeys.has(event.pubkey)) return; - if (event.pubkey === selfPubkeyRef.current) return; - if (event.content.trim().length <= 1) return; - // Legacy: skip [System]-prefixed messages from before kind:48106. - if (event.content.startsWith("[System]")) return; - invoke("speak_agent_message", { text: event.content }).catch((err) => { - console.warn( - "[huddle] TTS speak failed (backpressure or pipeline unavailable):", - err, - ); - }); + // Preserve arrival order while the initial authoritative membership + // lookup is pending. A failed lookup clears this buffer fail-closed. + initialMembershipGate.push(event); }) .then((dispose) => { if (disposed) { @@ -106,7 +160,9 @@ export function useTtsSubscription( return () => { disposed = true; + speakInOrder.setEnabled(false); cleanup?.(); + unlistenHuddleState?.(); window.clearInterval(agentRefreshId); }; }, [ephemeralChannelId, selfPubkeyRef]); diff --git a/desktop/src/features/settings/ui/SettingsPanels.tsx b/desktop/src/features/settings/ui/SettingsPanels.tsx index e74d1f383..5c997efbc 100644 --- a/desktop/src/features/settings/ui/SettingsPanels.tsx +++ b/desktop/src/features/settings/ui/SettingsPanels.tsx @@ -21,6 +21,7 @@ import { SunMoon, Ticket, UserRound, + Volume2, type LucideIcon, } from "lucide-react"; import type { @@ -83,10 +84,12 @@ import { SettingsOptionGroup, SettingsOptionRow } from "./SettingsOptionGroup"; import { ProfileSettingsCard } from "./ProfileSettingsCard"; import { UpdateChecker } from "../UpdateChecker"; import { SettingsSectionHeader } from "./SettingsSectionHeader"; +import { VoiceSettingsCard } from "./VoiceSettingsCard"; export type SettingsSection = | "profile" | "notifications" + | "voice" | "experimental" | "agents" | "channel-templates" @@ -106,6 +109,7 @@ export const DEFAULT_SETTINGS_SECTION: SettingsSection = "profile"; const SETTINGS_SECTION_VALUES: readonly SettingsSection[] = [ "profile", "notifications", + "voice", "experimental", "agents", "channel-templates", @@ -167,6 +171,11 @@ export const settingsSections: SettingsSectionDescriptor[] = [ label: "Notifications", icon: BellRing, }, + { + value: "voice", + label: "Voice", + icon: Volume2, + }, { value: "experimental", label: "Experiments", @@ -807,6 +816,8 @@ export function renderSettingsSection( onSetSoundForSlot={props.onSetSoundForSlot} /> ); + case "voice": + return ; case "experimental": return ; case "agents": diff --git a/desktop/src/features/settings/ui/SettingsView.tsx b/desktop/src/features/settings/ui/SettingsView.tsx index 20389b420..861388057 100644 --- a/desktop/src/features/settings/ui/SettingsView.tsx +++ b/desktop/src/features/settings/ui/SettingsView.tsx @@ -58,6 +58,7 @@ const settingsNavGroups: Array<{ "profile", "appearance", "notifications", + "voice", "shortcuts", "custom-emoji", "local-archive", diff --git a/desktop/src/features/settings/ui/VoiceSettingsCard.tsx b/desktop/src/features/settings/ui/VoiceSettingsCard.tsx new file mode 100644 index 000000000..3542d6f81 --- /dev/null +++ b/desktop/src/features/settings/ui/VoiceSettingsCard.tsx @@ -0,0 +1,252 @@ +import * as React from "react"; +import { ChevronDown, Play, Volume2 } from "lucide-react"; + +import { invokeTauri } from "@/shared/api/tauri"; +import { cn } from "@/shared/lib/cn"; +import { Button } from "@/shared/ui/button"; +import { + DropdownMenu, + DropdownMenuContent, + DropdownMenuRadioGroup, + DropdownMenuRadioItem, + DropdownMenuTrigger, +} from "@/shared/ui/dropdown-menu"; +import { Switch } from "@/shared/ui/switch"; +import { SettingsOptionGroup, SettingsOptionRow } from "./SettingsOptionGroup"; +import { SettingsSectionHeader } from "./SettingsSectionHeader"; +import { + selectedVoiceForBackend, + type VoiceRegistryEntry, + voiceOptionLabel, + voicesForBackend, +} from "./voiceSettingsLogic"; + +export type TtsSettings = { + version: number; + agentTextToSpeech: boolean; + voicePreferences: string[]; +}; + +export function VoiceSettingsCard() { + const [settings, setSettings] = React.useState(null); + const [registry, setRegistry] = React.useState([]); + const [busy, setBusy] = React.useState(false); + const [previewing, setPreviewing] = React.useState(false); + const [error, setError] = React.useState(null); + + React.useEffect(() => { + let disposed = false; + Promise.all([ + invokeTauri("get_tts_settings"), + invokeTauri("list_voice_registry"), + ]) + .then(([nextSettings, nextRegistry]) => { + if (!disposed) { + setSettings(nextSettings); + setRegistry(nextRegistry); + } + }) + .catch((loadError) => { + if (!disposed) { + setError( + loadError instanceof Error + ? loadError.message + : "Voice settings could not be loaded.", + ); + } + }); + return () => { + disposed = true; + }; + }, []); + + const saveEnabled = React.useCallback(async (enabled: boolean) => { + setBusy(true); + setError(null); + try { + const saved = await invokeTauri("set_tts_enabled", { + enabled, + }); + setSettings(saved); + } catch (saveError) { + try { + const state = await invokeTauri<{ tts_enabled: boolean }>( + "get_huddle_state", + ); + setSettings((current) => + current + ? { ...current, agentTextToSpeech: state.tts_enabled } + : current, + ); + } catch { + // Keep the last confirmed state when native reconciliation is + // unavailable; the visible save error makes the failure explicit. + } + setError( + saveError instanceof Error + ? saveError.message + : "Voice settings could not be saved.", + ); + } finally { + setBusy(false); + } + }, []); + + const savePocketVoice = React.useCallback(async (voiceKey: string) => { + setBusy(true); + setError(null); + try { + const saved = await invokeTauri("set_pocket_voice", { + voiceKey, + }); + setSettings(saved); + } catch (saveError) { + setError( + saveError instanceof Error + ? saveError.message + : "Voice settings could not be saved.", + ); + } finally { + setBusy(false); + } + }, []); + + const voices = voicesForBackend(registry, "pocket"); + const selectedVoice = selectedVoiceForBackend( + settings?.voicePreferences ?? [], + voices, + ); + const enabled = settings?.agentTextToSpeech ?? true; + const controlsDisabled = !settings || busy || !enabled; + + return ( +
+ + +
+ + +
+ +

+ Read new agent messages aloud in the order they arrive. +

+
+ { + if (settings) void saveEnabled(checked); + }} + /> +
+
+ +
+ + +
+

Voice

+

+ Voice files stay private on this device. +

+
+ +
+ + + + + + { + if (settings) void savePocketVoice(voiceKey); + }} + value={selectedVoice?.key} + > + {voices.map((voice) => ( + + {voiceOptionLabel(voice, voices)} + + ))} + + + + +
+
+
+
+ + {error && ( +

+ {error} +

+ )} +
+
+ ); +} diff --git a/desktop/src/features/settings/ui/voiceSettingsLogic.test.mjs b/desktop/src/features/settings/ui/voiceSettingsLogic.test.mjs new file mode 100644 index 000000000..d9e5ce9d0 --- /dev/null +++ b/desktop/src/features/settings/ui/voiceSettingsLogic.test.mjs @@ -0,0 +1,62 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { + selectedVoiceForBackend, + voiceOptionLabel, + voicesForBackend, +} from "./voiceSettingsLogic.ts"; + +const voice = (key, displayName, fallbackKey = "pocket:mary") => ({ + key, + displayName, + backend: "pocket", + backendName: "Pocket TTS", + availability: "bundled", + fallbackKey, + referenceFile: `${key}.wav`, + provenance: { + source: "bundled", + contentHash: null, + license: null, + sourceUrl: null, + }, +}); + +test("Pocket-only V1 filters the shared registry by backend", () => { + const registry = [ + voice("pocket:mary", "Mary", null), + { ...voice("siri:aaron", "Aaron"), backend: "siri" }, + ]; + assert.deepEqual( + voicesForBackend(registry, "pocket").map((entry) => entry.key), + ["pocket:mary"], + ); +}); + +test("local selection uses the first compatible qualified preference", () => { + const voices = [ + voice("pocket:mary", "Mary", null), + voice("pocket:marius", "Marius"), + ]; + assert.equal( + selectedVoiceForBackend( + ["siri:aaron", "pocket:marius", "pocket:mary"], + voices, + )?.key, + "pocket:marius", + ); +}); + +test("duplicate display labels remain distinct by content-derived key", () => { + const voices = [ + voice("pocket:imported:aaa", "Jim"), + voice("pocket:imported:bbb", "Jim"), + ]; + assert.equal( + selectedVoiceForBackend(["pocket:imported:bbb"], voices)?.key, + "pocket:imported:bbb", + ); + assert.equal(voiceOptionLabel(voices[0], voices), "Jim · aaa"); + assert.equal(voiceOptionLabel(voices[1], voices), "Jim · bbb"); +}); diff --git a/desktop/src/features/settings/ui/voiceSettingsLogic.ts b/desktop/src/features/settings/ui/voiceSettingsLogic.ts new file mode 100644 index 000000000..30352f2d0 --- /dev/null +++ b/desktop/src/features/settings/ui/voiceSettingsLogic.ts @@ -0,0 +1,57 @@ +export type VoiceAvailability = + | "bundled" + | "installed" + | "downloadable" + | "unavailable"; + +export type VoiceRegistryEntry = { + key: string; + displayName: string; + backend: string; + backendName: string; + availability: VoiceAvailability; + fallbackKey: string | null; + referenceFile: string | null; + provenance: { + source: string; + contentHash: string | null; + license: string | null; + sourceUrl: string | null; + }; +}; + +export function voicesForBackend( + registry: readonly VoiceRegistryEntry[], + backend: string, +): VoiceRegistryEntry[] { + return registry.filter( + (voice) => + voice.backend === backend && + (voice.availability === "bundled" || voice.availability === "installed"), + ); +} + +export function selectedVoiceForBackend( + preferences: readonly string[], + voices: readonly VoiceRegistryEntry[], +): VoiceRegistryEntry | undefined { + for (const key of preferences) { + const voice = voices.find((candidate) => candidate.key === key); + if (voice) return voice; + } + return voices.find((voice) => voice.fallbackKey === null) ?? voices[0]; +} + +export function voiceOptionLabel( + voice: VoiceRegistryEntry, + voices: readonly VoiceRegistryEntry[], +): string { + const duplicateLabel = voices.some( + (candidate) => + candidate.key !== voice.key && + candidate.displayName === voice.displayName, + ); + if (!duplicateLabel) return voice.displayName; + const identitySuffix = voice.key.split(":").at(-1)?.slice(-8) ?? voice.key; + return `${voice.displayName} · ${identitySuffix}`; +} diff --git a/desktop/src/shared/api/relayChannelFilters.test.mjs b/desktop/src/shared/api/relayChannelFilters.test.mjs index 3519c8214..7b8979904 100644 --- a/desktop/src/shared/api/relayChannelFilters.test.mjs +++ b/desktop/src/shared/api/relayChannelFilters.test.mjs @@ -6,6 +6,7 @@ import { buildChannelAuxFilter, buildChannelReactionAuxFilter, buildChannelStructuralAuxFilter, + buildHuddleTtsLiveFilter, } from "./relayChannelFilters.ts"; const CHANNEL = "36411e44-0e2d-4cfe-bd6e-567eb169db9f"; @@ -14,6 +15,14 @@ const IDS = [ "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", ]; +test("huddle TTS filter is future-only kind:9 with no same-second replay", () => { + assert.deepEqual(buildHuddleTtsLiveFilter(CHANNEL), { + kinds: [9], + "#h": [CHANNEL], + limit: 0, + }); +}); + // Regression: reaction (kind:7) and reaction-removal (kind:5) events carry only // an `e` tag, no channel `h` tag. An `#h`-scoped aux query never matches them, // so removed historical reactions reappear. The aux filters must key on `#e` diff --git a/desktop/src/shared/api/relayChannelFilters.ts b/desktop/src/shared/api/relayChannelFilters.ts index d0c7e7938..9acaef45c 100644 --- a/desktop/src/shared/api/relayChannelFilters.ts +++ b/desktop/src/shared/api/relayChannelFilters.ts @@ -6,6 +6,7 @@ import { KIND_DELETION, KIND_NIP29_DELETE_EVENT, KIND_REACTION, + KIND_STREAM_MESSAGE, KIND_STREAM_MESSAGE_EDIT, } from "@/shared/constants/kinds"; import type { RelaySubscriptionFilter } from "@/shared/api/relayClientShared"; @@ -40,6 +41,17 @@ export function buildChannelFilter( return filter; } +/** Strictly live huddle message filter: zero stored rows, future kind:9 only. */ +export function buildHuddleTtsLiveFilter( + channelId: string, +): RelaySubscriptionFilter { + return { + kinds: [KIND_STREAM_MESSAGE], + "#h": [channelId], + limit: 0, + }; +} + /** * History filter for cold-load and scrollback: message kinds *only*, so the * `limit` budget buys visible message depth. Auxiliary events (reactions, diff --git a/desktop/src/testing/e2eBridge.ts b/desktop/src/testing/e2eBridge.ts index 7b13273c6..9f6c5ca51 100644 --- a/desktop/src/testing/e2eBridge.ts +++ b/desktop/src/testing/e2eBridge.ts @@ -146,6 +146,11 @@ type MockSearchProfileSeed = { type E2eConfig = { mode?: "mock" | "relay"; mock?: { + ttsSettings?: { + version: number; + agentTextToSpeech: boolean; + voicePreferences: string[]; + }; /** Advertised HEAD for the first mock project without adding that branch. */ projectHeadBranch?: string; /** Builderlab account returned by hosted-community onboarding. Null/omitted = signed out. */ @@ -9678,6 +9683,88 @@ export function maybeInstallE2eTauriMocks() { window.__BUZZ_E2E_COMMAND_LOG__?.push({ command, payload }); switch (command) { + case "get_tts_settings": + return ( + activeConfig?.mock?.ttsSettings ?? { + version: 1, + agentTextToSpeech: true, + voicePreferences: ["pocket:mary"], + } + ); + case "list_voice_registry": + return [ + { + key: "pocket:mary", + displayName: "Mary", + backend: "pocket", + backendName: "Pocket TTS", + availability: "bundled", + fallbackKey: null, + referenceFile: "reference_sample.wav", + provenance: { + source: "bundled", + contentHash: + "a35b0468382218e9f37a9a7494d1e4b74deaf18d7ced22265b4e325bb55c183f", + license: "CC-BY-4.0", + sourceUrl: "https://datashare.ed.ac.uk/handle/10283/3443", + }, + }, + { + key: "pocket:marius", + displayName: "Marius", + backend: "pocket", + backendName: "Pocket TTS", + availability: "bundled", + fallbackKey: "pocket:mary", + referenceFile: "marius.wav", + provenance: { + source: "bundled", + contentHash: + "076968c3122520f3412eb7090e8c1c3f75fe57be1e24a2f96465583d84c71e16", + license: "CC0-1.0", + sourceUrl: + "https://huggingface.co/kyutai/tts-voices/blob/323332d33f997de8394f24a193e1a76df720e01a/voice-donations/Selfie.wav", + }, + }, + ]; + case "set_tts_enabled": { + const enabled = (payload as { enabled?: boolean })?.enabled; + if (typeof enabled !== "boolean") + throw new Error("Missing text-to-speech enabled state"); + const settings = { + version: 1, + agentTextToSpeech: enabled, + voicePreferences: activeConfig?.mock?.ttsSettings + ?.voicePreferences ?? ["pocket:mary"], + }; + if (activeConfig?.mock) activeConfig.mock.ttsSettings = settings; + return settings; + } + case "set_pocket_voice": { + const voiceKey = (payload as { voiceKey?: string })?.voiceKey; + if (!voiceKey) throw new Error("Missing Pocket voice key"); + const current = activeConfig?.mock?.ttsSettings ?? { + version: 1, + agentTextToSpeech: true, + voicePreferences: ["pocket:mary"], + }; + const firstPocketIndex = current.voicePreferences.findIndex((key) => + key.startsWith("pocket:"), + ); + const preferences = current.voicePreferences.filter( + (key) => !key.startsWith("pocket:"), + ); + preferences.splice( + firstPocketIndex < 0 ? preferences.length : firstPocketIndex, + 0, + voiceKey, + ); + const settings = { ...current, voicePreferences: preferences }; + if (activeConfig?.mock) activeConfig.mock.ttsSettings = settings; + return settings; + } + case "preview_pocket_voice": + return null; case "get_builderlab_auth": return activeConfig?.mock?.builderlabAuth ?? null; case "start_builderlab_login": { diff --git a/desktop/tests/e2e/voice-settings.spec.ts b/desktop/tests/e2e/voice-settings.spec.ts new file mode 100644 index 000000000..4cc8e75fe --- /dev/null +++ b/desktop/tests/e2e/voice-settings.spec.ts @@ -0,0 +1,78 @@ +import { expect, test } from "@playwright/test"; + +import { waitForAnimations } from "../helpers/animations"; +import { installMockBridge } from "../helpers/bridge"; +import { openSettings } from "../helpers/settings"; + +const SCREENSHOT_PATH = "test-results/voice-settings/pocket-voices.png"; + +test.describe("Pocket voice settings", () => { + test.use({ viewport: { width: 1100, height: 760 } }); + + test("selects and retains a bundled voice while text to speech is off", async ({ + page, + }) => { + await installMockBridge(page); + await page.goto("/", { waitUntil: "domcontentloaded" }); + await openSettings(page, "voice"); + + const card = page.getByTestId("settings-voice"); + await expect(card).toBeVisible(); + await expect( + page.getByText("Agent text to speech", { exact: true }), + ).toBeVisible(); + + await page.getByTestId("pocket-voice-selector").click(); + await page.getByRole("menuitemradio", { name: "Marius" }).click(); + await expect(page.getByTestId("pocket-voice-selector")).toContainText( + "Marius", + ); + + await page.getByTestId("agent-text-to-speech-toggle").click(); + await expect(page.getByTestId("pocket-voice-controls")).toHaveAttribute( + "aria-disabled", + "true", + ); + await expect(page.getByTestId("pocket-voice-selector")).toContainText( + "Marius", + ); + + const savedCommands = await page.evaluate(() => + (window.__BUZZ_E2E_COMMAND_LOG__ ?? []) + .filter((entry) => + ["set_pocket_voice", "set_tts_enabled"].includes(entry.command), + ) + .map((entry) => ({ command: entry.command, payload: entry.payload })), + ); + expect(savedCommands).toEqual([ + { + command: "set_pocket_voice", + payload: { voiceKey: "pocket:marius" }, + }, + { + command: "set_tts_enabled", + payload: { enabled: false }, + }, + ]); + }); + + test("captures the complete two-voice settings surface", async ({ page }) => { + await installMockBridge(page, { + ttsSettings: { + version: 1, + agentTextToSpeech: true, + voicePreferences: ["pocket:marius"], + }, + }); + await page.goto("/", { waitUntil: "domcontentloaded" }); + await openSettings(page, "voice"); + + const card = page.getByTestId("settings-voice"); + await expect(card).toBeVisible(); + await expect(page.getByTestId("pocket-voice-selector")).toContainText( + "Marius", + ); + await waitForAnimations(page); + await card.screenshot({ path: SCREENSHOT_PATH }); + }); +}); diff --git a/desktop/tests/helpers/bridge.ts b/desktop/tests/helpers/bridge.ts index ca4d62ddd..53de38c46 100644 --- a/desktop/tests/helpers/bridge.ts +++ b/desktop/tests/helpers/bridge.ts @@ -132,6 +132,11 @@ export type MockAgentMemoryListing = { }; type MockBridgeOptions = { + ttsSettings?: { + version: number; + agentTextToSpeech: boolean; + voicePreferences: string[]; + }; /** Advertised HEAD for the first mock project without adding that branch. */ projectHeadBranch?: string; /** Relay NIP-11 identity used to sign authoritative repository state. */ diff --git a/desktop/tests/helpers/settings.ts b/desktop/tests/helpers/settings.ts index a63b52453..c26c0e919 100644 --- a/desktop/tests/helpers/settings.ts +++ b/desktop/tests/helpers/settings.ts @@ -3,6 +3,7 @@ import { expect, type Page } from "@playwright/test"; type SettingsSection = | "profile" | "notifications" + | "voice" | "agents" | "channel-templates" | "compute"