From 59b60dbac18554aff186f27bff2eeafa3030d103 Mon Sep 17 00:00:00 2001 From: John Tennant Date: Tue, 28 Jul 2026 16:27:42 -0400 Subject: [PATCH] fix(desktop): stream Pocket model units without seams Signed-off-by: John Tennant --- desktop/src-tauri/src/huddle/pocket.rs | 12 ++ desktop/src-tauri/src/huddle/tts.rs | 146 ++++++++++++++-------- desktop/src-tauri/src/huddle/tts_tests.rs | 33 ++++- 3 files changed, 132 insertions(+), 59 deletions(-) diff --git a/desktop/src-tauri/src/huddle/pocket.rs b/desktop/src-tauri/src/huddle/pocket.rs index 02db11d42..2154a25c2 100644 --- a/desktop/src-tauri/src/huddle/pocket.rs +++ b/desktop/src-tauri/src/huddle/pocket.rs @@ -87,6 +87,18 @@ pub fn load_text_to_speech(model_dir: &str) -> Result { } impl PocketTts { + /// Split text into synthesis units that satisfy the bundle's exact + /// 50-token input limit. + pub fn split_text_into_chunks(&self, text: &str) -> Result, String> { + let Some(prepared) = prepare_april_prompt(text) else { + return Ok(Vec::new()); + }; + self.inner + .lock() + .map_err(|_| "Pocket TTS engine lock poisoned".to_string())? + .split_prompt(&prepared) + } + /// Synthesize text with the supplied reference voice. /// /// Pocket detects language from text and this model uses one synthesis diff --git a/desktop/src-tauri/src/huddle/tts.rs b/desktop/src-tauri/src/huddle/tts.rs index 2122723f7..2a31b8989 100644 --- a/desktop/src-tauri/src/huddle/tts.rs +++ b/desktop/src-tauri/src/huddle/tts.rs @@ -488,18 +488,17 @@ fn tts_worker( // Split into sentences, then group into synthesis chunks: the first // sentence stays alone (fast time-to-first-audio), the rest pack - // greedily up to MAX_CHUNK_CHARS. Each chunk is one `generate()` - // call; playback of chunk N overlaps synthesis of chunk N+1 - // (lookahead pipelining). The Pocket engine applies its exact 50-token - // limit internally; keeping those internal units within one playback - // chunk avoids adding fades and pauses at token-only boundaries. + // greedily up to MAX_CHUNK_CHARS. Playback of each model unit overlaps + // synthesis of the next one. The Pocket engine applies its exact + // 50-token split; keeping those units within one playback chunk avoids + // adding fades and pauses at token-only boundaries. let sentences: Vec = split_sentences(&text) .into_iter() .filter(|s| !s.trim().is_empty()) .collect(); let chunks = group_sentences_into_chunks(&sentences, MAX_CHUNK_CHARS); - for chunk in &chunks { + 'playback_chunks: for chunk in &chunks { if handle_cancel_or_shutdown( &cancel, &shutdown, @@ -516,51 +515,76 @@ fn tts_worker( continue; } - match engine.synth_chunk(text, "en", &style, SYNTH_STEPS) { - Ok(samples) if !samples.is_empty() => { - let mut audio = clamp_to_full_scale(samples); - // Fade-out only — fading-in would attenuate the consonant - // onset (see `apply_fade_out` docstring + the - // 2026-05-18 "first little sound is missing" regression). - apply_fade_out(&mut audio); - - // Build one contiguous buffer per synthesized sentence: - // lead-in cushion + audio + trailing gap. Keeping this as - // a single rodio source preserves the original queue/drain - // semantics (one append per sentence) while still giving - // every chunk a quiet device warm-up window. - let buf = - build_sentence_append_buffer(&mut first_append, audio, silence_buf_len); - - // Check-and-append under `player_ops`, serialized with - // the monitor: a barge-in may have arrived during - // synthesis (the blocking window the monitor thread - // exists for). Don't append the now-stale sentence — the - // human interrupted; speaking it anyway would talk over - // them. Holding the lock for the check + append means the - // monitor can never clear between our check passing and - // the buffer landing. The flag is deliberately NOT - // consumed here: the loop-top handle_cancel_or_shutdown - // does the full consume (drain queue, reset lead-in) on - // the next iteration. - let _ops = lock_player_ops(&player_ops); - if cancel.load(Ordering::Acquire) { - // Nothing appended; the loop-top consume re-arms - // `first_append` (the flag is still set — the worker - // is its only consumer). - break; - } - player.append(SamplesBuffer::new(channels, rate, buf)); - // NOTE: tts_active is set AFTER player.append(), not - // before. Setting it before synthesis would cause STT to - // discard user speech during the synthesis window as - // "echo" even though no audio is actually playing yet. - // See crossfire review C3. - tts_active.store(true, Ordering::Release); + let model_chunks = match engine.split_text_into_chunks(text) { + Ok(model_chunks) => model_chunks, + Err(error) => { + eprintln!("buzz-desktop: TTS chunking failed: {error}"); + break; } - Ok(_) => {} - Err(e) => { - eprintln!("buzz-desktop: TTS synth failed: {e}"); + }; + let model_chunk_count = model_chunks.len(); + for (model_chunk_index, model_chunk) in model_chunks.iter().enumerate() { + if handle_cancel_or_shutdown( + &cancel, + &shutdown, + &tts_active, + &text_rx, + Some((&player, &player_ops)), + ) { + first_append = true; + break 'playback_chunks; + } + + let ends_playback_chunk = model_chunk_index + 1 == model_chunk_count; + match engine.synth_chunk(model_chunk, "en", &style, SYNTH_STEPS) { + Ok(samples) if !samples.is_empty() => { + let mut audio = clamp_to_full_scale(samples); + if ends_playback_chunk { + // Fade only at the playback-chunk boundary. Applying + // it at the model's internal token boundary would + // create an audible dip between contiguous units. + apply_fade_out(&mut audio); + } + + let buf = build_sentence_append_buffer( + &mut first_append, + audio, + silence_buf_len, + model_chunk_index == 0, + ends_playback_chunk, + ); + + // Check-and-append under `player_ops`, serialized with + // the monitor: a barge-in may have arrived during + // synthesis (the blocking window the monitor thread + // exists for). Don't append the now-stale sentence — the + // human interrupted; speaking it anyway would talk over + // them. Holding the lock for the check + append means the + // monitor can never clear between our check passing and + // the buffer landing. The flag is deliberately NOT + // consumed here: the loop-top handle_cancel_or_shutdown + // does the full consume (drain queue, reset lead-in) on + // the next iteration. + let _ops = lock_player_ops(&player_ops); + if cancel.load(Ordering::Acquire) { + // Nothing appended; the loop-top consume re-arms + // `first_append` (the flag is still set — the worker + // is its only consumer). + break; + } + player.append(SamplesBuffer::new(channels, rate, buf)); + // NOTE: tts_active is set AFTER player.append(), not + // before. Setting it before synthesis would cause STT to + // discard user speech during the synthesis window as + // "echo" even though no audio is actually playing yet. + // See crossfire review C3. + tts_active.store(true, Ordering::Release); + } + Ok(_) => {} + Err(e) => { + eprintln!("buzz-desktop: TTS synth failed: {e}"); + break 'playback_chunks; + } } } } @@ -696,18 +720,34 @@ fn apply_fade_out(samples: &mut [f32]) { /// The worker uses it in the idle branch of the main loop to distinguish /// "never queued anything since last drain" from "drained after speaking", /// which controls when `tts_active` is released and the lead-in re-armed. +/// Add silence only at the outer playback boundary. +/// +/// A playback chunk may contain several model-sized synthesis units. The first +/// unit receives the onset cushion, the last receives the remaining gap, and +/// intermediate units stay sample-contiguous. fn build_sentence_append_buffer( first_append: &mut bool, audio: Vec, silence_buf_len: usize, + starts_playback_chunk: bool, + ends_playback_chunk: bool, ) -> Vec { if *first_append { *first_append = false; } - let trailing_silence_len = silence_buf_len.saturating_sub(SENTENCE_LEAD_IN_SAMPLES); - let mut buf = Vec::with_capacity(SENTENCE_LEAD_IN_SAMPLES + audio.len() + trailing_silence_len); - buf.extend(std::iter::repeat_n(0.0_f32, SENTENCE_LEAD_IN_SAMPLES)); + let lead_in_len = if starts_playback_chunk { + SENTENCE_LEAD_IN_SAMPLES + } else { + 0 + }; + let trailing_silence_len = if ends_playback_chunk { + silence_buf_len.saturating_sub(SENTENCE_LEAD_IN_SAMPLES) + } else { + 0 + }; + let mut buf = Vec::with_capacity(lead_in_len + audio.len() + trailing_silence_len); + buf.extend(std::iter::repeat_n(0.0_f32, lead_in_len)); buf.extend(audio); buf.extend(std::iter::repeat_n(0.0_f32, trailing_silence_len)); buf diff --git a/desktop/src-tauri/src/huddle/tts_tests.rs b/desktop/src-tauri/src/huddle/tts_tests.rs index 585850c69..79825e226 100644 --- a/desktop/src-tauri/src/huddle/tts_tests.rs +++ b/desktop/src-tauri/src/huddle/tts_tests.rs @@ -812,6 +812,8 @@ fn lead_in_pad_is_present_for_every_sentence_chunk() { &mut first, vec![0.5_f32; SENTENCE_AUDIO_LEN], SILENCE_BUF_LEN, + true, + true, ); assert_eq!(buf.len(), SENTENCE_AUDIO_LEN + SILENCE_BUF_LEN); @@ -840,11 +842,11 @@ fn lead_in_pad_is_present_for_every_sentence_chunk() { #[test] fn build_sentence_append_buffer_flips_first_append() { let mut first = true; - let _ = build_sentence_append_buffer(&mut first, vec![0.5; 100], 2400); + let _ = build_sentence_append_buffer(&mut first, vec![0.5; 100], 2400, true, true); assert!(!first, "first call must flip the flag"); // Subsequent call: still has a per-sentence lead-in, flag stays false. - let buf = build_sentence_append_buffer(&mut first, vec![0.5; 100], 2400); + let buf = build_sentence_append_buffer(&mut first, vec![0.5; 100], 2400, true, true); assert!(buf[..SENTENCE_LEAD_IN_SAMPLES].iter().all(|&s| s == 0.0)); assert!(!first); } @@ -853,7 +855,7 @@ fn build_sentence_append_buffer_flips_first_append() { #[test] fn first_sentence_leading_silence_is_exactly_lead_in() { let mut first = true; - let buf = build_sentence_append_buffer(&mut first, vec![0.5; 100], 2400); + let buf = build_sentence_append_buffer(&mut first, vec![0.5; 100], 2400, true, true); assert!(buf[..SENTENCE_LEAD_IN_SAMPLES].iter().all(|&s| s == 0.0)); assert_eq!(buf[SENTENCE_LEAD_IN_SAMPLES], 0.5); } @@ -863,8 +865,10 @@ fn first_sentence_leading_silence_is_exactly_lead_in() { fn sentence_gap_budget_is_preserved() { let mut first = true; let silence_buf_len = 2400; - let first_buf = build_sentence_append_buffer(&mut first, vec![0.5; 100], silence_buf_len); - let second_buf = build_sentence_append_buffer(&mut first, vec![0.5; 100], silence_buf_len); + let first_buf = + build_sentence_append_buffer(&mut first, vec![0.5; 100], silence_buf_len, true, true); + let second_buf = + build_sentence_append_buffer(&mut first, vec![0.5; 100], silence_buf_len, true, true); let first_tail = &first_buf[SENTENCE_LEAD_IN_SAMPLES + 100..]; let second_lead = &second_buf[..SENTENCE_LEAD_IN_SAMPLES]; @@ -877,7 +881,7 @@ fn sentence_gap_budget_is_preserved() { #[test] fn sentence_append_buffer_is_one_contiguous_source() { let mut first = true; - let buf = build_sentence_append_buffer(&mut first, vec![0.5; 100], 2400); + let buf = build_sentence_append_buffer(&mut first, vec![0.5; 100], 2400, true, true); assert_eq!(buf.len(), 2400 + 100); assert!(buf[..SENTENCE_LEAD_IN_SAMPLES].iter().all(|&s| s == 0.0)); @@ -888,6 +892,23 @@ fn sentence_append_buffer_is_one_contiguous_source() { ); } +/// Model-token splits remain contiguous: only the playback chunk as a whole +/// receives its onset cushion and trailing sentence gap. +#[test] +fn token_split_units_do_not_add_sentence_boundary_padding() { + let mut first = true; + let silence_buf_len = 2400; + let first_unit = + build_sentence_append_buffer(&mut first, vec![0.5; 100], silence_buf_len, true, false); + let last_unit = + build_sentence_append_buffer(&mut first, vec![0.25; 100], silence_buf_len, false, true); + + assert_eq!(first_unit.len(), SENTENCE_LEAD_IN_SAMPLES + 100); + assert_eq!(first_unit.last(), Some(&0.5)); + assert_eq!(last_unit.first(), Some(&0.25)); + assert_eq!(first_unit.len() + last_unit.len(), 200 + silence_buf_len); +} + // ── clamp_to_full_scale tests ───────────────────────────────────────────── /// In-range speech audio passes through bit-exact — no gain is applied.