From 17c367446878db0093b3525ac2f07e2bcc103a2d Mon Sep 17 00:00:00 2001 From: klopez4212 Date: Mon, 6 Jul 2026 16:42:45 +0100 Subject: [PATCH] fix(dictation): stream transcripts on-the-fly and suppress accent picker MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three fixes for the push-to-talk dictation experience: 1. **Streaming partial transcripts**: The STT engine now flushes a partial transcript every ~2 seconds of continuous speech (DICTATION_PARTIAL_FLUSH_SAMPLES). Previously, text only appeared after a silence gap or when the user released the key. Now text streams into the composer while you're still speaking. 2. **Final flush on shutdown**: When the STT worker exits (key released or stop_dictation called), it flushes any remaining speech buffer before shutting down. Previously, speech accumulated since the last silence gap was silently discarded. 3. **Suppress macOS accent picker**: Holding ⌘D triggered the macOS press-and-hold accent character popup (showing ð, đ, etc.). Fixed by setting ApplePressAndHoldEnabled=false for the app's bundle ID at startup. Also fixes the event ordering: the 'dictation-state: stopped' event is now emitted by the forwarder task after all pending transcripts have been delivered, ensuring the frontend receives the final text before the stopped signal. --- desktop/src-tauri/src/dictation.rs | 29 +++++++++------- desktop/src-tauri/src/huddle/stt.rs | 1 + desktop/src-tauri/src/lib.rs | 22 +++++++++++++ desktop/src-tauri/src/stt_engine.rs | 25 ++++++++++++++ .../features/dictation/hooks/useDictation.ts | 5 ++- .../dictation/hooks/useLocalDictation.ts | 33 ++++++++++++++++--- 6 files changed, 97 insertions(+), 18 deletions(-) diff --git a/desktop/src-tauri/src/dictation.rs b/desktop/src-tauri/src/dictation.rs index fd4d7d562..c3493ea22 100644 --- a/desktop/src-tauri/src/dictation.rs +++ b/desktop/src-tauri/src/dictation.rs @@ -18,7 +18,8 @@ use tauri::{Emitter, State}; use crate::app_state::AppState; use crate::huddle::models; use crate::stt_engine::{ - SttEngine, SttEngineConfig, DEFAULT_MAX_SPEECH_SAMPLES, DICTATION_SILENCE_FLUSH_FRAMES, + SttEngine, SttEngineConfig, DEFAULT_MAX_SPEECH_SAMPLES, DICTATION_PARTIAL_FLUSH_SAMPLES, + DICTATION_SILENCE_FLUSH_FRAMES, }; /// Tauri event name emitted when a dictation transcript segment is ready. @@ -67,6 +68,7 @@ pub async fn start_dictation(state: State<'_, AppState>) -> Result<(), String> { model_dir, silence_flush_frames: DICTATION_SILENCE_FLUSH_FRAMES, max_speech_samples: DEFAULT_MAX_SPEECH_SAMPLES, + partial_flush_samples: Some(DICTATION_PARTIAL_FLUSH_SAMPLES), tts_active: None, tts_cancel: None, ptt_active: None, @@ -100,20 +102,17 @@ pub async fn start_dictation(state: State<'_, AppState>) -> Result<(), String> { } /// `stop_dictation` — stop the active dictation session. +/// +/// The final transcript (if any) is emitted asynchronously by the forwarder +/// task. The `dictation-state: stopped` event is emitted by the forwarder +/// after all pending transcripts have been forwarded, ensuring the frontend +/// receives the final text before the stopped signal. #[tauri::command] pub fn stop_dictation(state: State<'_, AppState>) -> Result<(), String> { stop_dictation_inner(&state); - - // Emit state change to frontend. - if let Some(handle) = state - .app_handle - .lock() - .unwrap_or_else(|e| e.into_inner()) - .as_ref() - { - let _ = handle.emit(DICTATION_STATE_EVENT, "stopped"); - } - + // Note: `stopped` is emitted by the forwarder task after draining all + // pending transcripts — not here. This avoids a race where the frontend + // sees `stopped` before the final transcript arrives. Ok(()) } @@ -195,6 +194,10 @@ fn stop_dictation_inner(state: &AppState) { } /// Spawn an async task that reads transcribed text and emits Tauri events. +/// +/// When the channel closes (engine stopped), the forwarder emits +/// `dictation-state: stopped` so the frontend knows all pending transcripts +/// have been delivered. fn spawn_dictation_forwarder( mut text_rx: tokio::sync::mpsc::Receiver, app_handle: tauri::AppHandle, @@ -208,5 +211,7 @@ fn spawn_dictation_forwarder( break; // App window closed. } } + // All transcripts forwarded — signal the frontend that dictation is done. + let _ = app_handle.emit(DICTATION_STATE_EVENT, "stopped"); }); } diff --git a/desktop/src-tauri/src/huddle/stt.rs b/desktop/src-tauri/src/huddle/stt.rs index 5bbfe630f..de2be1031 100644 --- a/desktop/src-tauri/src/huddle/stt.rs +++ b/desktop/src-tauri/src/huddle/stt.rs @@ -67,6 +67,7 @@ impl SttPipeline { model_dir, silence_flush_frames: DEFAULT_SILENCE_FLUSH_FRAMES, max_speech_samples: DEFAULT_MAX_SPEECH_SAMPLES, + partial_flush_samples: None, tts_active: Some(tts_active), tts_cancel, ptt_active, diff --git a/desktop/src-tauri/src/lib.rs b/desktop/src-tauri/src/lib.rs index 6c31728a2..832453e8d 100644 --- a/desktop/src-tauri/src/lib.rs +++ b/desktop/src-tauri/src/lib.rs @@ -156,6 +156,28 @@ fn toggle_maximize(window: &tauri::Window) { #[cfg_attr(mobile, tauri::mobile_entry_point)] pub fn run() { + // Disable macOS press-and-hold accent picker so that holding shortcut keys + // (e.g. ⌘D for dictation push-to-talk) doesn't trigger the character popup. + // Writes to the app's own domain — does not affect other apps. + #[cfg(target_os = "macos")] + { + // Use the bundle ID for production; dev builds use the .dev suffix. + let bundle_id = if cfg!(debug_assertions) { + "xyz.block.buzz.app.dev" + } else { + "xyz.block.buzz.app" + }; + let _ = std::process::Command::new("defaults") + .args([ + "write", + bundle_id, + "ApplePressAndHoldEnabled", + "-bool", + "false", + ]) + .output(); + } + let builder = tauri::Builder::default() .plugin(tauri_plugin_single_instance::init(|app, argv, _cwd| { // Focus the existing window when a duplicate instance launches. diff --git a/desktop/src-tauri/src/stt_engine.rs b/desktop/src-tauri/src/stt_engine.rs index 7a501d273..bdc8aca4f 100644 --- a/desktop/src-tauri/src/stt_engine.rs +++ b/desktop/src-tauri/src/stt_engine.rs @@ -40,6 +40,11 @@ pub const DEFAULT_SILENCE_FLUSH_FRAMES: usize = 19; /// Silence frames for dictation (~400ms) — slightly longer for more coherent sentences. pub const DICTATION_SILENCE_FLUSH_FRAMES: usize = 25; +/// Periodic partial flush interval for dictation streaming (~2 seconds of speech +/// at 16 kHz). When speech exceeds this threshold without a silence gap, the engine +/// flushes a partial transcript so the user sees text appearing on-the-fly. +pub const DICTATION_PARTIAL_FLUSH_SAMPLES: usize = 16_000 * 2; + /// Default maximum speech buffer: 30 seconds at 16 kHz. pub const DEFAULT_MAX_SPEECH_SAMPLES: usize = 16_000 * 30; @@ -56,6 +61,10 @@ pub struct SttEngineConfig { pub silence_flush_frames: usize, /// Maximum speech buffer size in samples (OOM guard). pub max_speech_samples: usize, + /// Optional: when set, the engine flushes a partial transcript every N samples + /// of continuous speech (even without a silence gap). This enables on-the-fly + /// transcription for dictation. Set to `None` for huddle (silence-only flush). + pub partial_flush_samples: Option, /// Optional: shared flag set by TTS while audio is playing (echo prevention). pub tts_active: Option>, /// Optional: TTS cancel flag — set by STT on barge-in detection. @@ -329,10 +338,16 @@ fn stt_worker( ptt_active_flag.as_ref(), config.silence_flush_frames, config.max_speech_samples, + config.partial_flush_samples, ); } } } + + // ── Final flush — transcribe any remaining speech on shutdown/disconnect ── + if !speech_buf.is_empty() { + flush_to_stt(&speech_buf, &recognizer, &text_tx); + } } /// Resample a mono 48 kHz chunk to 16 kHz using rubato. @@ -377,6 +392,7 @@ fn process_16k_samples( ptt_active: Option<&Arc>, silence_flush_threshold: usize, max_speech_samples: usize, + partial_flush_samples: Option, ) { leftover.extend_from_slice(samples); @@ -456,6 +472,15 @@ fn process_16k_samples( *silence_frames = 0; *in_speech = false; } + // Periodic partial flush — emit intermediate transcript so text + // appears on-the-fly during continuous speech (dictation mode). + else if let Some(threshold) = partial_flush_samples { + if speech_buf.len() >= threshold { + flush_to_stt(speech_buf, recognizer, text_tx); + speech_buf.clear(); + // Stay in_speech — the user is still talking. + } + } } else if *in_speech { speech_buf.extend_from_slice(&frame); *silence_frames += 1; diff --git a/desktop/src/features/dictation/hooks/useDictation.ts b/desktop/src/features/dictation/hooks/useDictation.ts index 9265b171d..19aa3566c 100644 --- a/desktop/src/features/dictation/hooks/useDictation.ts +++ b/desktop/src/features/dictation/hooks/useDictation.ts @@ -45,7 +45,10 @@ export function useDictation({ if (!match) { setText(merged); - lastTranscriptRef.current = transcript; + // Reset to empty — each streaming partial is an independent segment + // (the native engine flushes and clears its buffer). The next transcript + // should be appended, not replace this one. + lastTranscriptRef.current = ""; return; } diff --git a/desktop/src/features/dictation/hooks/useLocalDictation.ts b/desktop/src/features/dictation/hooks/useLocalDictation.ts index 266b70ff3..b14bcc4a7 100644 --- a/desktop/src/features/dictation/hooks/useLocalDictation.ts +++ b/desktop/src/features/dictation/hooks/useLocalDictation.ts @@ -120,7 +120,6 @@ export function useLocalDictation({ const unlistenTranscript = await listen( DICTATION_TRANSCRIPT_EVENT, (event) => { - setIsTranscribing(false); if (event.payload) { onTranscriptTextRef.current(event.payload); } @@ -134,6 +133,15 @@ export function useLocalDictation({ if (event.payload === "stopped") { setIsRecording(false); setIsTranscribing(false); + // Clean up event listeners now that the session is fully done. + if (unlistenTranscriptRef.current) { + unlistenTranscriptRef.current(); + unlistenTranscriptRef.current = null; + } + if (unlistenStateRef.current) { + unlistenStateRef.current(); + unlistenStateRef.current = null; + } } }, ); @@ -230,12 +238,27 @@ export function useLocalDictation({ }, [cleanup, isEnabled, isRecording, isStarting]); const stopRecording = useCallback(() => { - cleanup(); + // Stop mic and audio pipeline immediately so the user gets visual feedback, + // but keep isTranscribing=true until the native `stopped` event arrives + // (which fires only after the final transcript has been forwarded). + if (streamRef.current) { + for (const track of streamRef.current.getTracks()) { + track.stop(); + } + streamRef.current = null; + } + if (workletRef.current) { + workletRef.current.disconnect(); + workletRef.current = null; + } + if (audioContextRef.current) { + void audioContextRef.current.close(); + audioContextRef.current = null; + } invoke("stop_dictation").catch(() => {}); setIsRecording(false); - // Keep isTranscribing briefly — final segment may still arrive. - setTimeout(() => setIsTranscribing(false), 500); - }, [cleanup]); + // isTranscribing stays true — cleared when `dictation-state: stopped` arrives. + }, []); const cancelRecording = useCallback(() => { cleanup();