diff --git a/desktop/src-tauri/src/dictation.rs b/desktop/src-tauri/src/dictation.rs index a03840d02..5dba7b767 100644 --- a/desktop/src-tauri/src/dictation.rs +++ b/desktop/src-tauri/src/dictation.rs @@ -69,7 +69,7 @@ pub async fn start_dictation(state: State<'_, AppState>) -> Result let model_dir = models::stt_model_dir().ok_or("STT model directory not found")?; // Stop any existing dictation session first. - stop_dictation_inner(&state); + stop_dictation_inner(&state, None); let config = SttEngineConfig { model_dir, @@ -119,9 +119,15 @@ pub async fn start_dictation(state: State<'_, AppState>) -> Result /// task. The `dictation-state: stopped` event is emitted by the forwarder /// after all pending transcripts have been forwarded, ensuring the frontend /// receives the final text before the stopped signal. +/// +/// `session` scopes the stop to a specific session: the engine is only torn +/// down when the currently-stored `session_id` matches. This prevents a +/// delayed/fire-and-forget stop from an old session (e.g. one deferred behind +/// a final audio flush) from killing a *newer* session the user started in the +/// meantime. Pass `None` for an unconditional stop (used on cancel/unmount). #[tauri::command] -pub fn stop_dictation(state: State<'_, AppState>) -> Result<(), String> { - stop_dictation_inner(&state); +pub fn stop_dictation(session: Option, state: State<'_, AppState>) -> Result<(), String> { + stop_dictation_inner(&state, session); // Note: `stopped` is emitted by the forwarder task after draining all // pending transcripts — not here. This avoids a race where the frontend // sees `stopped` before the final transcript arrives. @@ -190,13 +196,18 @@ pub struct DictationStatus { // ── Internal helpers ────────────────────────────────────────────────────────── -fn stop_dictation_inner(state: &AppState) { +fn stop_dictation_inner(state: &AppState, session: Option) { let old_engine = { let mut ds = state .dictation_state .lock() .unwrap_or_else(|e| e.into_inner()); - ds.engine.take() + // Session-scoped stop: only tear down when the requested session matches + // the one currently stored. A `None` session stops unconditionally. + match session { + Some(requested) if requested != ds.session_id => None, + _ => ds.engine.take(), + } }; if let Some(engine) = old_engine { engine.shutdown(); diff --git a/desktop/src-tauri/src/stt_engine.rs b/desktop/src-tauri/src/stt_engine.rs index bdc8aca4f..eefe950fa 100644 --- a/desktop/src-tauri/src/stt_engine.rs +++ b/desktop/src-tauri/src/stt_engine.rs @@ -344,6 +344,42 @@ fn stt_worker( } } + // ── Drain any audio still queued at shutdown ────────────────────────────── + // The caller flushes its final PCM batch (via `push_dictation_audio`) right + // before calling `stop_dictation`, which sets the shutdown flag. Without + // draining here, the loop above would break on the flag and skip that + // enqueued batch, so the tail of speech would never reach `speech_buf` and + // the final flush below would only transcribe older audio — dropping the + // last words. Process everything still in the channel before flushing. + while let Ok(bytes) = audio_rx.try_recv() { + let samples_48k = bytes_to_f32(&bytes); + input_buf_48k.extend_from_slice(&samples_48k); + + while input_buf_48k.len() >= chunk_in { + let chunk: Vec = input_buf_48k.drain(..chunk_in).collect(); + let resampled = resample_chunk(&mut resampler, &chunk); + process_16k_samples( + &resampled, + &mut leftover_16k, + &mut vad, + &mut speech_buf, + &mut silence_frames, + &mut in_speech, + &mut barge_in_frames, + &recognizer, + &text_tx, + has_tts, + &tts_active_flag, + tts_cancel_flag.as_deref(), + &mut tts_stopped_at, + ptt_active_flag.as_ref(), + config.silence_flush_frames, + config.max_speech_samples, + config.partial_flush_samples, + ); + } + } + // ── Final flush — transcribe any remaining speech on shutdown/disconnect ── if !speech_buf.is_empty() { flush_to_stt(&speech_buf, &recognizer, &text_tx); diff --git a/desktop/src/features/dictation/hooks/useLocalDictation.ts b/desktop/src/features/dictation/hooks/useLocalDictation.ts index 3ccebdfc4..6e3df3fbc 100644 --- a/desktop/src/features/dictation/hooks/useLocalDictation.ts +++ b/desktop/src/features/dictation/hooks/useLocalDictation.ts @@ -417,8 +417,12 @@ export function useLocalDictation({ // Flush remaining audio and THEN stop the native engine, ensuring the // final batch arrives before the engine shuts down and flushes its buffer. // isTranscribing stays true — cleared when `dictation-state: stopped` arrives. + // Scope the stop to THIS session: if the user restarts during the flush + // window, a new session may already be stored by the time this resolves — + // passing the session keeps this stop from tearing down the new engine. + const stoppingSession = nativeSessionRef.current; void flushAudioBatch().then(() => { - invoke("stop_dictation").catch(() => {}); + invoke("stop_dictation", { session: stoppingSession }).catch(() => {}); }); }, [flushAudioBatch]);