From e8f288d32279b56ace26075d8bcc568826062317 Mon Sep 17 00:00:00 2001 From: npub13fn4ahfnvaa2qwylvegdgeajqs0mph6v4qsw4jcqnw4mjh3hzh2quuucm5 <8a675edd33677aa0389f6650d467b2041fb0df4ca820eacb009babb95e3715d4@sprout-oss.stage.blox.sqprod.co> Date: Tue, 7 Jul 2026 12:18:29 +0100 Subject: [PATCH] fix(dictation): cancel during transcribing window + session-scope audio pushes Address two codex P2s on cross-session/late-arrival races: 1. Cancel dictation during the final transcribing window. The draftKey cancel effect only checked isRecording/isStarting, but stopRecording() keeps the transcript listener alive (isTranscribing=true) until the native stopped event. A channel/thread switch during that grace window didn't cancel, so a late transcript leaked into the newly restored draft. Include isTranscribing in the ownership check. 2. Scope push_dictation_audio to the owning session. Late flush chunks from a just-stopped session could be fed into a newer session's engine and transcribed into the new draft. Prepend an 8-byte LE u64 session header to each raw audio payload; native only feeds audio whose header matches the active session_id and drops stale chunks. --- desktop/src-tauri/src/dictation.rs | 42 ++++++++++++++--- .../dictation/hooks/useComposerDictation.ts | 20 ++++++--- .../dictation/hooks/useLocalDictation.ts | 45 ++++++++++++++----- 3 files changed, 85 insertions(+), 22 deletions(-) diff --git a/desktop/src-tauri/src/dictation.rs b/desktop/src-tauri/src/dictation.rs index 5dba7b767..b5351caa4 100644 --- a/desktop/src-tauri/src/dictation.rs +++ b/desktop/src-tauri/src/dictation.rs @@ -136,31 +136,59 @@ pub fn stop_dictation(session: Option, state: State<'_, AppState>) -> Resul /// `push_dictation_audio` — feed raw PCM bytes into the dictation pipeline. /// -/// Expects a raw binary body of f32 LE samples at 48 kHz mono. -/// If no dictation session is active, the bytes are silently discarded. +/// Expects a raw binary body: an 8-byte little-endian `u64` session header +/// followed by f32 LE samples at 48 kHz mono. The header scopes the push to a +/// specific session — bytes are only fed to the engine when the header matches +/// the currently-stored `session_id`. This prevents late audio from a +/// just-stopped session (whose final `flushAudioBatch()` chunks are still +/// arriving) from being accepted by a *newer* session the user started in the +/// meantime and transcribed into the new draft. +/// +/// If no dictation session is active, or the session header doesn't match the +/// active session, the bytes are silently discarded. #[tauri::command] pub fn push_dictation_audio( request: tauri::ipc::Request<'_>, state: State<'_, AppState>, ) -> Result<(), String> { - /// Maximum IPC audio batch size: 100 KB. + /// Size of the leading little-endian `u64` session header. + const SESSION_HEADER_BYTES: usize = 8; + /// Maximum IPC audio batch size (audio payload only, excluding the header): 100 KB. const MAX_AUDIO_BATCH_BYTES: usize = 100 * 1024; match request.body() { tauri::ipc::InvokeBody::Raw(bytes) => { - if bytes.len() > MAX_AUDIO_BATCH_BYTES { + if bytes.len() < SESSION_HEADER_BYTES { + return Err(format!( + "audio batch too small: {} bytes (need at least {} for session header)", + bytes.len(), + SESSION_HEADER_BYTES + )); + } + let (header, audio) = bytes.split_at(SESSION_HEADER_BYTES); + if audio.len() > MAX_AUDIO_BATCH_BYTES { return Err(format!( "audio batch too large: {} bytes (max {})", - bytes.len(), + audio.len(), MAX_AUDIO_BATCH_BYTES )); } + // `split_at` guarantees `header` is exactly `SESSION_HEADER_BYTES` long. + let session = u64::from_le_bytes( + header + .try_into() + .map_err(|_| "invalid session header".to_string())?, + ); let ds = state .dictation_state .lock() .unwrap_or_else(|e| e.into_inner()); - if let Some(ref engine) = ds.engine { - engine.push_audio(bytes.to_vec())?; + // Only feed audio tagged with the currently-active session. Late + // chunks from an old session are silently dropped. + if session == ds.session_id { + if let Some(ref engine) = ds.engine { + engine.push_audio(audio.to_vec())?; + } } Ok(()) } diff --git a/desktop/src/features/dictation/hooks/useComposerDictation.ts b/desktop/src/features/dictation/hooks/useComposerDictation.ts index 709662ccb..fc613b949 100644 --- a/desktop/src/features/dictation/hooks/useComposerDictation.ts +++ b/desktop/src/features/dictation/hooks/useComposerDictation.ts @@ -89,13 +89,23 @@ export function useComposerDictation({ // Cancel dictation when the channel/thread changes so that transcript events // from a stale local STT session don't leak into the wrong draft. - // Only cancel if this instance is actually recording — avoids killing another - // composer's session since the native engine is a singleton. - const isRecordingRef = useRef(false); - isRecordingRef.current = dictation.isRecording || dictation.isStarting; + // Only cancel if this instance owns a live/pending session — avoids killing + // another composer's session since the native engine is a singleton. + // + // Includes `isTranscribing`, not just `isRecording`/`isStarting`: + // `stopRecording()` clears `isRecording` immediately but deliberately keeps + // the transcript listener alive (isTranscribing=true) until the native + // `stopped` event, so the final local STT flush can still be appended. If the + // draftKey changes during that grace window, we must still cancel — otherwise + // the late transcript is appended through this same composer instance into the + // newly restored draft. `cancelRecording()` unlistens the transcript handler + // and performs a session-scoped native stop, so this is safe. + const isOwningSessionRef = useRef(false); + isOwningSessionRef.current = + dictation.isRecording || dictation.isStarting || dictation.isTranscribing; // biome-ignore lint/correctness/useExhaustiveDependencies: draftKey is the sole trigger useEffect(() => { - if (isRecordingRef.current) { + if (isOwningSessionRef.current) { dictation.cancelRecording(); } }, [draftKey]); diff --git a/desktop/src/features/dictation/hooks/useLocalDictation.ts b/desktop/src/features/dictation/hooks/useLocalDictation.ts index 16b7e2074..949515164 100644 --- a/desktop/src/features/dictation/hooks/useLocalDictation.ts +++ b/desktop/src/features/dictation/hooks/useLocalDictation.ts @@ -43,14 +43,22 @@ const AUDIO_BATCH_MS = 100; /** * Max samples per `push_dictation_audio` IPC call. The native command rejects - * any raw batch over 100 KB (`MAX_AUDIO_BATCH_BYTES` in `dictation.rs`); at - * 48 kHz f32 mono that is 25,600 samples (~0.53s). We chunk under that cap + * any raw audio payload over 100 KB (`MAX_AUDIO_BATCH_BYTES` in `dictation.rs`); + * at 48 kHz f32 mono that is 25,600 samples (~0.53s). We chunk under that cap * (24,000 samples / 96 KB, leaving headroom) so a stalled main thread that * lets the batch grow past ~0.5s can't produce a single oversized buffer that * native rejects and we silently drop. Chunks are sent in order. */ const MAX_IPC_SAMPLES = 24_000; +/** + * Size (bytes) of the little-endian `u64` session header prepended to each + * `push_dictation_audio` payload. Native reads this header and only feeds audio + * whose session matches the currently-active one, so late chunks from a + * just-stopped session can't leak into a newer session's transcript. + */ +const SESSION_HEADER_BYTES = 8; + /** * Local STT dictation hook using the Parakeet model via Tauri native commands. * @@ -134,6 +142,12 @@ export function useLocalDictation({ const batch = audioBatchRef.current; if (batch.length === 0) return Promise.resolve(); + // Tag this flush with the session that owns the buffered audio. Captured + // once here so every chunk of this batch carries the same session; native + // drops chunks whose session no longer matches the active one, so a late + // flush from a just-stopped session can't leak into a newer session. + const session = nativeSessionRef.current; + // Calculate total byte length and merge into a single buffer. let totalSamples = 0; for (const chunk of batch) { @@ -157,15 +171,26 @@ export function useLocalDictation({ start, Math.min(start + MAX_IPC_SAMPLES, merged.length), ); - // Copy into a fresh, tightly-bound buffer so the raw IPC payload is - // exactly this chunk (a subarray view shares the parent's ArrayBuffer). - const chunk = new Uint8Array( - slice.buffer.slice( - slice.byteOffset, - slice.byteOffset + slice.byteLength, - ), + // Build the IPC payload: an 8-byte LE u64 session header followed by + // the audio bytes. Copy the slice into the header-prefixed buffer so + // the raw payload is exactly this chunk (a subarray view shares the + // parent's ArrayBuffer). + const payload = new Uint8Array(SESSION_HEADER_BYTES + slice.byteLength); + new DataView(payload.buffer).setBigUint64( + 0, + BigInt(session), + true, // little-endian ); - await invokeRawBinary("push_dictation_audio", chunk).catch(() => {}); + payload.set( + new Uint8Array( + slice.buffer.slice( + slice.byteOffset, + slice.byteOffset + slice.byteLength, + ), + ), + SESSION_HEADER_BYTES, + ); + await invokeRawBinary("push_dictation_audio", payload).catch(() => {}); } }; return sendChunks();