fix(dictation): cancel during transcribing window + session-scope audio pushes

Address two codex P2s on cross-session/late-arrival races:

1. Cancel dictation during the final transcribing window. The draftKey
   cancel effect only checked isRecording/isStarting, but stopRecording()
   keeps the transcript listener alive (isTranscribing=true) until the
   native stopped event. A channel/thread switch during that grace window
   didn't cancel, so a late transcript leaked into the newly restored
   draft. Include isTranscribing in the ownership check.

2. Scope push_dictation_audio to the owning session. Late flush chunks
   from a just-stopped session could be fed into a newer session's engine
   and transcribed into the new draft. Prepend an 8-byte LE u64 session
   header to each raw audio payload; native only feeds audio whose header
   matches the active session_id and drops stale chunks.
This commit is contained in:
npub13fn4ahfnvaa2qwylvegdgeajqs0mph6v4qsw4jcqnw4mjh3hzh2quuucm5
2026-07-11 16:19:22 +01:00
parent 8bbdeec000
commit e8f288d322
3 changed files with 85 additions and 22 deletions
+35 -7
View File
@@ -136,31 +136,59 @@ pub fn stop_dictation(session: Option<u64>, state: State<'_, AppState>) -> Resul
/// `push_dictation_audio` — feed raw PCM bytes into the dictation pipeline.
///
/// Expects a raw binary body of f32 LE samples at 48 kHz mono.
/// If no dictation session is active, the bytes are silently discarded.
/// Expects a raw binary body: an 8-byte little-endian `u64` session header
/// followed by f32 LE samples at 48 kHz mono. The header scopes the push to a
/// specific session — bytes are only fed to the engine when the header matches
/// the currently-stored `session_id`. This prevents late audio from a
/// just-stopped session (whose final `flushAudioBatch()` chunks are still
/// arriving) from being accepted by a *newer* session the user started in the
/// meantime and transcribed into the new draft.
///
/// If no dictation session is active, or the session header doesn't match the
/// active session, the bytes are silently discarded.
#[tauri::command]
pub fn push_dictation_audio(
request: tauri::ipc::Request<'_>,
state: State<'_, AppState>,
) -> Result<(), String> {
/// Maximum IPC audio batch size: 100 KB.
/// Size of the leading little-endian `u64` session header.
const SESSION_HEADER_BYTES: usize = 8;
/// Maximum IPC audio batch size (audio payload only, excluding the header): 100 KB.
const MAX_AUDIO_BATCH_BYTES: usize = 100 * 1024;
match request.body() {
tauri::ipc::InvokeBody::Raw(bytes) => {
if bytes.len() > MAX_AUDIO_BATCH_BYTES {
if bytes.len() < SESSION_HEADER_BYTES {
return Err(format!(
"audio batch too small: {} bytes (need at least {} for session header)",
bytes.len(),
SESSION_HEADER_BYTES
));
}
let (header, audio) = bytes.split_at(SESSION_HEADER_BYTES);
if audio.len() > MAX_AUDIO_BATCH_BYTES {
return Err(format!(
"audio batch too large: {} bytes (max {})",
bytes.len(),
audio.len(),
MAX_AUDIO_BATCH_BYTES
));
}
// `split_at` guarantees `header` is exactly `SESSION_HEADER_BYTES` long.
let session = u64::from_le_bytes(
header
.try_into()
.map_err(|_| "invalid session header".to_string())?,
);
let ds = state
.dictation_state
.lock()
.unwrap_or_else(|e| e.into_inner());
if let Some(ref engine) = ds.engine {
engine.push_audio(bytes.to_vec())?;
// Only feed audio tagged with the currently-active session. Late
// chunks from an old session are silently dropped.
if session == ds.session_id {
if let Some(ref engine) = ds.engine {
engine.push_audio(audio.to_vec())?;
}
}
Ok(())
}
@@ -89,13 +89,23 @@ export function useComposerDictation({
// Cancel dictation when the channel/thread changes so that transcript events
// from a stale local STT session don't leak into the wrong draft.
// Only cancel if this instance is actually recording — avoids killing another
// composer's session since the native engine is a singleton.
const isRecordingRef = useRef(false);
isRecordingRef.current = dictation.isRecording || dictation.isStarting;
// Only cancel if this instance owns a live/pending session — avoids killing
// another composer's session since the native engine is a singleton.
//
// Includes `isTranscribing`, not just `isRecording`/`isStarting`:
// `stopRecording()` clears `isRecording` immediately but deliberately keeps
// the transcript listener alive (isTranscribing=true) until the native
// `stopped` event, so the final local STT flush can still be appended. If the
// draftKey changes during that grace window, we must still cancel — otherwise
// the late transcript is appended through this same composer instance into the
// newly restored draft. `cancelRecording()` unlistens the transcript handler
// and performs a session-scoped native stop, so this is safe.
const isOwningSessionRef = useRef(false);
isOwningSessionRef.current =
dictation.isRecording || dictation.isStarting || dictation.isTranscribing;
// biome-ignore lint/correctness/useExhaustiveDependencies: draftKey is the sole trigger
useEffect(() => {
if (isRecordingRef.current) {
if (isOwningSessionRef.current) {
dictation.cancelRecording();
}
}, [draftKey]);
@@ -43,14 +43,22 @@ const AUDIO_BATCH_MS = 100;
/**
* Max samples per `push_dictation_audio` IPC call. The native command rejects
* any raw batch over 100 KB (`MAX_AUDIO_BATCH_BYTES` in `dictation.rs`); at
* 48 kHz f32 mono that is 25,600 samples (~0.53s). We chunk under that cap
* any raw audio payload over 100 KB (`MAX_AUDIO_BATCH_BYTES` in `dictation.rs`);
* at 48 kHz f32 mono that is 25,600 samples (~0.53s). We chunk under that cap
* (24,000 samples / 96 KB, leaving headroom) so a stalled main thread that
* lets the batch grow past ~0.5s can't produce a single oversized buffer that
* native rejects and we silently drop. Chunks are sent in order.
*/
const MAX_IPC_SAMPLES = 24_000;
/**
* Size (bytes) of the little-endian `u64` session header prepended to each
* `push_dictation_audio` payload. Native reads this header and only feeds audio
* whose session matches the currently-active one, so late chunks from a
* just-stopped session can't leak into a newer session's transcript.
*/
const SESSION_HEADER_BYTES = 8;
/**
* Local STT dictation hook using the Parakeet model via Tauri native commands.
*
@@ -134,6 +142,12 @@ export function useLocalDictation({
const batch = audioBatchRef.current;
if (batch.length === 0) return Promise.resolve();
// Tag this flush with the session that owns the buffered audio. Captured
// once here so every chunk of this batch carries the same session; native
// drops chunks whose session no longer matches the active one, so a late
// flush from a just-stopped session can't leak into a newer session.
const session = nativeSessionRef.current;
// Calculate total byte length and merge into a single buffer.
let totalSamples = 0;
for (const chunk of batch) {
@@ -157,15 +171,26 @@ export function useLocalDictation({
start,
Math.min(start + MAX_IPC_SAMPLES, merged.length),
);
// Copy into a fresh, tightly-bound buffer so the raw IPC payload is
// exactly this chunk (a subarray view shares the parent's ArrayBuffer).
const chunk = new Uint8Array(
slice.buffer.slice(
slice.byteOffset,
slice.byteOffset + slice.byteLength,
),
// Build the IPC payload: an 8-byte LE u64 session header followed by
// the audio bytes. Copy the slice into the header-prefixed buffer so
// the raw payload is exactly this chunk (a subarray view shares the
// parent's ArrayBuffer).
const payload = new Uint8Array(SESSION_HEADER_BYTES + slice.byteLength);
new DataView(payload.buffer).setBigUint64(
0,
BigInt(session),
true, // little-endian
);
await invokeRawBinary("push_dictation_audio", chunk).catch(() => {});
payload.set(
new Uint8Array(
slice.buffer.slice(
slice.byteOffset,
slice.byteOffset + slice.byteLength,
),
),
SESSION_HEADER_BYTES,
);
await invokeRawBinary("push_dictation_audio", payload).catch(() => {});
}
};
return sendChunks();