mirror of
https://github.com/block/buzz.git
synced 2026-08-18 06:50:31 +02:00
fix(dictation): cancel during transcribing window + session-scope audio pushes
Address two codex P2s on cross-session/late-arrival races: 1. Cancel dictation during the final transcribing window. The draftKey cancel effect only checked isRecording/isStarting, but stopRecording() keeps the transcript listener alive (isTranscribing=true) until the native stopped event. A channel/thread switch during that grace window didn't cancel, so a late transcript leaked into the newly restored draft. Include isTranscribing in the ownership check. 2. Scope push_dictation_audio to the owning session. Late flush chunks from a just-stopped session could be fed into a newer session's engine and transcribed into the new draft. Prepend an 8-byte LE u64 session header to each raw audio payload; native only feeds audio whose header matches the active session_id and drops stale chunks.
This commit is contained in:
parent
8bbdeec000
commit
e8f288d322
@@ -136,31 +136,59 @@ pub fn stop_dictation(session: Option<u64>, state: State<'_, AppState>) -> Resul
|
||||
|
||||
/// `push_dictation_audio` — feed raw PCM bytes into the dictation pipeline.
|
||||
///
|
||||
/// Expects a raw binary body of f32 LE samples at 48 kHz mono.
|
||||
/// If no dictation session is active, the bytes are silently discarded.
|
||||
/// Expects a raw binary body: an 8-byte little-endian `u64` session header
|
||||
/// followed by f32 LE samples at 48 kHz mono. The header scopes the push to a
|
||||
/// specific session — bytes are only fed to the engine when the header matches
|
||||
/// the currently-stored `session_id`. This prevents late audio from a
|
||||
/// just-stopped session (whose final `flushAudioBatch()` chunks are still
|
||||
/// arriving) from being accepted by a *newer* session the user started in the
|
||||
/// meantime and transcribed into the new draft.
|
||||
///
|
||||
/// If no dictation session is active, or the session header doesn't match the
|
||||
/// active session, the bytes are silently discarded.
|
||||
#[tauri::command]
|
||||
pub fn push_dictation_audio(
|
||||
request: tauri::ipc::Request<'_>,
|
||||
state: State<'_, AppState>,
|
||||
) -> Result<(), String> {
|
||||
/// Maximum IPC audio batch size: 100 KB.
|
||||
/// Size of the leading little-endian `u64` session header.
|
||||
const SESSION_HEADER_BYTES: usize = 8;
|
||||
/// Maximum IPC audio batch size (audio payload only, excluding the header): 100 KB.
|
||||
const MAX_AUDIO_BATCH_BYTES: usize = 100 * 1024;
|
||||
|
||||
match request.body() {
|
||||
tauri::ipc::InvokeBody::Raw(bytes) => {
|
||||
if bytes.len() > MAX_AUDIO_BATCH_BYTES {
|
||||
if bytes.len() < SESSION_HEADER_BYTES {
|
||||
return Err(format!(
|
||||
"audio batch too small: {} bytes (need at least {} for session header)",
|
||||
bytes.len(),
|
||||
SESSION_HEADER_BYTES
|
||||
));
|
||||
}
|
||||
let (header, audio) = bytes.split_at(SESSION_HEADER_BYTES);
|
||||
if audio.len() > MAX_AUDIO_BATCH_BYTES {
|
||||
return Err(format!(
|
||||
"audio batch too large: {} bytes (max {})",
|
||||
bytes.len(),
|
||||
audio.len(),
|
||||
MAX_AUDIO_BATCH_BYTES
|
||||
));
|
||||
}
|
||||
// `split_at` guarantees `header` is exactly `SESSION_HEADER_BYTES` long.
|
||||
let session = u64::from_le_bytes(
|
||||
header
|
||||
.try_into()
|
||||
.map_err(|_| "invalid session header".to_string())?,
|
||||
);
|
||||
let ds = state
|
||||
.dictation_state
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner());
|
||||
if let Some(ref engine) = ds.engine {
|
||||
engine.push_audio(bytes.to_vec())?;
|
||||
// Only feed audio tagged with the currently-active session. Late
|
||||
// chunks from an old session are silently dropped.
|
||||
if session == ds.session_id {
|
||||
if let Some(ref engine) = ds.engine {
|
||||
engine.push_audio(audio.to_vec())?;
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -89,13 +89,23 @@ export function useComposerDictation({
|
||||
|
||||
// Cancel dictation when the channel/thread changes so that transcript events
|
||||
// from a stale local STT session don't leak into the wrong draft.
|
||||
// Only cancel if this instance is actually recording — avoids killing another
|
||||
// composer's session since the native engine is a singleton.
|
||||
const isRecordingRef = useRef(false);
|
||||
isRecordingRef.current = dictation.isRecording || dictation.isStarting;
|
||||
// Only cancel if this instance owns a live/pending session — avoids killing
|
||||
// another composer's session since the native engine is a singleton.
|
||||
//
|
||||
// Includes `isTranscribing`, not just `isRecording`/`isStarting`:
|
||||
// `stopRecording()` clears `isRecording` immediately but deliberately keeps
|
||||
// the transcript listener alive (isTranscribing=true) until the native
|
||||
// `stopped` event, so the final local STT flush can still be appended. If the
|
||||
// draftKey changes during that grace window, we must still cancel — otherwise
|
||||
// the late transcript is appended through this same composer instance into the
|
||||
// newly restored draft. `cancelRecording()` unlistens the transcript handler
|
||||
// and performs a session-scoped native stop, so this is safe.
|
||||
const isOwningSessionRef = useRef(false);
|
||||
isOwningSessionRef.current =
|
||||
dictation.isRecording || dictation.isStarting || dictation.isTranscribing;
|
||||
// biome-ignore lint/correctness/useExhaustiveDependencies: draftKey is the sole trigger
|
||||
useEffect(() => {
|
||||
if (isRecordingRef.current) {
|
||||
if (isOwningSessionRef.current) {
|
||||
dictation.cancelRecording();
|
||||
}
|
||||
}, [draftKey]);
|
||||
|
||||
@@ -43,14 +43,22 @@ const AUDIO_BATCH_MS = 100;
|
||||
|
||||
/**
|
||||
* Max samples per `push_dictation_audio` IPC call. The native command rejects
|
||||
* any raw batch over 100 KB (`MAX_AUDIO_BATCH_BYTES` in `dictation.rs`); at
|
||||
* 48 kHz f32 mono that is 25,600 samples (~0.53s). We chunk under that cap
|
||||
* any raw audio payload over 100 KB (`MAX_AUDIO_BATCH_BYTES` in `dictation.rs`);
|
||||
* at 48 kHz f32 mono that is 25,600 samples (~0.53s). We chunk under that cap
|
||||
* (24,000 samples / 96 KB, leaving headroom) so a stalled main thread that
|
||||
* lets the batch grow past ~0.5s can't produce a single oversized buffer that
|
||||
* native rejects and we silently drop. Chunks are sent in order.
|
||||
*/
|
||||
const MAX_IPC_SAMPLES = 24_000;
|
||||
|
||||
/**
|
||||
* Size (bytes) of the little-endian `u64` session header prepended to each
|
||||
* `push_dictation_audio` payload. Native reads this header and only feeds audio
|
||||
* whose session matches the currently-active one, so late chunks from a
|
||||
* just-stopped session can't leak into a newer session's transcript.
|
||||
*/
|
||||
const SESSION_HEADER_BYTES = 8;
|
||||
|
||||
/**
|
||||
* Local STT dictation hook using the Parakeet model via Tauri native commands.
|
||||
*
|
||||
@@ -134,6 +142,12 @@ export function useLocalDictation({
|
||||
const batch = audioBatchRef.current;
|
||||
if (batch.length === 0) return Promise.resolve();
|
||||
|
||||
// Tag this flush with the session that owns the buffered audio. Captured
|
||||
// once here so every chunk of this batch carries the same session; native
|
||||
// drops chunks whose session no longer matches the active one, so a late
|
||||
// flush from a just-stopped session can't leak into a newer session.
|
||||
const session = nativeSessionRef.current;
|
||||
|
||||
// Calculate total byte length and merge into a single buffer.
|
||||
let totalSamples = 0;
|
||||
for (const chunk of batch) {
|
||||
@@ -157,15 +171,26 @@ export function useLocalDictation({
|
||||
start,
|
||||
Math.min(start + MAX_IPC_SAMPLES, merged.length),
|
||||
);
|
||||
// Copy into a fresh, tightly-bound buffer so the raw IPC payload is
|
||||
// exactly this chunk (a subarray view shares the parent's ArrayBuffer).
|
||||
const chunk = new Uint8Array(
|
||||
slice.buffer.slice(
|
||||
slice.byteOffset,
|
||||
slice.byteOffset + slice.byteLength,
|
||||
),
|
||||
// Build the IPC payload: an 8-byte LE u64 session header followed by
|
||||
// the audio bytes. Copy the slice into the header-prefixed buffer so
|
||||
// the raw payload is exactly this chunk (a subarray view shares the
|
||||
// parent's ArrayBuffer).
|
||||
const payload = new Uint8Array(SESSION_HEADER_BYTES + slice.byteLength);
|
||||
new DataView(payload.buffer).setBigUint64(
|
||||
0,
|
||||
BigInt(session),
|
||||
true, // little-endian
|
||||
);
|
||||
await invokeRawBinary("push_dictation_audio", chunk).catch(() => {});
|
||||
payload.set(
|
||||
new Uint8Array(
|
||||
slice.buffer.slice(
|
||||
slice.byteOffset,
|
||||
slice.byteOffset + slice.byteLength,
|
||||
),
|
||||
),
|
||||
SESSION_HEADER_BYTES,
|
||||
);
|
||||
await invokeRawBinary("push_dictation_audio", payload).catch(() => {});
|
||||
}
|
||||
};
|
||||
return sendChunks();
|
||||
|
||||
Reference in New Issue
Block a user