mirror of
https://github.com/block/buzz.git
synced 2026-08-18 06:50:31 +02:00
fix(dictation): drain queued audio on shutdown; scope stop to session
Address two P2 review comments: - stt_engine: the worker broke out of its loop as soon as the shutdown flag was set, skipping any audio already enqueued by the caller's final push_dictation_audio flush. The shutdown flush then only transcribed older speech and dropped the last words. Now drain and process everything still in the audio channel before the final flush. - dictation/useLocalDictation: stop_dictation now takes an optional session id and only tears down the engine when it matches the currently-stored session_id. stopRecording captures its session and passes it to the deferred (post-flush) stop, so restarting during the Transcribing grace window no longer lets the stale stop kill the newly-started session. Unconditional callers (cancel, unmount, abort-bail, start) pass None.
This commit is contained in:
parent
79da7cddef
commit
2d1c522789
@@ -69,7 +69,7 @@ pub async fn start_dictation(state: State<'_, AppState>) -> Result<u64, String>
|
||||
let model_dir = models::stt_model_dir().ok_or("STT model directory not found")?;
|
||||
|
||||
// Stop any existing dictation session first.
|
||||
stop_dictation_inner(&state);
|
||||
stop_dictation_inner(&state, None);
|
||||
|
||||
let config = SttEngineConfig {
|
||||
model_dir,
|
||||
@@ -119,9 +119,15 @@ pub async fn start_dictation(state: State<'_, AppState>) -> Result<u64, String>
|
||||
/// task. The `dictation-state: stopped` event is emitted by the forwarder
|
||||
/// after all pending transcripts have been forwarded, ensuring the frontend
|
||||
/// receives the final text before the stopped signal.
|
||||
///
|
||||
/// `session` scopes the stop to a specific session: the engine is only torn
|
||||
/// down when the currently-stored `session_id` matches. This prevents a
|
||||
/// delayed/fire-and-forget stop from an old session (e.g. one deferred behind
|
||||
/// a final audio flush) from killing a *newer* session the user started in the
|
||||
/// meantime. Pass `None` for an unconditional stop (used on cancel/unmount).
|
||||
#[tauri::command]
|
||||
pub fn stop_dictation(state: State<'_, AppState>) -> Result<(), String> {
|
||||
stop_dictation_inner(&state);
|
||||
pub fn stop_dictation(session: Option<u64>, state: State<'_, AppState>) -> Result<(), String> {
|
||||
stop_dictation_inner(&state, session);
|
||||
// Note: `stopped` is emitted by the forwarder task after draining all
|
||||
// pending transcripts — not here. This avoids a race where the frontend
|
||||
// sees `stopped` before the final transcript arrives.
|
||||
@@ -190,13 +196,18 @@ pub struct DictationStatus {
|
||||
|
||||
// ── Internal helpers ──────────────────────────────────────────────────────────
|
||||
|
||||
fn stop_dictation_inner(state: &AppState) {
|
||||
fn stop_dictation_inner(state: &AppState, session: Option<u64>) {
|
||||
let old_engine = {
|
||||
let mut ds = state
|
||||
.dictation_state
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner());
|
||||
ds.engine.take()
|
||||
// Session-scoped stop: only tear down when the requested session matches
|
||||
// the one currently stored. A `None` session stops unconditionally.
|
||||
match session {
|
||||
Some(requested) if requested != ds.session_id => None,
|
||||
_ => ds.engine.take(),
|
||||
}
|
||||
};
|
||||
if let Some(engine) = old_engine {
|
||||
engine.shutdown();
|
||||
|
||||
@@ -344,6 +344,42 @@ fn stt_worker(
|
||||
}
|
||||
}
|
||||
|
||||
// ── Drain any audio still queued at shutdown ──────────────────────────────
|
||||
// The caller flushes its final PCM batch (via `push_dictation_audio`) right
|
||||
// before calling `stop_dictation`, which sets the shutdown flag. Without
|
||||
// draining here, the loop above would break on the flag and skip that
|
||||
// enqueued batch, so the tail of speech would never reach `speech_buf` and
|
||||
// the final flush below would only transcribe older audio — dropping the
|
||||
// last words. Process everything still in the channel before flushing.
|
||||
while let Ok(bytes) = audio_rx.try_recv() {
|
||||
let samples_48k = bytes_to_f32(&bytes);
|
||||
input_buf_48k.extend_from_slice(&samples_48k);
|
||||
|
||||
while input_buf_48k.len() >= chunk_in {
|
||||
let chunk: Vec<f32> = input_buf_48k.drain(..chunk_in).collect();
|
||||
let resampled = resample_chunk(&mut resampler, &chunk);
|
||||
process_16k_samples(
|
||||
&resampled,
|
||||
&mut leftover_16k,
|
||||
&mut vad,
|
||||
&mut speech_buf,
|
||||
&mut silence_frames,
|
||||
&mut in_speech,
|
||||
&mut barge_in_frames,
|
||||
&recognizer,
|
||||
&text_tx,
|
||||
has_tts,
|
||||
&tts_active_flag,
|
||||
tts_cancel_flag.as_deref(),
|
||||
&mut tts_stopped_at,
|
||||
ptt_active_flag.as_ref(),
|
||||
config.silence_flush_frames,
|
||||
config.max_speech_samples,
|
||||
config.partial_flush_samples,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// ── Final flush — transcribe any remaining speech on shutdown/disconnect ──
|
||||
if !speech_buf.is_empty() {
|
||||
flush_to_stt(&speech_buf, &recognizer, &text_tx);
|
||||
|
||||
@@ -417,8 +417,12 @@ export function useLocalDictation({
|
||||
// Flush remaining audio and THEN stop the native engine, ensuring the
|
||||
// final batch arrives before the engine shuts down and flushes its buffer.
|
||||
// isTranscribing stays true — cleared when `dictation-state: stopped` arrives.
|
||||
// Scope the stop to THIS session: if the user restarts during the flush
|
||||
// window, a new session may already be stored by the time this resolves —
|
||||
// passing the session keeps this stop from tearing down the new engine.
|
||||
const stoppingSession = nativeSessionRef.current;
|
||||
void flushAudioBatch().then(() => {
|
||||
invoke("stop_dictation").catch(() => {});
|
||||
invoke("stop_dictation", { session: stoppingSession }).catch(() => {});
|
||||
});
|
||||
}, [flushAudioBatch]);
|
||||
|
||||
|
||||
Reference in New Issue
Block a user