fix(dictation): stream transcripts on-the-fly and suppress accent picker

Three fixes for the push-to-talk dictation experience:

1. **Streaming partial transcripts**: The STT engine now flushes a partial
   transcript every ~2 seconds of continuous speech (DICTATION_PARTIAL_FLUSH_SAMPLES).
   Previously, text only appeared after a silence gap or when the user released
   the key. Now text streams into the composer while you're still speaking.

2. **Final flush on shutdown**: When the STT worker exits (key released or
   stop_dictation called), it flushes any remaining speech buffer before
   shutting down. Previously, speech accumulated since the last silence gap
   was silently discarded.

3. **Suppress macOS accent picker**: Holding ⌘D triggered the macOS
   press-and-hold accent character popup (showing ð, đ, etc.). Fixed by
   setting ApplePressAndHoldEnabled=false for the app's bundle ID at startup.

Also fixes the event ordering: the 'dictation-state: stopped' event is now
emitted by the forwarder task after all pending transcripts have been
delivered, ensuring the frontend receives the final text before the stopped
signal.
This commit is contained in:
klopez4212
2026-07-11 16:18:41 +01:00
parent 1bcfcd4171
commit 17c3674468
6 changed files with 97 additions and 18 deletions
+17 -12
View File
@@ -18,7 +18,8 @@ use tauri::{Emitter, State};
use crate::app_state::AppState;
use crate::huddle::models;
use crate::stt_engine::{
SttEngine, SttEngineConfig, DEFAULT_MAX_SPEECH_SAMPLES, DICTATION_SILENCE_FLUSH_FRAMES,
SttEngine, SttEngineConfig, DEFAULT_MAX_SPEECH_SAMPLES, DICTATION_PARTIAL_FLUSH_SAMPLES,
DICTATION_SILENCE_FLUSH_FRAMES,
};
/// Tauri event name emitted when a dictation transcript segment is ready.
@@ -67,6 +68,7 @@ pub async fn start_dictation(state: State<'_, AppState>) -> Result<(), String> {
model_dir,
silence_flush_frames: DICTATION_SILENCE_FLUSH_FRAMES,
max_speech_samples: DEFAULT_MAX_SPEECH_SAMPLES,
partial_flush_samples: Some(DICTATION_PARTIAL_FLUSH_SAMPLES),
tts_active: None,
tts_cancel: None,
ptt_active: None,
@@ -100,20 +102,17 @@ pub async fn start_dictation(state: State<'_, AppState>) -> Result<(), String> {
}
/// `stop_dictation` — stop the active dictation session.
///
/// The final transcript (if any) is emitted asynchronously by the forwarder
/// task. The `dictation-state: stopped` event is emitted by the forwarder
/// after all pending transcripts have been forwarded, ensuring the frontend
/// receives the final text before the stopped signal.
#[tauri::command]
pub fn stop_dictation(state: State<'_, AppState>) -> Result<(), String> {
stop_dictation_inner(&state);
// Emit state change to frontend.
if let Some(handle) = state
.app_handle
.lock()
.unwrap_or_else(|e| e.into_inner())
.as_ref()
{
let _ = handle.emit(DICTATION_STATE_EVENT, "stopped");
}
// Note: `stopped` is emitted by the forwarder task after draining all
// pending transcripts — not here. This avoids a race where the frontend
// sees `stopped` before the final transcript arrives.
Ok(())
}
@@ -195,6 +194,10 @@ fn stop_dictation_inner(state: &AppState) {
}
/// Spawn an async task that reads transcribed text and emits Tauri events.
///
/// When the channel closes (engine stopped), the forwarder emits
/// `dictation-state: stopped` so the frontend knows all pending transcripts
/// have been delivered.
fn spawn_dictation_forwarder(
mut text_rx: tokio::sync::mpsc::Receiver<String>,
app_handle: tauri::AppHandle,
@@ -208,5 +211,7 @@ fn spawn_dictation_forwarder(
break; // App window closed.
}
}
// All transcripts forwarded — signal the frontend that dictation is done.
let _ = app_handle.emit(DICTATION_STATE_EVENT, "stopped");
});
}
+1
View File
@@ -67,6 +67,7 @@ impl SttPipeline {
model_dir,
silence_flush_frames: DEFAULT_SILENCE_FLUSH_FRAMES,
max_speech_samples: DEFAULT_MAX_SPEECH_SAMPLES,
partial_flush_samples: None,
tts_active: Some(tts_active),
tts_cancel,
ptt_active,
+22
View File
@@ -156,6 +156,28 @@ fn toggle_maximize(window: &tauri::Window) {
#[cfg_attr(mobile, tauri::mobile_entry_point)]
pub fn run() {
// Disable macOS press-and-hold accent picker so that holding shortcut keys
// (e.g. ⌘D for dictation push-to-talk) doesn't trigger the character popup.
// Writes to the app's own domain — does not affect other apps.
#[cfg(target_os = "macos")]
{
// Use the bundle ID for production; dev builds use the .dev suffix.
let bundle_id = if cfg!(debug_assertions) {
"xyz.block.buzz.app.dev"
} else {
"xyz.block.buzz.app"
};
let _ = std::process::Command::new("defaults")
.args([
"write",
bundle_id,
"ApplePressAndHoldEnabled",
"-bool",
"false",
])
.output();
}
let builder = tauri::Builder::default()
.plugin(tauri_plugin_single_instance::init(|app, argv, _cwd| {
// Focus the existing window when a duplicate instance launches.
+25
View File
@@ -40,6 +40,11 @@ pub const DEFAULT_SILENCE_FLUSH_FRAMES: usize = 19;
/// Silence frames for dictation (~400ms) — slightly longer for more coherent sentences.
pub const DICTATION_SILENCE_FLUSH_FRAMES: usize = 25;
/// Periodic partial flush interval for dictation streaming (~2 seconds of speech
/// at 16 kHz). When speech exceeds this threshold without a silence gap, the engine
/// flushes a partial transcript so the user sees text appearing on-the-fly.
pub const DICTATION_PARTIAL_FLUSH_SAMPLES: usize = 16_000 * 2;
/// Default maximum speech buffer: 30 seconds at 16 kHz.
pub const DEFAULT_MAX_SPEECH_SAMPLES: usize = 16_000 * 30;
@@ -56,6 +61,10 @@ pub struct SttEngineConfig {
pub silence_flush_frames: usize,
/// Maximum speech buffer size in samples (OOM guard).
pub max_speech_samples: usize,
/// Optional: when set, the engine flushes a partial transcript every N samples
/// of continuous speech (even without a silence gap). This enables on-the-fly
/// transcription for dictation. Set to `None` for huddle (silence-only flush).
pub partial_flush_samples: Option<usize>,
/// Optional: shared flag set by TTS while audio is playing (echo prevention).
pub tts_active: Option<Arc<AtomicBool>>,
/// Optional: TTS cancel flag — set by STT on barge-in detection.
@@ -329,10 +338,16 @@ fn stt_worker(
ptt_active_flag.as_ref(),
config.silence_flush_frames,
config.max_speech_samples,
config.partial_flush_samples,
);
}
}
}
// ── Final flush — transcribe any remaining speech on shutdown/disconnect ──
if !speech_buf.is_empty() {
flush_to_stt(&speech_buf, &recognizer, &text_tx);
}
}
/// Resample a mono 48 kHz chunk to 16 kHz using rubato.
@@ -377,6 +392,7 @@ fn process_16k_samples(
ptt_active: Option<&Arc<AtomicBool>>,
silence_flush_threshold: usize,
max_speech_samples: usize,
partial_flush_samples: Option<usize>,
) {
leftover.extend_from_slice(samples);
@@ -456,6 +472,15 @@ fn process_16k_samples(
*silence_frames = 0;
*in_speech = false;
}
// Periodic partial flush — emit intermediate transcript so text
// appears on-the-fly during continuous speech (dictation mode).
else if let Some(threshold) = partial_flush_samples {
if speech_buf.len() >= threshold {
flush_to_stt(speech_buf, recognizer, text_tx);
speech_buf.clear();
// Stay in_speech — the user is still talking.
}
}
} else if *in_speech {
speech_buf.extend_from_slice(&frame);
*silence_frames += 1;
@@ -45,7 +45,10 @@ export function useDictation({
if (!match) {
setText(merged);
lastTranscriptRef.current = transcript;
// Reset to empty — each streaming partial is an independent segment
// (the native engine flushes and clears its buffer). The next transcript
// should be appended, not replace this one.
lastTranscriptRef.current = "";
return;
}
@@ -120,7 +120,6 @@ export function useLocalDictation({
const unlistenTranscript = await listen<string>(
DICTATION_TRANSCRIPT_EVENT,
(event) => {
setIsTranscribing(false);
if (event.payload) {
onTranscriptTextRef.current(event.payload);
}
@@ -134,6 +133,15 @@ export function useLocalDictation({
if (event.payload === "stopped") {
setIsRecording(false);
setIsTranscribing(false);
// Clean up event listeners now that the session is fully done.
if (unlistenTranscriptRef.current) {
unlistenTranscriptRef.current();
unlistenTranscriptRef.current = null;
}
if (unlistenStateRef.current) {
unlistenStateRef.current();
unlistenStateRef.current = null;
}
}
},
);
@@ -230,12 +238,27 @@ export function useLocalDictation({
}, [cleanup, isEnabled, isRecording, isStarting]);
const stopRecording = useCallback(() => {
cleanup();
// Stop mic and audio pipeline immediately so the user gets visual feedback,
// but keep isTranscribing=true until the native `stopped` event arrives
// (which fires only after the final transcript has been forwarded).
if (streamRef.current) {
for (const track of streamRef.current.getTracks()) {
track.stop();
}
streamRef.current = null;
}
if (workletRef.current) {
workletRef.current.disconnect();
workletRef.current = null;
}
if (audioContextRef.current) {
void audioContextRef.current.close();
audioContextRef.current = null;
}
invoke("stop_dictation").catch(() => {});
setIsRecording(false);
// Keep isTranscribing briefly — final segment may still arrive.
setTimeout(() => setIsTranscribing(false), 500);
}, [cleanup]);
// isTranscribing stays true — cleared when `dictation-state: stopped` arrives.
}, []);
const cancelRecording = useCallback(() => {
cleanup();