From 586d0de244e10d28e390dc880f18a32a9c6e597b Mon Sep 17 00:00:00 2001 From: klopez4212 Date: Sun, 5 Jul 2026 08:12:56 +0100 Subject: [PATCH] =?UTF-8?q?fix:=20address=20review=20comments=20=E2=80=94?= =?UTF-8?q?=20nested=20audio=20schema,=20item=20separators,=20safe=20auto-?= =?UTF-8?q?submit=20clear?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Relay: restructure OpenAI client-secrets payload to use the current typed transcription schema (audio.input.transcription) instead of the deprecated top-level input_audio_transcription field. - realtimeAudio: insert space separators between transcript items when neither the preceding nor following text has whitespace, preventing multi-utterance runs from merging into unreadable text. - useDictation: remove premature setText('') after auto-submit — the send flow handles clearing on success, so dictated text survives if a mention dialog opens or the send is blocked. Signed-off-by: klopez4212 --- crates/buzz-relay/src/api/transcribe.rs | 16 +++++++++------- .../src/features/dictation/hooks/useDictation.ts | 7 ++++++- .../src/features/dictation/lib/realtimeAudio.ts | 15 +++++++-------- 3 files changed, 22 insertions(+), 16 deletions(-) diff --git a/crates/buzz-relay/src/api/transcribe.rs b/crates/buzz-relay/src/api/transcribe.rs index f66c584e4..97263a87f 100644 --- a/crates/buzz-relay/src/api/transcribe.rs +++ b/crates/buzz-relay/src/api/transcribe.rs @@ -81,14 +81,16 @@ pub async fn create_transcribe_session( .header("Authorization", format!("Bearer {api_key}")) .header("Content-Type", "application/json") .json(&serde_json::json!({ - "session": { - "type": "transcription", - "input_audio_transcription": { - "model": model, - }, - "turn_detection": { - "type": "server_vad", + "type": "transcription", + "audio": { + "input": { + "transcription": { + "model": model, + } } + }, + "turn_detection": { + "type": "server_vad", } })) .timeout(std::time::Duration::from_secs(10)) diff --git a/desktop/src/features/dictation/hooks/useDictation.ts b/desktop/src/features/dictation/hooks/useDictation.ts index ad42ef569..5e21fed16 100644 --- a/desktop/src/features/dictation/hooks/useDictation.ts +++ b/desktop/src/features/dictation/hooks/useDictation.ts @@ -63,8 +63,13 @@ export function useDictation({ return; } + // Set the text first so the composer shows the final dictated content, + // then trigger send. We intentionally do NOT clear the composer here — + // the send flow in MessageComposer handles clearing on successful send. + // If a mention dialog opens (non-member mention), the text stays in the + // composer so the user doesn't lose their dictated message. + setText(textWithoutPhrase.trim()); onSend(textWithoutPhrase.trim()); - setText(""); lastTranscriptRef.current = ""; }, [autoSubmitPhrases, getText, onSend, isSendBlockedRef, setText], diff --git a/desktop/src/features/dictation/lib/realtimeAudio.ts b/desktop/src/features/dictation/lib/realtimeAudio.ts index 4aaaeb148..2ad6e010d 100644 --- a/desktop/src/features/dictation/lib/realtimeAudio.ts +++ b/desktop/src/features/dictation/lib/realtimeAudio.ts @@ -89,7 +89,12 @@ export function getTranscriptText(state: TranscriptSegmentState): string { for (const id of state.itemOrder) { const seg = state.items.get(id); if (!seg) continue; - result += seg.finalized ?? seg.pending; + const text = seg.finalized ?? seg.pending; + if (!text) continue; + if (result && !result.endsWith(" ") && !text.startsWith(" ")) { + result += " "; + } + result += text; } return result; } @@ -136,13 +141,7 @@ export function mergeTranscriptEvent( } // Reconstruct full text from all items in order. - let result = ""; - for (const id of state.itemOrder) { - const seg = state.items.get(id); - if (!seg) continue; - result += seg.finalized ?? seg.pending; - } - return result; + return getTranscriptText(state); } // ── Audio buffer capture ──────────────────────────────────────────────────