diff --git a/desktop/src/features/dictation/hooks/useRealtimeDictation.ts b/desktop/src/features/dictation/hooks/useRealtimeDictation.ts index 15a7d6ed0..1b40953b1 100644 --- a/desktop/src/features/dictation/hooks/useRealtimeDictation.ts +++ b/desktop/src/features/dictation/hooks/useRealtimeDictation.ts @@ -11,6 +11,7 @@ import { BUFFER_COMMITTED_EVENT, TRANSCRIPT_COMPLETED_EVENT, TRANSCRIPT_DELTA_EVENT, + commitAudioBuffer, connectPeerConnection, createAudioBufferCapture, createPeerConnection, @@ -18,6 +19,7 @@ import { flushAudioBuffer, getTranscriptText, mergeTranscriptEvent, + requiresManualCommit, } from "../lib/realtimeAudio"; interface UseRealtimeDictationOptions { @@ -58,6 +60,8 @@ export function useRealtimeDictation({ createTranscriptSegmentState(), ); const activeRunIdRef = useRef(0); + const manualCommitRef = useRef(false); + const commitIntervalRef = useRef | null>(null); const onRecordingStartRef = useRef(onRecordingStart); const onTranscriptTextRef = useRef(onTranscriptText); @@ -82,6 +86,45 @@ export function useRealtimeDictation({ }, []); const cleanupResources = useCallback(() => { + // Clear periodic commit interval if active. + if (commitIntervalRef.current) { + clearInterval(commitIntervalRef.current); + commitIntervalRef.current = null; + } + + // For manual-commit models (no server VAD), commit any buffered audio + // before tearing down the connection so OpenAI processes the final chunk. + // We keep the data channel open briefly to receive the transcript response. + const dc = dataChannelRef.current; + const needsCommit = + manualCommitRef.current && dc && dc.readyState === "open"; + manualCommitRef.current = false; + + if (needsCommit && dc) { + commitAudioBuffer(dc); + // Stop the mic immediately so no new audio is sent after commit. + for (const track of streamRef.current?.getTracks() ?? []) { + track.stop(); + } + streamRef.current = null; + audioCaptureRef.current?.close(); + audioCaptureRef.current = null; + // Delay full teardown to allow the final transcript to arrive. + const pc = peerConnectionRef.current; + peerConnectionRef.current = null; + dataChannelRef.current = null; + const runId = activeRunIdRef.current; + setTimeout(() => { + // Only tear down if no new run started in the meantime. + if (activeRunIdRef.current === runId) { + activeRunIdRef.current += 1; + } + dc.close(); + pc?.close(); + }, 3000); + return; + } + activeRunIdRef.current += 1; closeResources({ audioCapture: audioCaptureRef.current, @@ -181,6 +224,7 @@ export function useRealtimeDictation({ closeResources({ audioCapture, stream }); return; } + manualCommitRef.current = requiresManualCommit(session.model); // 4. Set up WebRTC peerConnection = createPeerConnection(); @@ -203,6 +247,7 @@ export function useRealtimeDictation({ // Flush buffered audio once data channel opens const channelToFlush = dataChannel; const captureToFlush = audioCapture; + const useManualCommit = manualCommitRef.current; dataChannel.addEventListener("open", () => { // If the user stopped (or restarted) recording between the SDP // exchange and the channel opening, drop this run's buffered audio. @@ -211,6 +256,20 @@ export function useRealtimeDictation({ return; } flushAudioBuffer(channelToFlush, captureToFlush.chunks); + // For manual-commit models, commit the initial buffered audio and + // start a periodic commit interval so streaming transcripts flow + // during recording (server VAD models commit automatically). + if (useManualCommit) { + commitAudioBuffer(channelToFlush); + // Commit every 2s to produce streaming transcript segments while + // the user is still speaking. Each commit triggers a transcription + // of the audio accumulated since the last commit. + commitIntervalRef.current = setInterval(() => { + if (channelToFlush.readyState === "open") { + commitAudioBuffer(channelToFlush); + } + }, 2000); + } captureToFlush.close(); audioCaptureRef.current = null; }); diff --git a/desktop/src/features/dictation/lib/realtimeAudio.test.mjs b/desktop/src/features/dictation/lib/realtimeAudio.test.mjs index adf0ca663..0121ef436 100644 --- a/desktop/src/features/dictation/lib/realtimeAudio.test.mjs +++ b/desktop/src/features/dictation/lib/realtimeAudio.test.mjs @@ -306,3 +306,29 @@ describe("mergeTranscriptEvent", () => { assert.equal(result, "Alpha. Bravo."); }); }); + +// Inline the logic to keep the test self-contained. +function requiresManualCommit(model) { + return model.includes("realtime-whisper"); +} + +describe("requiresManualCommit", () => { + it("returns true for gpt-realtime-whisper", () => { + assert.equal(requiresManualCommit("gpt-realtime-whisper"), true); + }); + + it("returns true for versioned realtime-whisper model", () => { + assert.equal( + requiresManualCommit("gpt-4o-realtime-whisper-20250512"), + true, + ); + }); + + it("returns false for whisper-1", () => { + assert.equal(requiresManualCommit("whisper-1"), false); + }); + + it("returns false for gpt-4o-transcribe", () => { + assert.equal(requiresManualCommit("gpt-4o-transcribe"), false); + }); +}); diff --git a/desktop/src/features/dictation/lib/realtimeAudio.ts b/desktop/src/features/dictation/lib/realtimeAudio.ts index afac2ad33..4c46fec7b 100644 --- a/desktop/src/features/dictation/lib/realtimeAudio.ts +++ b/desktop/src/features/dictation/lib/realtimeAudio.ts @@ -235,3 +235,22 @@ export function flushAudioBuffer( } chunks.length = 0; } + +/** + * Send `input_audio_buffer.commit` to finalize buffered audio for transcription. + * + * Required when server VAD is disabled (e.g. `realtime-whisper` models) — + * without a commit, appended audio is never processed. For models using + * server VAD, the server commits automatically on speech boundaries. + */ +export function commitAudioBuffer(dataChannel: RTCDataChannel): void { + dataChannel.send(JSON.stringify({ type: "input_audio_buffer.commit" })); +} + +/** + * Whether a transcription model requires manual audio commit (no server VAD). + * Models containing "realtime-whisper" use manual commit per OpenAI guidance. + */ +export function requiresManualCommit(model: string): boolean { + return model.includes("realtime-whisper"); +}