fix(dictation): commit buffered audio for manual-VAD models

When BUZZ_TRANSCRIPTION_MODEL is a realtime-whisper variant, the relay
now omits server_vad from the session config. Without server VAD, OpenAI
buffers audio indefinitely until a manual input_audio_buffer.commit is
sent. This commit adds the client-side counterpart:

1. Export commitAudioBuffer() and requiresManualCommit() from
   realtimeAudio.ts.
2. On session creation, track whether the model requires manual commit.
3. After flushing the pre-connection buffer, commit immediately and start
   a 2s periodic commit interval so streaming transcripts flow during
   recording.
4. On stop, send a final commit before teardown and keep the data channel
   open briefly (3s) to receive the last transcript response.
5. Clear the commit interval on cleanup.

Without this, realtime-whisper sessions would stream/buffer audio but
never produce transcripts because no commit was ever sent.
This commit is contained in:
klopez4212
2026-07-11 16:18:37 +01:00
parent 6be052e0c2
commit 625037e35d
3 changed files with 104 additions and 0 deletions
@@ -11,6 +11,7 @@ import {
BUFFER_COMMITTED_EVENT,
TRANSCRIPT_COMPLETED_EVENT,
TRANSCRIPT_DELTA_EVENT,
commitAudioBuffer,
connectPeerConnection,
createAudioBufferCapture,
createPeerConnection,
@@ -18,6 +19,7 @@ import {
flushAudioBuffer,
getTranscriptText,
mergeTranscriptEvent,
requiresManualCommit,
} from "../lib/realtimeAudio";
interface UseRealtimeDictationOptions {
@@ -58,6 +60,8 @@ export function useRealtimeDictation({
createTranscriptSegmentState(),
);
const activeRunIdRef = useRef(0);
const manualCommitRef = useRef(false);
const commitIntervalRef = useRef<ReturnType<typeof setInterval> | null>(null);
const onRecordingStartRef = useRef(onRecordingStart);
const onTranscriptTextRef = useRef(onTranscriptText);
@@ -82,6 +86,45 @@ export function useRealtimeDictation({
}, []);
const cleanupResources = useCallback(() => {
// Clear periodic commit interval if active.
if (commitIntervalRef.current) {
clearInterval(commitIntervalRef.current);
commitIntervalRef.current = null;
}
// For manual-commit models (no server VAD), commit any buffered audio
// before tearing down the connection so OpenAI processes the final chunk.
// We keep the data channel open briefly to receive the transcript response.
const dc = dataChannelRef.current;
const needsCommit =
manualCommitRef.current && dc && dc.readyState === "open";
manualCommitRef.current = false;
if (needsCommit && dc) {
commitAudioBuffer(dc);
// Stop the mic immediately so no new audio is sent after commit.
for (const track of streamRef.current?.getTracks() ?? []) {
track.stop();
}
streamRef.current = null;
audioCaptureRef.current?.close();
audioCaptureRef.current = null;
// Delay full teardown to allow the final transcript to arrive.
const pc = peerConnectionRef.current;
peerConnectionRef.current = null;
dataChannelRef.current = null;
const runId = activeRunIdRef.current;
setTimeout(() => {
// Only tear down if no new run started in the meantime.
if (activeRunIdRef.current === runId) {
activeRunIdRef.current += 1;
}
dc.close();
pc?.close();
}, 3000);
return;
}
activeRunIdRef.current += 1;
closeResources({
audioCapture: audioCaptureRef.current,
@@ -181,6 +224,7 @@ export function useRealtimeDictation({
closeResources({ audioCapture, stream });
return;
}
manualCommitRef.current = requiresManualCommit(session.model);
// 4. Set up WebRTC
peerConnection = createPeerConnection();
@@ -203,6 +247,7 @@ export function useRealtimeDictation({
// Flush buffered audio once data channel opens
const channelToFlush = dataChannel;
const captureToFlush = audioCapture;
const useManualCommit = manualCommitRef.current;
dataChannel.addEventListener("open", () => {
// If the user stopped (or restarted) recording between the SDP
// exchange and the channel opening, drop this run's buffered audio.
@@ -211,6 +256,20 @@ export function useRealtimeDictation({
return;
}
flushAudioBuffer(channelToFlush, captureToFlush.chunks);
// For manual-commit models, commit the initial buffered audio and
// start a periodic commit interval so streaming transcripts flow
// during recording (server VAD models commit automatically).
if (useManualCommit) {
commitAudioBuffer(channelToFlush);
// Commit every 2s to produce streaming transcript segments while
// the user is still speaking. Each commit triggers a transcription
// of the audio accumulated since the last commit.
commitIntervalRef.current = setInterval(() => {
if (channelToFlush.readyState === "open") {
commitAudioBuffer(channelToFlush);
}
}, 2000);
}
captureToFlush.close();
audioCaptureRef.current = null;
});
@@ -306,3 +306,29 @@ describe("mergeTranscriptEvent", () => {
assert.equal(result, "Alpha. Bravo.");
});
});
// Inline the logic to keep the test self-contained.
function requiresManualCommit(model) {
return model.includes("realtime-whisper");
}
describe("requiresManualCommit", () => {
it("returns true for gpt-realtime-whisper", () => {
assert.equal(requiresManualCommit("gpt-realtime-whisper"), true);
});
it("returns true for versioned realtime-whisper model", () => {
assert.equal(
requiresManualCommit("gpt-4o-realtime-whisper-20250512"),
true,
);
});
it("returns false for whisper-1", () => {
assert.equal(requiresManualCommit("whisper-1"), false);
});
it("returns false for gpt-4o-transcribe", () => {
assert.equal(requiresManualCommit("gpt-4o-transcribe"), false);
});
});
@@ -235,3 +235,22 @@ export function flushAudioBuffer(
}
chunks.length = 0;
}
/**
* Send `input_audio_buffer.commit` to finalize buffered audio for transcription.
*
* Required when server VAD is disabled (e.g. `realtime-whisper` models) —
* without a commit, appended audio is never processed. For models using
* server VAD, the server commits automatically on speech boundaries.
*/
export function commitAudioBuffer(dataChannel: RTCDataChannel): void {
dataChannel.send(JSON.stringify({ type: "input_audio_buffer.commit" }));
}
/**
* Whether a transcription model requires manual audio commit (no server VAD).
* Models containing "realtime-whisper" use manual commit per OpenAI guidance.
*/
export function requiresManualCommit(model: string): boolean {
return model.includes("realtime-whisper");
}