mirror of
https://github.com/block/buzz.git
synced 2026-08-18 06:50:31 +02:00
fix(dictation): commit buffered audio for manual-VAD models
When BUZZ_TRANSCRIPTION_MODEL is a realtime-whisper variant, the relay now omits server_vad from the session config. Without server VAD, OpenAI buffers audio indefinitely until a manual input_audio_buffer.commit is sent. This commit adds the client-side counterpart: 1. Export commitAudioBuffer() and requiresManualCommit() from realtimeAudio.ts. 2. On session creation, track whether the model requires manual commit. 3. After flushing the pre-connection buffer, commit immediately and start a 2s periodic commit interval so streaming transcripts flow during recording. 4. On stop, send a final commit before teardown and keep the data channel open briefly (3s) to receive the last transcript response. 5. Clear the commit interval on cleanup. Without this, realtime-whisper sessions would stream/buffer audio but never produce transcripts because no commit was ever sent.
This commit is contained in:
@@ -11,6 +11,7 @@ import {
|
||||
BUFFER_COMMITTED_EVENT,
|
||||
TRANSCRIPT_COMPLETED_EVENT,
|
||||
TRANSCRIPT_DELTA_EVENT,
|
||||
commitAudioBuffer,
|
||||
connectPeerConnection,
|
||||
createAudioBufferCapture,
|
||||
createPeerConnection,
|
||||
@@ -18,6 +19,7 @@ import {
|
||||
flushAudioBuffer,
|
||||
getTranscriptText,
|
||||
mergeTranscriptEvent,
|
||||
requiresManualCommit,
|
||||
} from "../lib/realtimeAudio";
|
||||
|
||||
interface UseRealtimeDictationOptions {
|
||||
@@ -58,6 +60,8 @@ export function useRealtimeDictation({
|
||||
createTranscriptSegmentState(),
|
||||
);
|
||||
const activeRunIdRef = useRef(0);
|
||||
const manualCommitRef = useRef(false);
|
||||
const commitIntervalRef = useRef<ReturnType<typeof setInterval> | null>(null);
|
||||
const onRecordingStartRef = useRef(onRecordingStart);
|
||||
const onTranscriptTextRef = useRef(onTranscriptText);
|
||||
|
||||
@@ -82,6 +86,45 @@ export function useRealtimeDictation({
|
||||
}, []);
|
||||
|
||||
const cleanupResources = useCallback(() => {
|
||||
// Clear periodic commit interval if active.
|
||||
if (commitIntervalRef.current) {
|
||||
clearInterval(commitIntervalRef.current);
|
||||
commitIntervalRef.current = null;
|
||||
}
|
||||
|
||||
// For manual-commit models (no server VAD), commit any buffered audio
|
||||
// before tearing down the connection so OpenAI processes the final chunk.
|
||||
// We keep the data channel open briefly to receive the transcript response.
|
||||
const dc = dataChannelRef.current;
|
||||
const needsCommit =
|
||||
manualCommitRef.current && dc && dc.readyState === "open";
|
||||
manualCommitRef.current = false;
|
||||
|
||||
if (needsCommit && dc) {
|
||||
commitAudioBuffer(dc);
|
||||
// Stop the mic immediately so no new audio is sent after commit.
|
||||
for (const track of streamRef.current?.getTracks() ?? []) {
|
||||
track.stop();
|
||||
}
|
||||
streamRef.current = null;
|
||||
audioCaptureRef.current?.close();
|
||||
audioCaptureRef.current = null;
|
||||
// Delay full teardown to allow the final transcript to arrive.
|
||||
const pc = peerConnectionRef.current;
|
||||
peerConnectionRef.current = null;
|
||||
dataChannelRef.current = null;
|
||||
const runId = activeRunIdRef.current;
|
||||
setTimeout(() => {
|
||||
// Only tear down if no new run started in the meantime.
|
||||
if (activeRunIdRef.current === runId) {
|
||||
activeRunIdRef.current += 1;
|
||||
}
|
||||
dc.close();
|
||||
pc?.close();
|
||||
}, 3000);
|
||||
return;
|
||||
}
|
||||
|
||||
activeRunIdRef.current += 1;
|
||||
closeResources({
|
||||
audioCapture: audioCaptureRef.current,
|
||||
@@ -181,6 +224,7 @@ export function useRealtimeDictation({
|
||||
closeResources({ audioCapture, stream });
|
||||
return;
|
||||
}
|
||||
manualCommitRef.current = requiresManualCommit(session.model);
|
||||
|
||||
// 4. Set up WebRTC
|
||||
peerConnection = createPeerConnection();
|
||||
@@ -203,6 +247,7 @@ export function useRealtimeDictation({
|
||||
// Flush buffered audio once data channel opens
|
||||
const channelToFlush = dataChannel;
|
||||
const captureToFlush = audioCapture;
|
||||
const useManualCommit = manualCommitRef.current;
|
||||
dataChannel.addEventListener("open", () => {
|
||||
// If the user stopped (or restarted) recording between the SDP
|
||||
// exchange and the channel opening, drop this run's buffered audio.
|
||||
@@ -211,6 +256,20 @@ export function useRealtimeDictation({
|
||||
return;
|
||||
}
|
||||
flushAudioBuffer(channelToFlush, captureToFlush.chunks);
|
||||
// For manual-commit models, commit the initial buffered audio and
|
||||
// start a periodic commit interval so streaming transcripts flow
|
||||
// during recording (server VAD models commit automatically).
|
||||
if (useManualCommit) {
|
||||
commitAudioBuffer(channelToFlush);
|
||||
// Commit every 2s to produce streaming transcript segments while
|
||||
// the user is still speaking. Each commit triggers a transcription
|
||||
// of the audio accumulated since the last commit.
|
||||
commitIntervalRef.current = setInterval(() => {
|
||||
if (channelToFlush.readyState === "open") {
|
||||
commitAudioBuffer(channelToFlush);
|
||||
}
|
||||
}, 2000);
|
||||
}
|
||||
captureToFlush.close();
|
||||
audioCaptureRef.current = null;
|
||||
});
|
||||
|
||||
@@ -306,3 +306,29 @@ describe("mergeTranscriptEvent", () => {
|
||||
assert.equal(result, "Alpha. Bravo.");
|
||||
});
|
||||
});
|
||||
|
||||
// Inline the logic to keep the test self-contained.
|
||||
function requiresManualCommit(model) {
|
||||
return model.includes("realtime-whisper");
|
||||
}
|
||||
|
||||
describe("requiresManualCommit", () => {
|
||||
it("returns true for gpt-realtime-whisper", () => {
|
||||
assert.equal(requiresManualCommit("gpt-realtime-whisper"), true);
|
||||
});
|
||||
|
||||
it("returns true for versioned realtime-whisper model", () => {
|
||||
assert.equal(
|
||||
requiresManualCommit("gpt-4o-realtime-whisper-20250512"),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
it("returns false for whisper-1", () => {
|
||||
assert.equal(requiresManualCommit("whisper-1"), false);
|
||||
});
|
||||
|
||||
it("returns false for gpt-4o-transcribe", () => {
|
||||
assert.equal(requiresManualCommit("gpt-4o-transcribe"), false);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -235,3 +235,22 @@ export function flushAudioBuffer(
|
||||
}
|
||||
chunks.length = 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Send `input_audio_buffer.commit` to finalize buffered audio for transcription.
|
||||
*
|
||||
* Required when server VAD is disabled (e.g. `realtime-whisper` models) —
|
||||
* without a commit, appended audio is never processed. For models using
|
||||
* server VAD, the server commits automatically on speech boundaries.
|
||||
*/
|
||||
export function commitAudioBuffer(dataChannel: RTCDataChannel): void {
|
||||
dataChannel.send(JSON.stringify({ type: "input_audio_buffer.commit" }));
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether a transcription model requires manual audio commit (no server VAD).
|
||||
* Models containing "realtime-whisper" use manual commit per OpenAI guidance.
|
||||
*/
|
||||
export function requiresManualCommit(model: string): boolean {
|
||||
return model.includes("realtime-whisper");
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user