mirror of
https://github.com/block/buzz.git
synced 2026-08-18 06:50:31 +02:00
fix(dictation): correct expires_after schema, grace window for all models
Two fixes: 1. (transcribe.rs) expires_after must be an object with anchor + seconds fields per OpenAI's client-secret schema, not a bare integer. The bare number would cause OpenAI to reject the request with a 400, making dictation fail with a 502 from the relay. 2. (useRealtimeDictation.ts) The user-stop grace window (3s delayed teardown with run kept valid) now applies to ALL models, not just manual-commit ones. For server-VAD models, the final VAD completion event may still be in-flight when the user clicks the mic to stop — without the grace window, that event was dropped and the tail of the dictated message disappeared.
This commit is contained in:
@@ -209,7 +209,10 @@ fn build_session_payload(model: &str) -> Value {
|
||||
"input": audio_input
|
||||
}
|
||||
},
|
||||
"expires_after": 60
|
||||
"expires_after": {
|
||||
"anchor": "created_at",
|
||||
"seconds": 60
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
@@ -433,9 +436,11 @@ mod tests {
|
||||
fn build_session_payload_sets_short_expires_after() {
|
||||
use super::build_session_payload;
|
||||
let payload = build_session_payload("whisper-1");
|
||||
assert_eq!(payload["expires_after"], 60);
|
||||
assert_eq!(payload["expires_after"]["anchor"], "created_at");
|
||||
assert_eq!(payload["expires_after"]["seconds"], 60);
|
||||
let payload2 = build_session_payload("gpt-realtime-whisper");
|
||||
assert_eq!(payload2["expires_after"], 60);
|
||||
assert_eq!(payload2["expires_after"]["anchor"], "created_at");
|
||||
assert_eq!(payload2["expires_after"]["seconds"], 60);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -108,31 +108,33 @@ export function useRealtimeDictation({
|
||||
manualCommitRef.current && dc && dc.readyState === "open";
|
||||
manualCommitRef.current = false;
|
||||
|
||||
// Send final commit for manual-commit models.
|
||||
if (needsCommit && dc) {
|
||||
commitAudioBuffer(dc);
|
||||
// Stop the mic immediately so no new audio is sent after commit.
|
||||
}
|
||||
|
||||
// Determine whether we need a grace window before full teardown.
|
||||
// User-initiated stop (!invalidateRun) needs a grace period for BOTH:
|
||||
// - Manual-commit models: final commit's transcript response
|
||||
// - Server-VAD models: in-flight VAD completion events
|
||||
const needsGrace = !invalidateRun && dc && dc.readyState === "open";
|
||||
|
||||
if (needsGrace && dc) {
|
||||
// Stop the mic immediately so no new audio is sent.
|
||||
for (const track of streamRef.current?.getTracks() ?? []) {
|
||||
track.stop();
|
||||
}
|
||||
streamRef.current = null;
|
||||
audioCaptureRef.current?.close();
|
||||
audioCaptureRef.current = null;
|
||||
// Delay WebRTC teardown briefly so the commit reaches OpenAI and
|
||||
// the transcript response can arrive.
|
||||
// Delay WebRTC teardown briefly so final transcript events arrive.
|
||||
const pc = peerConnectionRef.current;
|
||||
peerConnectionRef.current = null;
|
||||
dataChannelRef.current = null;
|
||||
const runId = activeRunIdRef.current;
|
||||
|
||||
if (invalidateRun) {
|
||||
// Send/navigation: reject all late events immediately.
|
||||
activeRunIdRef.current += 1;
|
||||
}
|
||||
// else: user stop — keep the run valid so the final transcript arrives.
|
||||
|
||||
setTimeout(() => {
|
||||
// After the grace window, invalidate if we haven't already (user stop)
|
||||
// or if no new run started (send/navigation case already bumped).
|
||||
// After the grace window, invalidate the run and tear down.
|
||||
if (activeRunIdRef.current === runId) {
|
||||
activeRunIdRef.current += 1;
|
||||
}
|
||||
@@ -142,6 +144,7 @@ export function useRealtimeDictation({
|
||||
return;
|
||||
}
|
||||
|
||||
// Immediate teardown: send/navigation cancellation, or no open channel.
|
||||
activeRunIdRef.current += 1;
|
||||
closeResources({
|
||||
audioCapture: audioCaptureRef.current,
|
||||
|
||||
Reference in New Issue
Block a user