fix(dictation): correct expires_after schema, grace window for all models

Two fixes:

1. (transcribe.rs) expires_after must be an object with anchor + seconds
   fields per OpenAI's client-secret schema, not a bare integer. The bare
   number would cause OpenAI to reject the request with a 400, making
   dictation fail with a 502 from the relay.

2. (useRealtimeDictation.ts) The user-stop grace window (3s delayed
   teardown with run kept valid) now applies to ALL models, not just
   manual-commit ones. For server-VAD models, the final VAD completion
   event may still be in-flight when the user clicks the mic to stop —
   without the grace window, that event was dropped and the tail of the
   dictated message disappeared.
This commit is contained in:
klopez4212
2026-07-11 16:18:38 +01:00
parent bf8ca927ee
commit 25fd7ca78d
2 changed files with 22 additions and 14 deletions
+8 -3
View File
@@ -209,7 +209,10 @@ fn build_session_payload(model: &str) -> Value {
"input": audio_input
}
},
"expires_after": 60
"expires_after": {
"anchor": "created_at",
"seconds": 60
}
})
}
@@ -433,9 +436,11 @@ mod tests {
fn build_session_payload_sets_short_expires_after() {
use super::build_session_payload;
let payload = build_session_payload("whisper-1");
assert_eq!(payload["expires_after"], 60);
assert_eq!(payload["expires_after"]["anchor"], "created_at");
assert_eq!(payload["expires_after"]["seconds"], 60);
let payload2 = build_session_payload("gpt-realtime-whisper");
assert_eq!(payload2["expires_after"], 60);
assert_eq!(payload2["expires_after"]["anchor"], "created_at");
assert_eq!(payload2["expires_after"]["seconds"], 60);
}
#[test]
@@ -108,31 +108,33 @@ export function useRealtimeDictation({
manualCommitRef.current && dc && dc.readyState === "open";
manualCommitRef.current = false;
// Send final commit for manual-commit models.
if (needsCommit && dc) {
commitAudioBuffer(dc);
// Stop the mic immediately so no new audio is sent after commit.
}
// Determine whether we need a grace window before full teardown.
// User-initiated stop (!invalidateRun) needs a grace period for BOTH:
// - Manual-commit models: final commit's transcript response
// - Server-VAD models: in-flight VAD completion events
const needsGrace = !invalidateRun && dc && dc.readyState === "open";
if (needsGrace && dc) {
// Stop the mic immediately so no new audio is sent.
for (const track of streamRef.current?.getTracks() ?? []) {
track.stop();
}
streamRef.current = null;
audioCaptureRef.current?.close();
audioCaptureRef.current = null;
// Delay WebRTC teardown briefly so the commit reaches OpenAI and
// the transcript response can arrive.
// Delay WebRTC teardown briefly so final transcript events arrive.
const pc = peerConnectionRef.current;
peerConnectionRef.current = null;
dataChannelRef.current = null;
const runId = activeRunIdRef.current;
if (invalidateRun) {
// Send/navigation: reject all late events immediately.
activeRunIdRef.current += 1;
}
// else: user stop — keep the run valid so the final transcript arrives.
setTimeout(() => {
// After the grace window, invalidate if we haven't already (user stop)
// or if no new run started (send/navigation case already bumped).
// After the grace window, invalidate the run and tear down.
if (activeRunIdRef.current === runId) {
activeRunIdRef.current += 1;
}
@@ -142,6 +144,7 @@ export function useRealtimeDictation({
return;
}
// Immediate teardown: send/navigation cancellation, or no open channel.
activeRunIdRef.current += 1;
closeResources({
audioCapture: audioCaptureRef.current,