refactor(dictation): drop OpenAI Realtime path, add ⌘D shortcut

- Remove useRealtimeDictation, realtimeAudio, realtimeBufferWorklet,
  and transcribeSession API — dictation is now local-only (Parakeet).
- useDictation simplified to use only useLocalDictation.
- Add ⌘D (Ctrl+D on Windows) global shortcut to toggle dictation.
- DictationButton tooltip shows 'Voice Dictation ⌘D' with kbd styling
  matching the search bar's ⌘K pattern.
- Shortcut registered in keyboard-shortcuts.ts settings list.
- AppShell dispatches custom event; useComposerDictation listens.
This commit is contained in:
klopez4212
2026-07-11 16:18:40 +01:00
parent d3fd07b10d
commit ba958389e7
10 changed files with 45 additions and 1106 deletions
+6
View File
@@ -595,6 +595,12 @@ export function AppShell() {
return;
}
if (key === "d" && !event.shiftKey) {
event.preventDefault();
window.dispatchEvent(new CustomEvent("buzz:toggle-dictation"));
return;
}
if (key === "a" && event.shiftKey) {
event.preventDefault();
void goHome();
@@ -1,76 +0,0 @@
import { getRelayHttpUrl, signRelayEvent } from "@/shared/api/tauri";
export interface TranscribeStatus {
configured: boolean;
model: string;
}
export interface TranscribeConnectResponse {
sdp: string;
model: string;
}
/** NIP-98 event kind for HTTP request authorization. */
const NIP98_KIND = 27235;
/**
* Build a NIP-98 `Authorization: Nostr <base64>` header for an HTTP request.
*
* The relay verifies the signed event's `u` tag against its own
* host-derived expected URL, so `url` must be the exact absolute URL being
* fetched (scheme + host + path). The `method` tag must match the request.
*/
async function nip98AuthHeader(url: string, method: string): Promise<string> {
const nonce = crypto.randomUUID();
const event = await signRelayEvent({
kind: NIP98_KIND,
content: "",
tags: [
["u", url],
["method", method],
["nonce", nonce],
],
});
const json = JSON.stringify(event);
// btoa needs a binary string; encode UTF-8 first so non-ASCII survives.
const base64 = btoa(String.fromCharCode(...new TextEncoder().encode(json)));
return `Nostr ${base64}`;
}
export async function getTranscribeStatus(): Promise<TranscribeStatus> {
const baseUrl = await getRelayHttpUrl();
const url = `${baseUrl}/transcribe/status`;
const response = await fetch(url, {
headers: { Authorization: await nip98AuthHeader(url, "GET") },
});
if (!response.ok) {
throw new Error(`Transcribe status check failed: ${response.status}`);
}
return response.json();
}
/**
* Mint an OpenAI Realtime session and complete the WebRTC SDP exchange in a
* single relay round-trip. The relay holds the OpenAI bearer token server-side
* — the client never sees it. This also works correctly across multiple relay
* replicas since no server-side session state is needed between requests.
*/
export async function transcribeConnect(
sdp: string,
): Promise<TranscribeConnectResponse> {
const baseUrl = await getRelayHttpUrl();
const url = `${baseUrl}/transcribe/connect`;
const response = await fetch(url, {
method: "POST",
headers: {
"Content-Type": "application/json",
Authorization: await nip98AuthHeader(url, "POST"),
},
body: JSON.stringify({ sdp }),
});
if (!response.ok) {
const body = await response.text().catch(() => "");
throw new Error(`Transcribe connect failed (${response.status}): ${body}`);
}
return response.json();
}
@@ -64,12 +64,23 @@ export function useComposerDictation({
// Auto-cancel dictation when the composer becomes disabled mid-recording
// (e.g. channel becomes read-only, parent send state disables thread composer).
// Without this, the WebRTC session keeps running with no way to stop it.
// Without this, the STT session keeps running with no way to stop it.
useEffect(() => {
if (disabled && dictation.isRecording) {
dictation.cancelRecording();
}
}, [disabled, dictation.isRecording, dictation.cancelRecording]);
// ⌘D global shortcut — dispatched from AppShell's keydown handler.
useEffect(() => {
function handleToggle() {
dictation.toggleRecording();
}
window.addEventListener("buzz:toggle-dictation", handleToggle);
return () => {
window.removeEventListener("buzz:toggle-dictation", handleToggle);
};
}, [dictation.toggleRecording]);
return dictation;
}
@@ -7,7 +7,6 @@ import {
replaceTrailingTranscribedText,
} from "../lib/voiceInput";
import { useLocalDictation } from "./useLocalDictation";
import { useRealtimeDictation } from "./useRealtimeDictation";
interface UseDictationOptions {
/** Returns the current composer text (must be fresh — synced from editor). */
@@ -71,25 +70,13 @@ export function useDictation({
[autoSubmitPhrases, getText, onSend, isSendBlockedRef, setText],
);
// Try local STT first (offline, no API key needed).
const localDictation = useLocalDictation({
const dictation = useLocalDictation({
onRecordingStart: () => {
lastTranscriptRef.current = "";
},
onTranscriptText: handleTranscript,
});
// Fall back to cloud (OpenAI Realtime) if local is unavailable.
const cloudDictation = useRealtimeDictation({
disabled: localDictation.isEnabled, // Disable cloud when local is available
onRecordingStart: () => {
lastTranscriptRef.current = "";
},
onTranscriptText: handleTranscript,
});
// Use whichever is enabled — local takes priority.
const dictation = localDictation.isEnabled ? localDictation : cloudDictation;
stopRecordingRef.current = dictation.stopRecording;
return dictation;
@@ -1,363 +0,0 @@
import { useCallback, useEffect, useRef, useState } from "react";
import { toast } from "sonner";
import { getTranscribeStatus } from "../api/transcribeSession";
import {
type AudioBufferCapture,
type TranscriptEvent,
type TranscriptSegmentState,
BUFFER_COMMITTED_EVENT,
TRANSCRIPT_COMPLETED_EVENT,
TRANSCRIPT_DELTA_EVENT,
TRANSCRIPT_FAILED_EVENT,
commitAudioBuffer,
connectPeerConnection,
createAudioBufferCapture,
createPeerConnection,
createTranscriptSegmentState,
flushAudioBuffer,
getTranscriptText,
mergeTranscriptEvent,
requiresManualCommit,
} from "../lib/realtimeAudio";
interface UseRealtimeDictationOptions {
disabled?: boolean;
onRecordingStart?: () => void;
onTranscriptText: (text: string) => void;
}
function closeResources(resources: {
audioCapture?: AudioBufferCapture | null;
dataChannel?: RTCDataChannel | null;
peerConnection?: RTCPeerConnection | null;
stream?: MediaStream | null;
}) {
resources.audioCapture?.close();
resources.dataChannel?.close();
resources.peerConnection?.close();
for (const track of resources.stream?.getTracks() ?? []) {
track.stop();
}
}
export function useRealtimeDictation({
disabled = false,
onRecordingStart,
onTranscriptText,
}: UseRealtimeDictationOptions) {
const [isRecording, setIsRecording] = useState(false);
const [isStarting, setIsStarting] = useState(false);
const [isTranscribing, setIsTranscribing] = useState(false);
const [isConfigured, setIsConfigured] = useState(false);
const peerConnectionRef = useRef<RTCPeerConnection | null>(null);
const dataChannelRef = useRef<RTCDataChannel | null>(null);
const streamRef = useRef<MediaStream | null>(null);
const audioCaptureRef = useRef<AudioBufferCapture | null>(null);
const segmentStateRef = useRef<TranscriptSegmentState>(
createTranscriptSegmentState(),
);
const activeRunIdRef = useRef(0);
const manualCommitRef = useRef(false);
const commitIntervalRef = useRef<ReturnType<typeof setInterval> | null>(null);
const onRecordingStartRef = useRef(onRecordingStart);
const onTranscriptTextRef = useRef(onTranscriptText);
onRecordingStartRef.current = onRecordingStart;
onTranscriptTextRef.current = onTranscriptText;
const isEnabled = !disabled && isConfigured;
// Check if transcription is configured on mount
useEffect(() => {
let cancelled = false;
getTranscribeStatus()
.then((status) => {
if (!cancelled) setIsConfigured(status.configured);
})
.catch(() => {
if (!cancelled) setIsConfigured(false);
});
return () => {
cancelled = true;
};
}, []);
/**
* Tear down recording resources.
*
* @param invalidateRun - When true, immediately marks the run stale so
* late transcript events are rejected (used by send/edit-save/navigation).
* When false (user-initiated stop), the run stays valid during the grace
* window so the final commit's transcript is delivered to the composer.
*/
const cleanupResources = useCallback((invalidateRun = true) => {
// Clear periodic commit interval if active.
if (commitIntervalRef.current) {
clearInterval(commitIntervalRef.current);
commitIntervalRef.current = null;
}
// For manual-commit models (no server VAD), commit any buffered audio
// before tearing down the connection so OpenAI processes the final chunk.
// We keep the data channel open briefly to receive the transcript response.
const dc = dataChannelRef.current;
const needsCommit =
manualCommitRef.current && dc && dc.readyState === "open";
manualCommitRef.current = false;
// Send final commit for manual-commit models.
if (needsCommit && dc) {
commitAudioBuffer(dc);
}
// Determine whether we need a grace window before full teardown.
// User-initiated stop (!invalidateRun) needs a grace period for BOTH:
// - Manual-commit models: final commit's transcript response
// - Server-VAD models: in-flight VAD completion events
const needsGrace = !invalidateRun && dc && dc.readyState === "open";
if (needsGrace && dc) {
// Stop the mic immediately so no new audio is sent.
for (const track of streamRef.current?.getTracks() ?? []) {
track.stop();
}
streamRef.current = null;
audioCaptureRef.current?.close();
audioCaptureRef.current = null;
// Delay WebRTC teardown briefly so final transcript events arrive.
const pc = peerConnectionRef.current;
peerConnectionRef.current = null;
dataChannelRef.current = null;
const runId = activeRunIdRef.current;
setTimeout(() => {
// After the grace window, invalidate the run and tear down.
if (activeRunIdRef.current === runId) {
activeRunIdRef.current += 1;
}
dc.close();
pc?.close();
}, 3000);
return;
}
// Immediate teardown: send/navigation cancellation, or no open channel.
activeRunIdRef.current += 1;
closeResources({
audioCapture: audioCaptureRef.current,
dataChannel: dataChannelRef.current,
peerConnection: peerConnectionRef.current,
stream: streamRef.current,
});
audioCaptureRef.current = null;
dataChannelRef.current = null;
peerConnectionRef.current = null;
streamRef.current = null;
}, []);
/** Full cleanup — invalidates the run (used by send/navigation). */
const cleanup = useCallback(() => {
cleanupResources(true);
setIsRecording(false);
setIsStarting(false);
setIsTranscribing(false);
}, [cleanupResources]);
/** User-initiated stop — preserves final transcript for manual-commit models. */
const userStop = useCallback(() => {
cleanupResources(false);
setIsRecording(false);
setIsStarting(false);
// Note: isTranscribing stays true briefly while the final transcript arrives.
}, [cleanupResources]);
useEffect(() => cleanupResources, [cleanupResources]);
const handleRealtimeEvent = useCallback(
(runId: number, event: TranscriptEvent) => {
// Ignore events from a stale run (e.g. user sent/stopped while
// transcripts were still in-flight from the data channel).
if (activeRunIdRef.current !== runId) return;
if (event.type === "error") {
console.error("OpenAI realtime server error", event);
toast.error(event.error?.message ?? "Voice input error");
return;
}
if (event.type === TRANSCRIPT_FAILED_EVENT) {
console.warn("OpenAI transcription failed for item", event);
setIsTranscribing(false);
toast.error(event.error?.message ?? "Transcription failed");
return;
}
if (
event.type !== TRANSCRIPT_DELTA_EVENT &&
event.type !== TRANSCRIPT_COMPLETED_EVENT &&
event.type !== BUFFER_COMMITTED_EVENT
) {
return;
}
const prevText = getTranscriptText(segmentStateRef.current);
const merged = mergeTranscriptEvent(segmentStateRef.current, event);
// Always update transcribing state — a completed event must clear the
// indicator even when the final text matches the accumulated deltas.
const stillTranscribing = event.type !== TRANSCRIPT_COMPLETED_EVENT;
setIsTranscribing(stillTranscribing);
if (merged === prevText) return;
onTranscriptTextRef.current(merged);
},
[],
);
const startRecording = useCallback(async () => {
if (!isEnabled || isStarting || isRecording) return;
const runId = activeRunIdRef.current + 1;
activeRunIdRef.current = runId;
const isStaleRun = () => activeRunIdRef.current !== runId;
let stream: MediaStream | null = null;
let audioCapture: AudioBufferCapture | null = null;
let peerConnection: RTCPeerConnection | null = null;
let dataChannel: RTCDataChannel | null = null;
setIsStarting(true);
segmentStateRef.current = createTranscriptSegmentState();
onRecordingStartRef.current?.();
try {
// 1. Capture mic immediately for instant feedback
stream = await navigator.mediaDevices.getUserMedia({
audio: {
autoGainControl: true,
echoCancellation: true,
noiseSuppression: true,
},
});
if (isStaleRun()) {
closeResources({ stream });
return;
}
streamRef.current = stream;
setIsRecording(true);
// 2. Buffer PCM via AudioWorklet while network calls proceed
audioCapture = await createAudioBufferCapture(stream);
if (isStaleRun()) {
closeResources({ audioCapture, stream });
return;
}
audioCaptureRef.current = audioCapture;
// 3. Set up WebRTC peer connection
peerConnection = createPeerConnection();
peerConnectionRef.current = peerConnection;
const activeStream = stream;
stream.getAudioTracks().forEach((track) => {
peerConnection?.addTrack(track, activeStream);
});
dataChannel = peerConnection.createDataChannel("oai-events");
dataChannelRef.current = dataChannel;
dataChannel.addEventListener("message", (message) => {
try {
handleRealtimeEvent(runId, JSON.parse(String(message.data)));
} catch {
// Ignore non-JSON events
}
});
// Flush buffered audio once data channel opens
const channelToFlush = dataChannel;
const captureToFlush = audioCapture;
dataChannel.addEventListener("open", () => {
if (isStaleRun()) {
captureToFlush.close();
return;
}
flushAudioBuffer(channelToFlush, captureToFlush.chunks);
if (manualCommitRef.current) {
commitAudioBuffer(channelToFlush);
commitIntervalRef.current = setInterval(() => {
if (channelToFlush.readyState === "open") {
commitAudioBuffer(channelToFlush);
}
}, 2000);
}
captureToFlush.close();
audioCaptureRef.current = null;
});
// 4. Session creation + SDP exchange in a single relay round-trip.
// No cross-replica state needed — works in HA deployments.
const { model } = await connectPeerConnection({ peerConnection });
if (isStaleRun()) {
closeResources({ audioCapture, dataChannel, peerConnection, stream });
return;
}
manualCommitRef.current = requiresManualCommit(model);
} catch (error) {
closeResources({ audioCapture, dataChannel, peerConnection, stream });
if (!isStaleRun()) {
audioCaptureRef.current = null;
dataChannelRef.current = null;
peerConnectionRef.current = null;
streamRef.current = null;
setIsRecording(false);
setIsTranscribing(false);
const message =
error instanceof Error ? error.message : "Voice input failed";
if (/not allowed|denied|permission/i.test(message)) {
toast.error("Microphone access denied", {
description:
"Allow microphone access in System Settings to use dictation.",
});
} else if (/not found|no audio/i.test(message)) {
toast.error("No microphone found", {
description: "Connect a microphone and try again.",
});
} else {
toast.error("Voice input failed", { description: message });
}
}
} finally {
if (!isStaleRun()) setIsStarting(false);
}
}, [handleRealtimeEvent, isEnabled, isRecording, isStarting]);
/** User-initiated stop (mic button) — preserves final transcript. */
const stopRecording = useCallback(() => userStop(), [userStop]);
/**
* Cancel recording and reject all pending transcripts. Used by send,
* edit-save, and navigation to prevent late events from refilling the
* composer after the content has been dispatched.
*/
const cancelRecording = useCallback(() => cleanup(), [cleanup]);
const toggleRecording = useCallback(() => {
if (isRecording || isStarting) {
stopRecording();
return;
}
void startRecording();
}, [isRecording, isStarting, startRecording, stopRecording]);
return {
isEnabled,
isRecording,
isStarting,
isTranscribing,
startRecording,
stopRecording,
cancelRecording,
toggleRecording,
};
}
@@ -1,334 +0,0 @@
import { describe, it } from "node:test";
import assert from "node:assert/strict";
// Inline the logic to keep the test self-contained and avoid bundler issues.
const TRANSCRIPT_DELTA_EVENT =
"conversation.item.input_audio_transcription.delta";
const TRANSCRIPT_COMPLETED_EVENT =
"conversation.item.input_audio_transcription.completed";
const BUFFER_COMMITTED_EVENT = "input_audio_buffer.committed";
function createTranscriptSegmentState() {
return { itemOrder: [], items: new Map() };
}
function getOrCreateItem(state, itemId, previousItemId) {
let seg = state.items.get(itemId);
if (!seg) {
seg = { pending: "", finalized: null };
state.items.set(itemId, seg);
if (previousItemId) {
const prevIndex = state.itemOrder.indexOf(previousItemId);
if (prevIndex !== -1) {
state.itemOrder.splice(prevIndex + 1, 0, itemId);
} else {
state.itemOrder.push(itemId);
}
} else {
state.itemOrder.push(itemId);
}
}
return seg;
}
function getTranscriptText(state) {
let result = "";
for (const id of state.itemOrder) {
const seg = state.items.get(id);
if (!seg) continue;
const text = seg.finalized ?? seg.pending;
if (!text) continue;
if (result && !result.endsWith(" ") && !text.startsWith(" ")) {
result += " ";
}
result += text;
}
return result;
}
function mergeTranscriptEvent(state, event) {
const itemId = event.item_id ?? "__default__";
if (event.type === BUFFER_COMMITTED_EVENT) {
getOrCreateItem(state, itemId, event.previous_item_id ?? undefined);
} else if (event.type === TRANSCRIPT_DELTA_EVENT) {
const seg = getOrCreateItem(state, itemId);
const delta = event.delta ?? "";
if (delta) {
seg.pending += delta;
}
} else if (event.type === TRANSCRIPT_COMPLETED_EVENT) {
const seg = getOrCreateItem(state, itemId);
seg.finalized = event.transcript ?? "";
}
return getTranscriptText(state);
}
describe("mergeTranscriptEvent", () => {
it("accumulates delta events for a single item", () => {
const state = createTranscriptSegmentState();
const r1 = mergeTranscriptEvent(state, {
type: TRANSCRIPT_DELTA_EVENT,
item_id: "item_1",
delta: "hello ",
});
assert.equal(r1, "hello ");
const r2 = mergeTranscriptEvent(state, {
type: TRANSCRIPT_DELTA_EVENT,
item_id: "item_1",
delta: "world",
});
assert.equal(r2, "hello world");
});
it("replaces deltas with finalized text on completed event", () => {
const state = createTranscriptSegmentState();
mergeTranscriptEvent(state, {
type: TRANSCRIPT_DELTA_EVENT,
item_id: "item_1",
delta: "hello world",
});
const result = mergeTranscriptEvent(state, {
type: TRANSCRIPT_COMPLETED_EVENT,
item_id: "item_1",
transcript: "Hello, world.",
});
assert.equal(result, "Hello, world.");
});
it("handles multiple items in order", () => {
const state = createTranscriptSegmentState();
// First item
mergeTranscriptEvent(state, {
type: TRANSCRIPT_DELTA_EVENT,
item_id: "item_1",
delta: "first ",
});
mergeTranscriptEvent(state, {
type: TRANSCRIPT_COMPLETED_EVENT,
item_id: "item_1",
transcript: "First.",
});
// Second item
mergeTranscriptEvent(state, {
type: TRANSCRIPT_DELTA_EVENT,
item_id: "item_2",
delta: "second",
});
assert.equal(state.items.get("item_2").pending, "second");
const result = mergeTranscriptEvent(state, {
type: TRANSCRIPT_COMPLETED_EVENT,
item_id: "item_2",
transcript: "Second.",
});
assert.equal(result, "First. Second.");
});
it("handles out-of-order completed events by item id", () => {
const state = createTranscriptSegmentState();
// Both items start with deltas
mergeTranscriptEvent(state, {
type: TRANSCRIPT_DELTA_EVENT,
item_id: "item_1",
delta: "first",
});
mergeTranscriptEvent(state, {
type: TRANSCRIPT_DELTA_EVENT,
item_id: "item_2",
delta: "second",
});
// item_2 completes before item_1
mergeTranscriptEvent(state, {
type: TRANSCRIPT_COMPLETED_EVENT,
item_id: "item_2",
transcript: "Second.",
});
// item_1 still shows pending
let result = mergeTranscriptEvent(state, {
type: TRANSCRIPT_DELTA_EVENT,
item_id: "item_1",
delta: " more",
});
assert.equal(result, "first more Second.");
// item_1 finally completes
result = mergeTranscriptEvent(state, {
type: TRANSCRIPT_COMPLETED_EVENT,
item_id: "item_1",
transcript: "First more.",
});
assert.equal(result, "First more. Second.");
});
it("does not duplicate text on completed event", () => {
const state = createTranscriptSegmentState();
mergeTranscriptEvent(state, {
type: TRANSCRIPT_DELTA_EVENT,
item_id: "item_1",
delta: "hello world",
});
const result = mergeTranscriptEvent(state, {
type: TRANSCRIPT_COMPLETED_EVENT,
item_id: "item_1",
transcript: "Hello, world.",
});
assert.equal(result, "Hello, world.");
});
it("falls back to __default__ when item_id is missing", () => {
const state = createTranscriptSegmentState();
mergeTranscriptEvent(state, {
type: TRANSCRIPT_DELTA_EVENT,
delta: "no id",
});
const result = mergeTranscriptEvent(state, {
type: TRANSCRIPT_COMPLETED_EVENT,
transcript: "No id.",
});
assert.equal(result, "No id.");
});
it("preserves committed item order when completions arrive out of order", () => {
const state = createTranscriptSegmentState();
// Server commits items in order: item_1 then item_2
mergeTranscriptEvent(state, {
type: BUFFER_COMMITTED_EVENT,
item_id: "item_1",
previous_item_id: null,
});
mergeTranscriptEvent(state, {
type: BUFFER_COMMITTED_EVENT,
item_id: "item_2",
previous_item_id: "item_1",
});
// But item_2's completion arrives first
mergeTranscriptEvent(state, {
type: TRANSCRIPT_COMPLETED_EVENT,
item_id: "item_2",
transcript: "Second.",
});
// Then item_1's completion arrives
const result = mergeTranscriptEvent(state, {
type: TRANSCRIPT_COMPLETED_EVENT,
item_id: "item_1",
transcript: "First.",
});
// Order must follow committed order, not arrival order
assert.equal(result, "First. Second.");
});
it("inserts item after previous_item_id even if later items already exist", () => {
const state = createTranscriptSegmentState();
// Commit item_1
mergeTranscriptEvent(state, {
type: BUFFER_COMMITTED_EVENT,
item_id: "item_1",
});
// Commit item_3 (item_2 not yet committed)
mergeTranscriptEvent(state, {
type: BUFFER_COMMITTED_EVENT,
item_id: "item_3",
previous_item_id: "item_1",
});
// Now commit item_2 between item_1 and item_3
// (previous_item_id = item_1, so it goes after item_1)
mergeTranscriptEvent(state, {
type: BUFFER_COMMITTED_EVENT,
item_id: "item_2",
previous_item_id: "item_1",
});
// Add transcripts
mergeTranscriptEvent(state, {
type: TRANSCRIPT_COMPLETED_EVENT,
item_id: "item_1",
transcript: "One.",
});
mergeTranscriptEvent(state, {
type: TRANSCRIPT_COMPLETED_EVENT,
item_id: "item_2",
transcript: "Two.",
});
const result = mergeTranscriptEvent(state, {
type: TRANSCRIPT_COMPLETED_EVENT,
item_id: "item_3",
transcript: "Three.",
});
// item_2 was inserted after item_1 (before item_3)
assert.equal(result, "One. Two. Three.");
});
it("handles completion-only flow with committed order", () => {
const state = createTranscriptSegmentState();
// Server commits both items
mergeTranscriptEvent(state, {
type: BUFFER_COMMITTED_EVENT,
item_id: "item_A",
});
mergeTranscriptEvent(state, {
type: BUFFER_COMMITTED_EVENT,
item_id: "item_B",
previous_item_id: "item_A",
});
// Only completed events arrive (no deltas) — in reverse order
mergeTranscriptEvent(state, {
type: TRANSCRIPT_COMPLETED_EVENT,
item_id: "item_B",
transcript: "Bravo.",
});
const result = mergeTranscriptEvent(state, {
type: TRANSCRIPT_COMPLETED_EVENT,
item_id: "item_A",
transcript: "Alpha.",
});
assert.equal(result, "Alpha. Bravo.");
});
});
// Inline the logic to keep the test self-contained.
function requiresManualCommit(model) {
return model.includes("realtime-whisper");
}
describe("requiresManualCommit", () => {
it("returns true for gpt-realtime-whisper", () => {
assert.equal(requiresManualCommit("gpt-realtime-whisper"), true);
});
it("returns true for versioned realtime-whisper model", () => {
assert.equal(
requiresManualCommit("gpt-4o-realtime-whisper-20250512"),
true,
);
});
it("returns false for whisper-1", () => {
assert.equal(requiresManualCommit("whisper-1"), false);
});
it("returns false for gpt-4o-transcribe", () => {
assert.equal(requiresManualCommit("gpt-4o-transcribe"), false);
});
});
@@ -1,253 +0,0 @@
import { transcribeConnect } from "../api/transcribeSession";
import {
REALTIME_BUFFER_PROCESSOR_NAME,
createWorkletBlobUrl,
} from "./realtimeBufferWorklet";
export const TRANSCRIPT_DELTA_EVENT =
"conversation.item.input_audio_transcription.delta";
export const TRANSCRIPT_COMPLETED_EVENT =
"conversation.item.input_audio_transcription.completed";
export const TRANSCRIPT_FAILED_EVENT =
"conversation.item.input_audio_transcription.failed";
export const BUFFER_COMMITTED_EVENT = "input_audio_buffer.committed";
const MAX_BUFFER_CHUNKS = 500; // ~10s at 20ms per chunk
export type TranscriptEvent = {
type?: string;
item_id?: string;
previous_item_id?: string;
content_index?: number;
delta?: string;
transcript?: string;
message?: string;
error?: { message?: string };
};
export function createPeerConnection(): RTCPeerConnection {
return new RTCPeerConnection();
}
/**
* Complete the WebRTC SDP exchange via the relay.
*
* The relay mints the OpenAI session and proxies the SDP exchange in a single
* request — the client never sees the bearer token, and no server-side session
* state is needed between requests (works across HA relay replicas).
*
* Returns the model name from the relay so the caller can configure VAD mode.
*/
export async function connectPeerConnection(options: {
peerConnection: RTCPeerConnection;
}): Promise<{ model: string }> {
const offer = await options.peerConnection.createOffer();
await options.peerConnection.setLocalDescription(offer);
const { sdp: answerSdp, model } = await transcribeConnect(offer.sdp ?? "");
await options.peerConnection.setRemoteDescription({
type: "answer",
sdp: answerSdp,
});
return { model };
}
/**
* Per-item tracking for a single transcription turn. OpenAI's Realtime API
* sends delta and completed events tagged with an `item_id`; completed events
* for different turns can arrive out of order, so we reconcile by item id.
*/
interface ItemSegment {
/** Accumulated delta text (replaced by finalized on completion). */
pending: string;
/** Finalized text (set once the completed event arrives). */
finalized: string | null;
}
/**
* State for tracking transcription segments keyed by item id.
* Maintains insertion order so the full transcript is reconstructed
* in the order items were first seen.
*/
export interface TranscriptSegmentState {
/** Ordered item ids (insertion order = utterance order). */
itemOrder: string[];
/** Per-item segment data. */
items: Map<string, ItemSegment>;
}
export function createTranscriptSegmentState(): TranscriptSegmentState {
return { itemOrder: [], items: new Map() };
}
/** Get the current full transcript text from segment state. */
export function getTranscriptText(state: TranscriptSegmentState): string {
let result = "";
for (const id of state.itemOrder) {
const seg = state.items.get(id);
if (!seg) continue;
const text = seg.finalized ?? seg.pending;
if (!text) continue;
if (result && !result.endsWith(" ") && !text.startsWith(" ")) {
result += " ";
}
result += text;
}
return result;
}
/**
* Internal: get or create the segment for an item.
* When `previousItemId` is provided (from committed events), the new item is
* inserted after that item in `itemOrder` to preserve the server's utterance
* order — even if transcript events for a later item arrive first.
*/
function getOrCreateItem(
state: TranscriptSegmentState,
itemId: string,
previousItemId?: string,
): ItemSegment {
let seg = state.items.get(itemId);
if (!seg) {
seg = { pending: "", finalized: null };
state.items.set(itemId, seg);
if (previousItemId) {
const prevIndex = state.itemOrder.indexOf(previousItemId);
if (prevIndex !== -1) {
state.itemOrder.splice(prevIndex + 1, 0, itemId);
} else {
// Previous item not yet seen — append (best effort).
state.itemOrder.push(itemId);
}
} else {
state.itemOrder.push(itemId);
}
}
return seg;
}
/**
* Merge a transcript event into the segment state, keyed by `item_id`.
*
* - Committed events: register the item in the correct order using
* `previous_item_id` from the server, before any transcript arrives.
* - Delta events: append to the item's `pending` text.
* - Completed events: store `finalized` text, replacing accumulated deltas.
*
* Returns the full merged text across all items in order.
*/
export function mergeTranscriptEvent(
state: TranscriptSegmentState,
event: TranscriptEvent,
): string {
// Use item_id from the event; fall back to a synthetic key for events
// that lack one (shouldn't happen in practice, but be defensive).
const itemId = event.item_id ?? "__default__";
if (event.type === BUFFER_COMMITTED_EVENT) {
// Register the item in the correct position before transcripts arrive.
getOrCreateItem(state, itemId, event.previous_item_id ?? undefined);
} else if (event.type === TRANSCRIPT_DELTA_EVENT) {
const seg = getOrCreateItem(state, itemId);
const delta = event.delta ?? "";
if (delta) {
seg.pending += delta;
}
} else if (event.type === TRANSCRIPT_COMPLETED_EVENT) {
const seg = getOrCreateItem(state, itemId);
seg.finalized = event.transcript ?? "";
}
// Reconstruct full text from all items in order.
return getTranscriptText(state);
}
// ── Audio buffer capture ──────────────────────────────────────────────────
export interface AudioBufferCapture {
chunks: Int16Array[];
close(): void;
}
export async function createAudioBufferCapture(
stream: MediaStream,
): Promise<AudioBufferCapture> {
const audioContext = new AudioContext();
const blobUrl = createWorkletBlobUrl();
try {
await audioContext.audioWorklet.addModule(blobUrl);
} finally {
URL.revokeObjectURL(blobUrl);
}
const source = audioContext.createMediaStreamSource(stream);
const worklet = new AudioWorkletNode(
audioContext,
REALTIME_BUFFER_PROCESSOR_NAME,
);
source.connect(worklet);
worklet.connect(audioContext.destination);
const chunks: Int16Array[] = [];
worklet.port.onmessage = (event: MessageEvent<ArrayBuffer>) => {
if (chunks.length < MAX_BUFFER_CHUNKS) {
chunks.push(new Int16Array(event.data));
}
};
return {
chunks,
close() {
worklet.disconnect();
source.disconnect();
void audioContext.close();
},
};
}
// ── Flush buffered PCM into the data channel ──────────────────────────────
function int16ToBase64(pcm: Int16Array): string {
const bytes = new Uint8Array(pcm.buffer, pcm.byteOffset, pcm.byteLength);
let binary = "";
for (let i = 0; i < bytes.length; i++) {
binary += String.fromCharCode(bytes[i]);
}
return btoa(binary);
}
export function flushAudioBuffer(
dataChannel: RTCDataChannel,
chunks: Int16Array[],
): void {
for (const chunk of chunks) {
dataChannel.send(
JSON.stringify({
type: "input_audio_buffer.append",
audio: int16ToBase64(chunk),
}),
);
}
chunks.length = 0;
}
/**
* Send `input_audio_buffer.commit` to finalize buffered audio for transcription.
*
* Required when server VAD is disabled (e.g. `realtime-whisper` models) —
* without a commit, appended audio is never processed. For models using
* server VAD, the server commits automatically on speech boundaries.
*/
export function commitAudioBuffer(dataChannel: RTCDataChannel): void {
dataChannel.send(JSON.stringify({ type: "input_audio_buffer.commit" }));
}
/**
* Whether a transcription model requires manual audio commit (no server VAD).
* Models containing "realtime-whisper" use manual commit per OpenAI guidance.
*/
export function requiresManualCommit(model: string): boolean {
return model.includes("realtime-whisper");
}
@@ -1,53 +0,0 @@
const TARGET_SAMPLE_RATE = 24000;
const FRAME_SAMPLES = 480; // 20ms at 24kHz
export const REALTIME_BUFFER_PROCESSOR_NAME = "realtime-buffer-processor";
export const REALTIME_BUFFER_WORKLET_SOURCE = /* js */ `
class RealtimeBufferProcessor extends AudioWorkletProcessor {
constructor() {
super();
this._ratio = sampleRate / ${TARGET_SAMPLE_RATE};
this._offset = 0;
this._buf = new Float32Array(${FRAME_SAMPLES});
this._idx = 0;
}
process(inputs) {
const input = inputs[0]?.[0];
if (!input) return true;
while (this._offset < input.length) {
const i = Math.floor(this._offset);
const frac = this._offset - i;
const s0 = input[i];
const s1 = i + 1 < input.length ? input[i + 1] : s0;
this._buf[this._idx++] = s0 + frac * (s1 - s0);
if (this._idx >= ${FRAME_SAMPLES}) {
const pcm = new Int16Array(${FRAME_SAMPLES});
for (let j = 0; j < ${FRAME_SAMPLES}; j++) {
const s = Math.max(-1, Math.min(1, this._buf[j]));
pcm[j] = s < 0 ? s * 0x8000 : s * 0x7fff;
}
this.port.postMessage(pcm.buffer, [pcm.buffer]);
this._idx = 0;
}
this._offset += this._ratio;
}
this._offset -= input.length;
return true;
}
}
registerProcessor('${REALTIME_BUFFER_PROCESSOR_NAME}', RealtimeBufferProcessor);
`;
/** Create a blob URL that can be passed to `audioWorklet.addModule()`. */
export function createWorkletBlobUrl(): string {
return URL.createObjectURL(
new Blob([REALTIME_BUFFER_WORKLET_SOURCE], {
type: "application/javascript",
}),
);
}
@@ -2,6 +2,7 @@ import { Mic } from "lucide-react";
import { Button } from "@/shared/ui/button";
import { Tooltip, TooltipContent, TooltipTrigger } from "@/shared/ui/tooltip";
import { cn } from "@/shared/lib/cn";
import { isMacPlatform } from "@/shared/lib/platform";
interface DictationState {
isEnabled: boolean;
@@ -22,25 +23,19 @@ export function DictationButton({
}: DictationButtonProps) {
if (!dictation.isEnabled) return null;
const shortcutHint = isMacPlatform() ? "⌘D" : "Ctrl+D";
const tooltipText = dictation.isRecording
? "Stop recording"
: dictation.isTranscribing
? "Transcribing…"
: "Dictate message";
// Allow the stop action whenever the mic is live (isRecording), even if
// the session setup is still in progress (isStarting) or the composer is
// disabled. Only block the button when idle + disabled, or when startup
// hasn't captured the mic yet (isStarting && !isRecording).
const isDisabled = dictation.isRecording
? false
: disabled || dictation.isStarting;
: null;
return (
<Tooltip>
<TooltipTrigger asChild>
<Button
aria-label={tooltipText}
aria-label={tooltipText ?? "Voice Dictation"}
aria-pressed={dictation.isRecording}
className={cn(
"rounded-full",
@@ -48,7 +43,9 @@ export function DictationButton({
"bg-destructive text-destructive-foreground hover:bg-destructive/90 hover:text-destructive-foreground active:bg-destructive active:text-destructive-foreground",
dictation.isTranscribing && "animate-pulse",
)}
disabled={isDisabled}
disabled={
dictation.isRecording ? false : disabled || dictation.isStarting
}
onClick={dictation.toggleRecording}
size="icon"
type="button"
@@ -57,7 +54,16 @@ export function DictationButton({
<Mic />
</Button>
</TooltipTrigger>
<TooltipContent>{tooltipText}</TooltipContent>
<TooltipContent>
{tooltipText ?? (
<span className="flex items-center gap-1.5">
Voice Dictation
<kbd className="rounded border border-foreground/10 px-1 py-0.5 text-2xs font-medium text-foreground/60">
{shortcutHint}
</kbd>
</span>
)}
</TooltipContent>
</Tooltip>
);
}
@@ -181,6 +181,14 @@ export const KEYBOARD_SHORTCUTS: KeyboardShortcut[] = [
keysWindows: "Ctrl+Space",
category: "Messages",
},
{
id: "voice-dictation",
label: "Voice Dictation",
description: "Start or stop voice dictation in the composer",
keys: "⌘D",
keysWindows: "Ctrl+D",
category: "Messages",
},
// Formatting
{