diff --git a/desktop/src/app/AppShell.tsx b/desktop/src/app/AppShell.tsx index 3a4146caf..5d06c7227 100644 --- a/desktop/src/app/AppShell.tsx +++ b/desktop/src/app/AppShell.tsx @@ -595,6 +595,12 @@ export function AppShell() { return; } + if (key === "d" && !event.shiftKey) { + event.preventDefault(); + window.dispatchEvent(new CustomEvent("buzz:toggle-dictation")); + return; + } + if (key === "a" && event.shiftKey) { event.preventDefault(); void goHome(); diff --git a/desktop/src/features/dictation/api/transcribeSession.ts b/desktop/src/features/dictation/api/transcribeSession.ts deleted file mode 100644 index c0f0cc03f..000000000 --- a/desktop/src/features/dictation/api/transcribeSession.ts +++ /dev/null @@ -1,76 +0,0 @@ -import { getRelayHttpUrl, signRelayEvent } from "@/shared/api/tauri"; - -export interface TranscribeStatus { - configured: boolean; - model: string; -} - -export interface TranscribeConnectResponse { - sdp: string; - model: string; -} - -/** NIP-98 event kind for HTTP request authorization. */ -const NIP98_KIND = 27235; - -/** - * Build a NIP-98 `Authorization: Nostr ` header for an HTTP request. - * - * The relay verifies the signed event's `u` tag against its own - * host-derived expected URL, so `url` must be the exact absolute URL being - * fetched (scheme + host + path). The `method` tag must match the request. - */ -async function nip98AuthHeader(url: string, method: string): Promise { - const nonce = crypto.randomUUID(); - const event = await signRelayEvent({ - kind: NIP98_KIND, - content: "", - tags: [ - ["u", url], - ["method", method], - ["nonce", nonce], - ], - }); - const json = JSON.stringify(event); - // btoa needs a binary string; encode UTF-8 first so non-ASCII survives. - const base64 = btoa(String.fromCharCode(...new TextEncoder().encode(json))); - return `Nostr ${base64}`; -} - -export async function getTranscribeStatus(): Promise { - const baseUrl = await getRelayHttpUrl(); - const url = `${baseUrl}/transcribe/status`; - const response = await fetch(url, { - headers: { Authorization: await nip98AuthHeader(url, "GET") }, - }); - if (!response.ok) { - throw new Error(`Transcribe status check failed: ${response.status}`); - } - return response.json(); -} - -/** - * Mint an OpenAI Realtime session and complete the WebRTC SDP exchange in a - * single relay round-trip. The relay holds the OpenAI bearer token server-side - * — the client never sees it. This also works correctly across multiple relay - * replicas since no server-side session state is needed between requests. - */ -export async function transcribeConnect( - sdp: string, -): Promise { - const baseUrl = await getRelayHttpUrl(); - const url = `${baseUrl}/transcribe/connect`; - const response = await fetch(url, { - method: "POST", - headers: { - "Content-Type": "application/json", - Authorization: await nip98AuthHeader(url, "POST"), - }, - body: JSON.stringify({ sdp }), - }); - if (!response.ok) { - const body = await response.text().catch(() => ""); - throw new Error(`Transcribe connect failed (${response.status}): ${body}`); - } - return response.json(); -} diff --git a/desktop/src/features/dictation/hooks/useComposerDictation.ts b/desktop/src/features/dictation/hooks/useComposerDictation.ts index b9d8fd3c2..46e252d15 100644 --- a/desktop/src/features/dictation/hooks/useComposerDictation.ts +++ b/desktop/src/features/dictation/hooks/useComposerDictation.ts @@ -64,12 +64,23 @@ export function useComposerDictation({ // Auto-cancel dictation when the composer becomes disabled mid-recording // (e.g. channel becomes read-only, parent send state disables thread composer). - // Without this, the WebRTC session keeps running with no way to stop it. + // Without this, the STT session keeps running with no way to stop it. useEffect(() => { if (disabled && dictation.isRecording) { dictation.cancelRecording(); } }, [disabled, dictation.isRecording, dictation.cancelRecording]); + // ⌘D global shortcut — dispatched from AppShell's keydown handler. + useEffect(() => { + function handleToggle() { + dictation.toggleRecording(); + } + window.addEventListener("buzz:toggle-dictation", handleToggle); + return () => { + window.removeEventListener("buzz:toggle-dictation", handleToggle); + }; + }, [dictation.toggleRecording]); + return dictation; } diff --git a/desktop/src/features/dictation/hooks/useDictation.ts b/desktop/src/features/dictation/hooks/useDictation.ts index 36050a086..9265b171d 100644 --- a/desktop/src/features/dictation/hooks/useDictation.ts +++ b/desktop/src/features/dictation/hooks/useDictation.ts @@ -7,7 +7,6 @@ import { replaceTrailingTranscribedText, } from "../lib/voiceInput"; import { useLocalDictation } from "./useLocalDictation"; -import { useRealtimeDictation } from "./useRealtimeDictation"; interface UseDictationOptions { /** Returns the current composer text (must be fresh — synced from editor). */ @@ -71,25 +70,13 @@ export function useDictation({ [autoSubmitPhrases, getText, onSend, isSendBlockedRef, setText], ); - // Try local STT first (offline, no API key needed). - const localDictation = useLocalDictation({ + const dictation = useLocalDictation({ onRecordingStart: () => { lastTranscriptRef.current = ""; }, onTranscriptText: handleTranscript, }); - // Fall back to cloud (OpenAI Realtime) if local is unavailable. - const cloudDictation = useRealtimeDictation({ - disabled: localDictation.isEnabled, // Disable cloud when local is available - onRecordingStart: () => { - lastTranscriptRef.current = ""; - }, - onTranscriptText: handleTranscript, - }); - - // Use whichever is enabled — local takes priority. - const dictation = localDictation.isEnabled ? localDictation : cloudDictation; stopRecordingRef.current = dictation.stopRecording; return dictation; diff --git a/desktop/src/features/dictation/hooks/useRealtimeDictation.ts b/desktop/src/features/dictation/hooks/useRealtimeDictation.ts deleted file mode 100644 index 07518e26a..000000000 --- a/desktop/src/features/dictation/hooks/useRealtimeDictation.ts +++ /dev/null @@ -1,363 +0,0 @@ -import { useCallback, useEffect, useRef, useState } from "react"; -import { toast } from "sonner"; -import { getTranscribeStatus } from "../api/transcribeSession"; -import { - type AudioBufferCapture, - type TranscriptEvent, - type TranscriptSegmentState, - BUFFER_COMMITTED_EVENT, - TRANSCRIPT_COMPLETED_EVENT, - TRANSCRIPT_DELTA_EVENT, - TRANSCRIPT_FAILED_EVENT, - commitAudioBuffer, - connectPeerConnection, - createAudioBufferCapture, - createPeerConnection, - createTranscriptSegmentState, - flushAudioBuffer, - getTranscriptText, - mergeTranscriptEvent, - requiresManualCommit, -} from "../lib/realtimeAudio"; - -interface UseRealtimeDictationOptions { - disabled?: boolean; - onRecordingStart?: () => void; - onTranscriptText: (text: string) => void; -} - -function closeResources(resources: { - audioCapture?: AudioBufferCapture | null; - dataChannel?: RTCDataChannel | null; - peerConnection?: RTCPeerConnection | null; - stream?: MediaStream | null; -}) { - resources.audioCapture?.close(); - resources.dataChannel?.close(); - resources.peerConnection?.close(); - for (const track of resources.stream?.getTracks() ?? []) { - track.stop(); - } -} - -export function useRealtimeDictation({ - disabled = false, - onRecordingStart, - onTranscriptText, -}: UseRealtimeDictationOptions) { - const [isRecording, setIsRecording] = useState(false); - const [isStarting, setIsStarting] = useState(false); - const [isTranscribing, setIsTranscribing] = useState(false); - const [isConfigured, setIsConfigured] = useState(false); - - const peerConnectionRef = useRef(null); - const dataChannelRef = useRef(null); - const streamRef = useRef(null); - const audioCaptureRef = useRef(null); - const segmentStateRef = useRef( - createTranscriptSegmentState(), - ); - const activeRunIdRef = useRef(0); - const manualCommitRef = useRef(false); - const commitIntervalRef = useRef | null>(null); - const onRecordingStartRef = useRef(onRecordingStart); - const onTranscriptTextRef = useRef(onTranscriptText); - - onRecordingStartRef.current = onRecordingStart; - onTranscriptTextRef.current = onTranscriptText; - - const isEnabled = !disabled && isConfigured; - - // Check if transcription is configured on mount - useEffect(() => { - let cancelled = false; - getTranscribeStatus() - .then((status) => { - if (!cancelled) setIsConfigured(status.configured); - }) - .catch(() => { - if (!cancelled) setIsConfigured(false); - }); - return () => { - cancelled = true; - }; - }, []); - - /** - * Tear down recording resources. - * - * @param invalidateRun - When true, immediately marks the run stale so - * late transcript events are rejected (used by send/edit-save/navigation). - * When false (user-initiated stop), the run stays valid during the grace - * window so the final commit's transcript is delivered to the composer. - */ - const cleanupResources = useCallback((invalidateRun = true) => { - // Clear periodic commit interval if active. - if (commitIntervalRef.current) { - clearInterval(commitIntervalRef.current); - commitIntervalRef.current = null; - } - - // For manual-commit models (no server VAD), commit any buffered audio - // before tearing down the connection so OpenAI processes the final chunk. - // We keep the data channel open briefly to receive the transcript response. - const dc = dataChannelRef.current; - const needsCommit = - manualCommitRef.current && dc && dc.readyState === "open"; - manualCommitRef.current = false; - - // Send final commit for manual-commit models. - if (needsCommit && dc) { - commitAudioBuffer(dc); - } - - // Determine whether we need a grace window before full teardown. - // User-initiated stop (!invalidateRun) needs a grace period for BOTH: - // - Manual-commit models: final commit's transcript response - // - Server-VAD models: in-flight VAD completion events - const needsGrace = !invalidateRun && dc && dc.readyState === "open"; - - if (needsGrace && dc) { - // Stop the mic immediately so no new audio is sent. - for (const track of streamRef.current?.getTracks() ?? []) { - track.stop(); - } - streamRef.current = null; - audioCaptureRef.current?.close(); - audioCaptureRef.current = null; - // Delay WebRTC teardown briefly so final transcript events arrive. - const pc = peerConnectionRef.current; - peerConnectionRef.current = null; - dataChannelRef.current = null; - const runId = activeRunIdRef.current; - - setTimeout(() => { - // After the grace window, invalidate the run and tear down. - if (activeRunIdRef.current === runId) { - activeRunIdRef.current += 1; - } - dc.close(); - pc?.close(); - }, 3000); - return; - } - - // Immediate teardown: send/navigation cancellation, or no open channel. - activeRunIdRef.current += 1; - closeResources({ - audioCapture: audioCaptureRef.current, - dataChannel: dataChannelRef.current, - peerConnection: peerConnectionRef.current, - stream: streamRef.current, - }); - audioCaptureRef.current = null; - dataChannelRef.current = null; - peerConnectionRef.current = null; - streamRef.current = null; - }, []); - - /** Full cleanup — invalidates the run (used by send/navigation). */ - const cleanup = useCallback(() => { - cleanupResources(true); - setIsRecording(false); - setIsStarting(false); - setIsTranscribing(false); - }, [cleanupResources]); - - /** User-initiated stop — preserves final transcript for manual-commit models. */ - const userStop = useCallback(() => { - cleanupResources(false); - setIsRecording(false); - setIsStarting(false); - // Note: isTranscribing stays true briefly while the final transcript arrives. - }, [cleanupResources]); - - useEffect(() => cleanupResources, [cleanupResources]); - - const handleRealtimeEvent = useCallback( - (runId: number, event: TranscriptEvent) => { - // Ignore events from a stale run (e.g. user sent/stopped while - // transcripts were still in-flight from the data channel). - if (activeRunIdRef.current !== runId) return; - - if (event.type === "error") { - console.error("OpenAI realtime server error", event); - toast.error(event.error?.message ?? "Voice input error"); - return; - } - - if (event.type === TRANSCRIPT_FAILED_EVENT) { - console.warn("OpenAI transcription failed for item", event); - setIsTranscribing(false); - toast.error(event.error?.message ?? "Transcription failed"); - return; - } - - if ( - event.type !== TRANSCRIPT_DELTA_EVENT && - event.type !== TRANSCRIPT_COMPLETED_EVENT && - event.type !== BUFFER_COMMITTED_EVENT - ) { - return; - } - - const prevText = getTranscriptText(segmentStateRef.current); - const merged = mergeTranscriptEvent(segmentStateRef.current, event); - - // Always update transcribing state — a completed event must clear the - // indicator even when the final text matches the accumulated deltas. - const stillTranscribing = event.type !== TRANSCRIPT_COMPLETED_EVENT; - setIsTranscribing(stillTranscribing); - - if (merged === prevText) return; - onTranscriptTextRef.current(merged); - }, - [], - ); - - const startRecording = useCallback(async () => { - if (!isEnabled || isStarting || isRecording) return; - - const runId = activeRunIdRef.current + 1; - activeRunIdRef.current = runId; - const isStaleRun = () => activeRunIdRef.current !== runId; - - let stream: MediaStream | null = null; - let audioCapture: AudioBufferCapture | null = null; - let peerConnection: RTCPeerConnection | null = null; - let dataChannel: RTCDataChannel | null = null; - - setIsStarting(true); - segmentStateRef.current = createTranscriptSegmentState(); - onRecordingStartRef.current?.(); - - try { - // 1. Capture mic immediately for instant feedback - stream = await navigator.mediaDevices.getUserMedia({ - audio: { - autoGainControl: true, - echoCancellation: true, - noiseSuppression: true, - }, - }); - if (isStaleRun()) { - closeResources({ stream }); - return; - } - streamRef.current = stream; - setIsRecording(true); - - // 2. Buffer PCM via AudioWorklet while network calls proceed - audioCapture = await createAudioBufferCapture(stream); - if (isStaleRun()) { - closeResources({ audioCapture, stream }); - return; - } - audioCaptureRef.current = audioCapture; - - // 3. Set up WebRTC peer connection - peerConnection = createPeerConnection(); - peerConnectionRef.current = peerConnection; - const activeStream = stream; - stream.getAudioTracks().forEach((track) => { - peerConnection?.addTrack(track, activeStream); - }); - - dataChannel = peerConnection.createDataChannel("oai-events"); - dataChannelRef.current = dataChannel; - dataChannel.addEventListener("message", (message) => { - try { - handleRealtimeEvent(runId, JSON.parse(String(message.data))); - } catch { - // Ignore non-JSON events - } - }); - - // Flush buffered audio once data channel opens - const channelToFlush = dataChannel; - const captureToFlush = audioCapture; - dataChannel.addEventListener("open", () => { - if (isStaleRun()) { - captureToFlush.close(); - return; - } - flushAudioBuffer(channelToFlush, captureToFlush.chunks); - if (manualCommitRef.current) { - commitAudioBuffer(channelToFlush); - commitIntervalRef.current = setInterval(() => { - if (channelToFlush.readyState === "open") { - commitAudioBuffer(channelToFlush); - } - }, 2000); - } - captureToFlush.close(); - audioCaptureRef.current = null; - }); - - // 4. Session creation + SDP exchange in a single relay round-trip. - // No cross-replica state needed — works in HA deployments. - const { model } = await connectPeerConnection({ peerConnection }); - if (isStaleRun()) { - closeResources({ audioCapture, dataChannel, peerConnection, stream }); - return; - } - manualCommitRef.current = requiresManualCommit(model); - } catch (error) { - closeResources({ audioCapture, dataChannel, peerConnection, stream }); - if (!isStaleRun()) { - audioCaptureRef.current = null; - dataChannelRef.current = null; - peerConnectionRef.current = null; - streamRef.current = null; - setIsRecording(false); - setIsTranscribing(false); - - const message = - error instanceof Error ? error.message : "Voice input failed"; - if (/not allowed|denied|permission/i.test(message)) { - toast.error("Microphone access denied", { - description: - "Allow microphone access in System Settings to use dictation.", - }); - } else if (/not found|no audio/i.test(message)) { - toast.error("No microphone found", { - description: "Connect a microphone and try again.", - }); - } else { - toast.error("Voice input failed", { description: message }); - } - } - } finally { - if (!isStaleRun()) setIsStarting(false); - } - }, [handleRealtimeEvent, isEnabled, isRecording, isStarting]); - - /** User-initiated stop (mic button) — preserves final transcript. */ - const stopRecording = useCallback(() => userStop(), [userStop]); - - /** - * Cancel recording and reject all pending transcripts. Used by send, - * edit-save, and navigation to prevent late events from refilling the - * composer after the content has been dispatched. - */ - const cancelRecording = useCallback(() => cleanup(), [cleanup]); - - const toggleRecording = useCallback(() => { - if (isRecording || isStarting) { - stopRecording(); - return; - } - void startRecording(); - }, [isRecording, isStarting, startRecording, stopRecording]); - - return { - isEnabled, - isRecording, - isStarting, - isTranscribing, - startRecording, - stopRecording, - cancelRecording, - toggleRecording, - }; -} diff --git a/desktop/src/features/dictation/lib/realtimeAudio.test.mjs b/desktop/src/features/dictation/lib/realtimeAudio.test.mjs deleted file mode 100644 index 0121ef436..000000000 --- a/desktop/src/features/dictation/lib/realtimeAudio.test.mjs +++ /dev/null @@ -1,334 +0,0 @@ -import { describe, it } from "node:test"; -import assert from "node:assert/strict"; - -// Inline the logic to keep the test self-contained and avoid bundler issues. -const TRANSCRIPT_DELTA_EVENT = - "conversation.item.input_audio_transcription.delta"; -const TRANSCRIPT_COMPLETED_EVENT = - "conversation.item.input_audio_transcription.completed"; -const BUFFER_COMMITTED_EVENT = "input_audio_buffer.committed"; - -function createTranscriptSegmentState() { - return { itemOrder: [], items: new Map() }; -} - -function getOrCreateItem(state, itemId, previousItemId) { - let seg = state.items.get(itemId); - if (!seg) { - seg = { pending: "", finalized: null }; - state.items.set(itemId, seg); - if (previousItemId) { - const prevIndex = state.itemOrder.indexOf(previousItemId); - if (prevIndex !== -1) { - state.itemOrder.splice(prevIndex + 1, 0, itemId); - } else { - state.itemOrder.push(itemId); - } - } else { - state.itemOrder.push(itemId); - } - } - return seg; -} - -function getTranscriptText(state) { - let result = ""; - for (const id of state.itemOrder) { - const seg = state.items.get(id); - if (!seg) continue; - const text = seg.finalized ?? seg.pending; - if (!text) continue; - if (result && !result.endsWith(" ") && !text.startsWith(" ")) { - result += " "; - } - result += text; - } - return result; -} - -function mergeTranscriptEvent(state, event) { - const itemId = event.item_id ?? "__default__"; - - if (event.type === BUFFER_COMMITTED_EVENT) { - getOrCreateItem(state, itemId, event.previous_item_id ?? undefined); - } else if (event.type === TRANSCRIPT_DELTA_EVENT) { - const seg = getOrCreateItem(state, itemId); - const delta = event.delta ?? ""; - if (delta) { - seg.pending += delta; - } - } else if (event.type === TRANSCRIPT_COMPLETED_EVENT) { - const seg = getOrCreateItem(state, itemId); - seg.finalized = event.transcript ?? ""; - } - - return getTranscriptText(state); -} - -describe("mergeTranscriptEvent", () => { - it("accumulates delta events for a single item", () => { - const state = createTranscriptSegmentState(); - const r1 = mergeTranscriptEvent(state, { - type: TRANSCRIPT_DELTA_EVENT, - item_id: "item_1", - delta: "hello ", - }); - assert.equal(r1, "hello "); - - const r2 = mergeTranscriptEvent(state, { - type: TRANSCRIPT_DELTA_EVENT, - item_id: "item_1", - delta: "world", - }); - assert.equal(r2, "hello world"); - }); - - it("replaces deltas with finalized text on completed event", () => { - const state = createTranscriptSegmentState(); - mergeTranscriptEvent(state, { - type: TRANSCRIPT_DELTA_EVENT, - item_id: "item_1", - delta: "hello world", - }); - - const result = mergeTranscriptEvent(state, { - type: TRANSCRIPT_COMPLETED_EVENT, - item_id: "item_1", - transcript: "Hello, world.", - }); - assert.equal(result, "Hello, world."); - }); - - it("handles multiple items in order", () => { - const state = createTranscriptSegmentState(); - - // First item - mergeTranscriptEvent(state, { - type: TRANSCRIPT_DELTA_EVENT, - item_id: "item_1", - delta: "first ", - }); - mergeTranscriptEvent(state, { - type: TRANSCRIPT_COMPLETED_EVENT, - item_id: "item_1", - transcript: "First.", - }); - - // Second item - mergeTranscriptEvent(state, { - type: TRANSCRIPT_DELTA_EVENT, - item_id: "item_2", - delta: "second", - }); - assert.equal(state.items.get("item_2").pending, "second"); - - const result = mergeTranscriptEvent(state, { - type: TRANSCRIPT_COMPLETED_EVENT, - item_id: "item_2", - transcript: "Second.", - }); - assert.equal(result, "First. Second."); - }); - - it("handles out-of-order completed events by item id", () => { - const state = createTranscriptSegmentState(); - - // Both items start with deltas - mergeTranscriptEvent(state, { - type: TRANSCRIPT_DELTA_EVENT, - item_id: "item_1", - delta: "first", - }); - mergeTranscriptEvent(state, { - type: TRANSCRIPT_DELTA_EVENT, - item_id: "item_2", - delta: "second", - }); - - // item_2 completes before item_1 - mergeTranscriptEvent(state, { - type: TRANSCRIPT_COMPLETED_EVENT, - item_id: "item_2", - transcript: "Second.", - }); - - // item_1 still shows pending - let result = mergeTranscriptEvent(state, { - type: TRANSCRIPT_DELTA_EVENT, - item_id: "item_1", - delta: " more", - }); - assert.equal(result, "first more Second."); - - // item_1 finally completes - result = mergeTranscriptEvent(state, { - type: TRANSCRIPT_COMPLETED_EVENT, - item_id: "item_1", - transcript: "First more.", - }); - assert.equal(result, "First more. Second."); - }); - - it("does not duplicate text on completed event", () => { - const state = createTranscriptSegmentState(); - - mergeTranscriptEvent(state, { - type: TRANSCRIPT_DELTA_EVENT, - item_id: "item_1", - delta: "hello world", - }); - - const result = mergeTranscriptEvent(state, { - type: TRANSCRIPT_COMPLETED_EVENT, - item_id: "item_1", - transcript: "Hello, world.", - }); - assert.equal(result, "Hello, world."); - }); - - it("falls back to __default__ when item_id is missing", () => { - const state = createTranscriptSegmentState(); - mergeTranscriptEvent(state, { - type: TRANSCRIPT_DELTA_EVENT, - delta: "no id", - }); - const result = mergeTranscriptEvent(state, { - type: TRANSCRIPT_COMPLETED_EVENT, - transcript: "No id.", - }); - assert.equal(result, "No id."); - }); - - it("preserves committed item order when completions arrive out of order", () => { - const state = createTranscriptSegmentState(); - - // Server commits items in order: item_1 then item_2 - mergeTranscriptEvent(state, { - type: BUFFER_COMMITTED_EVENT, - item_id: "item_1", - previous_item_id: null, - }); - mergeTranscriptEvent(state, { - type: BUFFER_COMMITTED_EVENT, - item_id: "item_2", - previous_item_id: "item_1", - }); - - // But item_2's completion arrives first - mergeTranscriptEvent(state, { - type: TRANSCRIPT_COMPLETED_EVENT, - item_id: "item_2", - transcript: "Second.", - }); - - // Then item_1's completion arrives - const result = mergeTranscriptEvent(state, { - type: TRANSCRIPT_COMPLETED_EVENT, - item_id: "item_1", - transcript: "First.", - }); - - // Order must follow committed order, not arrival order - assert.equal(result, "First. Second."); - }); - - it("inserts item after previous_item_id even if later items already exist", () => { - const state = createTranscriptSegmentState(); - - // Commit item_1 - mergeTranscriptEvent(state, { - type: BUFFER_COMMITTED_EVENT, - item_id: "item_1", - }); - - // Commit item_3 (item_2 not yet committed) - mergeTranscriptEvent(state, { - type: BUFFER_COMMITTED_EVENT, - item_id: "item_3", - previous_item_id: "item_1", - }); - - // Now commit item_2 between item_1 and item_3 - // (previous_item_id = item_1, so it goes after item_1) - mergeTranscriptEvent(state, { - type: BUFFER_COMMITTED_EVENT, - item_id: "item_2", - previous_item_id: "item_1", - }); - - // Add transcripts - mergeTranscriptEvent(state, { - type: TRANSCRIPT_COMPLETED_EVENT, - item_id: "item_1", - transcript: "One.", - }); - mergeTranscriptEvent(state, { - type: TRANSCRIPT_COMPLETED_EVENT, - item_id: "item_2", - transcript: "Two.", - }); - const result = mergeTranscriptEvent(state, { - type: TRANSCRIPT_COMPLETED_EVENT, - item_id: "item_3", - transcript: "Three.", - }); - - // item_2 was inserted after item_1 (before item_3) - assert.equal(result, "One. Two. Three."); - }); - - it("handles completion-only flow with committed order", () => { - const state = createTranscriptSegmentState(); - - // Server commits both items - mergeTranscriptEvent(state, { - type: BUFFER_COMMITTED_EVENT, - item_id: "item_A", - }); - mergeTranscriptEvent(state, { - type: BUFFER_COMMITTED_EVENT, - item_id: "item_B", - previous_item_id: "item_A", - }); - - // Only completed events arrive (no deltas) — in reverse order - mergeTranscriptEvent(state, { - type: TRANSCRIPT_COMPLETED_EVENT, - item_id: "item_B", - transcript: "Bravo.", - }); - const result = mergeTranscriptEvent(state, { - type: TRANSCRIPT_COMPLETED_EVENT, - item_id: "item_A", - transcript: "Alpha.", - }); - - assert.equal(result, "Alpha. Bravo."); - }); -}); - -// Inline the logic to keep the test self-contained. -function requiresManualCommit(model) { - return model.includes("realtime-whisper"); -} - -describe("requiresManualCommit", () => { - it("returns true for gpt-realtime-whisper", () => { - assert.equal(requiresManualCommit("gpt-realtime-whisper"), true); - }); - - it("returns true for versioned realtime-whisper model", () => { - assert.equal( - requiresManualCommit("gpt-4o-realtime-whisper-20250512"), - true, - ); - }); - - it("returns false for whisper-1", () => { - assert.equal(requiresManualCommit("whisper-1"), false); - }); - - it("returns false for gpt-4o-transcribe", () => { - assert.equal(requiresManualCommit("gpt-4o-transcribe"), false); - }); -}); diff --git a/desktop/src/features/dictation/lib/realtimeAudio.ts b/desktop/src/features/dictation/lib/realtimeAudio.ts deleted file mode 100644 index 1ce20b0dd..000000000 --- a/desktop/src/features/dictation/lib/realtimeAudio.ts +++ /dev/null @@ -1,253 +0,0 @@ -import { transcribeConnect } from "../api/transcribeSession"; -import { - REALTIME_BUFFER_PROCESSOR_NAME, - createWorkletBlobUrl, -} from "./realtimeBufferWorklet"; - -export const TRANSCRIPT_DELTA_EVENT = - "conversation.item.input_audio_transcription.delta"; -export const TRANSCRIPT_COMPLETED_EVENT = - "conversation.item.input_audio_transcription.completed"; -export const TRANSCRIPT_FAILED_EVENT = - "conversation.item.input_audio_transcription.failed"; -export const BUFFER_COMMITTED_EVENT = "input_audio_buffer.committed"; - -const MAX_BUFFER_CHUNKS = 500; // ~10s at 20ms per chunk - -export type TranscriptEvent = { - type?: string; - item_id?: string; - previous_item_id?: string; - content_index?: number; - delta?: string; - transcript?: string; - message?: string; - error?: { message?: string }; -}; - -export function createPeerConnection(): RTCPeerConnection { - return new RTCPeerConnection(); -} - -/** - * Complete the WebRTC SDP exchange via the relay. - * - * The relay mints the OpenAI session and proxies the SDP exchange in a single - * request — the client never sees the bearer token, and no server-side session - * state is needed between requests (works across HA relay replicas). - * - * Returns the model name from the relay so the caller can configure VAD mode. - */ -export async function connectPeerConnection(options: { - peerConnection: RTCPeerConnection; -}): Promise<{ model: string }> { - const offer = await options.peerConnection.createOffer(); - await options.peerConnection.setLocalDescription(offer); - - const { sdp: answerSdp, model } = await transcribeConnect(offer.sdp ?? ""); - - await options.peerConnection.setRemoteDescription({ - type: "answer", - sdp: answerSdp, - }); - - return { model }; -} - -/** - * Per-item tracking for a single transcription turn. OpenAI's Realtime API - * sends delta and completed events tagged with an `item_id`; completed events - * for different turns can arrive out of order, so we reconcile by item id. - */ -interface ItemSegment { - /** Accumulated delta text (replaced by finalized on completion). */ - pending: string; - /** Finalized text (set once the completed event arrives). */ - finalized: string | null; -} - -/** - * State for tracking transcription segments keyed by item id. - * Maintains insertion order so the full transcript is reconstructed - * in the order items were first seen. - */ -export interface TranscriptSegmentState { - /** Ordered item ids (insertion order = utterance order). */ - itemOrder: string[]; - /** Per-item segment data. */ - items: Map; -} - -export function createTranscriptSegmentState(): TranscriptSegmentState { - return { itemOrder: [], items: new Map() }; -} - -/** Get the current full transcript text from segment state. */ -export function getTranscriptText(state: TranscriptSegmentState): string { - let result = ""; - for (const id of state.itemOrder) { - const seg = state.items.get(id); - if (!seg) continue; - const text = seg.finalized ?? seg.pending; - if (!text) continue; - if (result && !result.endsWith(" ") && !text.startsWith(" ")) { - result += " "; - } - result += text; - } - return result; -} - -/** - * Internal: get or create the segment for an item. - * When `previousItemId` is provided (from committed events), the new item is - * inserted after that item in `itemOrder` to preserve the server's utterance - * order — even if transcript events for a later item arrive first. - */ -function getOrCreateItem( - state: TranscriptSegmentState, - itemId: string, - previousItemId?: string, -): ItemSegment { - let seg = state.items.get(itemId); - if (!seg) { - seg = { pending: "", finalized: null }; - state.items.set(itemId, seg); - if (previousItemId) { - const prevIndex = state.itemOrder.indexOf(previousItemId); - if (prevIndex !== -1) { - state.itemOrder.splice(prevIndex + 1, 0, itemId); - } else { - // Previous item not yet seen — append (best effort). - state.itemOrder.push(itemId); - } - } else { - state.itemOrder.push(itemId); - } - } - return seg; -} - -/** - * Merge a transcript event into the segment state, keyed by `item_id`. - * - * - Committed events: register the item in the correct order using - * `previous_item_id` from the server, before any transcript arrives. - * - Delta events: append to the item's `pending` text. - * - Completed events: store `finalized` text, replacing accumulated deltas. - * - * Returns the full merged text across all items in order. - */ -export function mergeTranscriptEvent( - state: TranscriptSegmentState, - event: TranscriptEvent, -): string { - // Use item_id from the event; fall back to a synthetic key for events - // that lack one (shouldn't happen in practice, but be defensive). - const itemId = event.item_id ?? "__default__"; - - if (event.type === BUFFER_COMMITTED_EVENT) { - // Register the item in the correct position before transcripts arrive. - getOrCreateItem(state, itemId, event.previous_item_id ?? undefined); - } else if (event.type === TRANSCRIPT_DELTA_EVENT) { - const seg = getOrCreateItem(state, itemId); - const delta = event.delta ?? ""; - if (delta) { - seg.pending += delta; - } - } else if (event.type === TRANSCRIPT_COMPLETED_EVENT) { - const seg = getOrCreateItem(state, itemId); - seg.finalized = event.transcript ?? ""; - } - - // Reconstruct full text from all items in order. - return getTranscriptText(state); -} - -// ── Audio buffer capture ────────────────────────────────────────────────── - -export interface AudioBufferCapture { - chunks: Int16Array[]; - close(): void; -} - -export async function createAudioBufferCapture( - stream: MediaStream, -): Promise { - const audioContext = new AudioContext(); - const blobUrl = createWorkletBlobUrl(); - try { - await audioContext.audioWorklet.addModule(blobUrl); - } finally { - URL.revokeObjectURL(blobUrl); - } - - const source = audioContext.createMediaStreamSource(stream); - const worklet = new AudioWorkletNode( - audioContext, - REALTIME_BUFFER_PROCESSOR_NAME, - ); - source.connect(worklet); - worklet.connect(audioContext.destination); - - const chunks: Int16Array[] = []; - worklet.port.onmessage = (event: MessageEvent) => { - if (chunks.length < MAX_BUFFER_CHUNKS) { - chunks.push(new Int16Array(event.data)); - } - }; - - return { - chunks, - close() { - worklet.disconnect(); - source.disconnect(); - void audioContext.close(); - }, - }; -} - -// ── Flush buffered PCM into the data channel ────────────────────────────── - -function int16ToBase64(pcm: Int16Array): string { - const bytes = new Uint8Array(pcm.buffer, pcm.byteOffset, pcm.byteLength); - let binary = ""; - for (let i = 0; i < bytes.length; i++) { - binary += String.fromCharCode(bytes[i]); - } - return btoa(binary); -} - -export function flushAudioBuffer( - dataChannel: RTCDataChannel, - chunks: Int16Array[], -): void { - for (const chunk of chunks) { - dataChannel.send( - JSON.stringify({ - type: "input_audio_buffer.append", - audio: int16ToBase64(chunk), - }), - ); - } - chunks.length = 0; -} - -/** - * Send `input_audio_buffer.commit` to finalize buffered audio for transcription. - * - * Required when server VAD is disabled (e.g. `realtime-whisper` models) — - * without a commit, appended audio is never processed. For models using - * server VAD, the server commits automatically on speech boundaries. - */ -export function commitAudioBuffer(dataChannel: RTCDataChannel): void { - dataChannel.send(JSON.stringify({ type: "input_audio_buffer.commit" })); -} - -/** - * Whether a transcription model requires manual audio commit (no server VAD). - * Models containing "realtime-whisper" use manual commit per OpenAI guidance. - */ -export function requiresManualCommit(model: string): boolean { - return model.includes("realtime-whisper"); -} diff --git a/desktop/src/features/dictation/lib/realtimeBufferWorklet.ts b/desktop/src/features/dictation/lib/realtimeBufferWorklet.ts deleted file mode 100644 index ac8ffaabd..000000000 --- a/desktop/src/features/dictation/lib/realtimeBufferWorklet.ts +++ /dev/null @@ -1,53 +0,0 @@ -const TARGET_SAMPLE_RATE = 24000; -const FRAME_SAMPLES = 480; // 20ms at 24kHz - -export const REALTIME_BUFFER_PROCESSOR_NAME = "realtime-buffer-processor"; - -export const REALTIME_BUFFER_WORKLET_SOURCE = /* js */ ` -class RealtimeBufferProcessor extends AudioWorkletProcessor { - constructor() { - super(); - this._ratio = sampleRate / ${TARGET_SAMPLE_RATE}; - this._offset = 0; - this._buf = new Float32Array(${FRAME_SAMPLES}); - this._idx = 0; - } - - process(inputs) { - const input = inputs[0]?.[0]; - if (!input) return true; - - while (this._offset < input.length) { - const i = Math.floor(this._offset); - const frac = this._offset - i; - const s0 = input[i]; - const s1 = i + 1 < input.length ? input[i + 1] : s0; - this._buf[this._idx++] = s0 + frac * (s1 - s0); - - if (this._idx >= ${FRAME_SAMPLES}) { - const pcm = new Int16Array(${FRAME_SAMPLES}); - for (let j = 0; j < ${FRAME_SAMPLES}; j++) { - const s = Math.max(-1, Math.min(1, this._buf[j])); - pcm[j] = s < 0 ? s * 0x8000 : s * 0x7fff; - } - this.port.postMessage(pcm.buffer, [pcm.buffer]); - this._idx = 0; - } - this._offset += this._ratio; - } - this._offset -= input.length; - return true; - } -} - -registerProcessor('${REALTIME_BUFFER_PROCESSOR_NAME}', RealtimeBufferProcessor); -`; - -/** Create a blob URL that can be passed to `audioWorklet.addModule()`. */ -export function createWorkletBlobUrl(): string { - return URL.createObjectURL( - new Blob([REALTIME_BUFFER_WORKLET_SOURCE], { - type: "application/javascript", - }), - ); -} diff --git a/desktop/src/features/dictation/ui/DictationButton.tsx b/desktop/src/features/dictation/ui/DictationButton.tsx index 387a0ee32..8f23d904b 100644 --- a/desktop/src/features/dictation/ui/DictationButton.tsx +++ b/desktop/src/features/dictation/ui/DictationButton.tsx @@ -2,6 +2,7 @@ import { Mic } from "lucide-react"; import { Button } from "@/shared/ui/button"; import { Tooltip, TooltipContent, TooltipTrigger } from "@/shared/ui/tooltip"; import { cn } from "@/shared/lib/cn"; +import { isMacPlatform } from "@/shared/lib/platform"; interface DictationState { isEnabled: boolean; @@ -22,25 +23,19 @@ export function DictationButton({ }: DictationButtonProps) { if (!dictation.isEnabled) return null; + const shortcutHint = isMacPlatform() ? "⌘D" : "Ctrl+D"; + const tooltipText = dictation.isRecording ? "Stop recording" : dictation.isTranscribing ? "Transcribing…" - : "Dictate message"; - - // Allow the stop action whenever the mic is live (isRecording), even if - // the session setup is still in progress (isStarting) or the composer is - // disabled. Only block the button when idle + disabled, or when startup - // hasn't captured the mic yet (isStarting && !isRecording). - const isDisabled = dictation.isRecording - ? false - : disabled || dictation.isStarting; + : null; return ( - {tooltipText} + + {tooltipText ?? ( + + Voice Dictation + + {shortcutHint} + + + )} + ); } diff --git a/desktop/src/shared/lib/keyboard-shortcuts.ts b/desktop/src/shared/lib/keyboard-shortcuts.ts index 6a67f631a..8887ec4e1 100644 --- a/desktop/src/shared/lib/keyboard-shortcuts.ts +++ b/desktop/src/shared/lib/keyboard-shortcuts.ts @@ -181,6 +181,14 @@ export const KEYBOARD_SHORTCUTS: KeyboardShortcut[] = [ keysWindows: "Ctrl+Space", category: "Messages", }, + { + id: "voice-dictation", + label: "Voice Dictation", + description: "Start or stop voice dictation in the composer", + keys: "⌘D", + keysWindows: "Ctrl+D", + category: "Messages", + }, // Formatting {