mirror of
https://github.com/block/buzz.git
synced 2026-08-18 06:50:31 +02:00
refactor(dictation): drop OpenAI Realtime path, add ⌘D shortcut
- Remove useRealtimeDictation, realtimeAudio, realtimeBufferWorklet, and transcribeSession API — dictation is now local-only (Parakeet). - useDictation simplified to use only useLocalDictation. - Add ⌘D (Ctrl+D on Windows) global shortcut to toggle dictation. - DictationButton tooltip shows 'Voice Dictation ⌘D' with kbd styling matching the search bar's ⌘K pattern. - Shortcut registered in keyboard-shortcuts.ts settings list. - AppShell dispatches custom event; useComposerDictation listens.
This commit is contained in:
@@ -595,6 +595,12 @@ export function AppShell() {
|
||||
return;
|
||||
}
|
||||
|
||||
if (key === "d" && !event.shiftKey) {
|
||||
event.preventDefault();
|
||||
window.dispatchEvent(new CustomEvent("buzz:toggle-dictation"));
|
||||
return;
|
||||
}
|
||||
|
||||
if (key === "a" && event.shiftKey) {
|
||||
event.preventDefault();
|
||||
void goHome();
|
||||
|
||||
@@ -1,76 +0,0 @@
|
||||
import { getRelayHttpUrl, signRelayEvent } from "@/shared/api/tauri";
|
||||
|
||||
export interface TranscribeStatus {
|
||||
configured: boolean;
|
||||
model: string;
|
||||
}
|
||||
|
||||
export interface TranscribeConnectResponse {
|
||||
sdp: string;
|
||||
model: string;
|
||||
}
|
||||
|
||||
/** NIP-98 event kind for HTTP request authorization. */
|
||||
const NIP98_KIND = 27235;
|
||||
|
||||
/**
|
||||
* Build a NIP-98 `Authorization: Nostr <base64>` header for an HTTP request.
|
||||
*
|
||||
* The relay verifies the signed event's `u` tag against its own
|
||||
* host-derived expected URL, so `url` must be the exact absolute URL being
|
||||
* fetched (scheme + host + path). The `method` tag must match the request.
|
||||
*/
|
||||
async function nip98AuthHeader(url: string, method: string): Promise<string> {
|
||||
const nonce = crypto.randomUUID();
|
||||
const event = await signRelayEvent({
|
||||
kind: NIP98_KIND,
|
||||
content: "",
|
||||
tags: [
|
||||
["u", url],
|
||||
["method", method],
|
||||
["nonce", nonce],
|
||||
],
|
||||
});
|
||||
const json = JSON.stringify(event);
|
||||
// btoa needs a binary string; encode UTF-8 first so non-ASCII survives.
|
||||
const base64 = btoa(String.fromCharCode(...new TextEncoder().encode(json)));
|
||||
return `Nostr ${base64}`;
|
||||
}
|
||||
|
||||
export async function getTranscribeStatus(): Promise<TranscribeStatus> {
|
||||
const baseUrl = await getRelayHttpUrl();
|
||||
const url = `${baseUrl}/transcribe/status`;
|
||||
const response = await fetch(url, {
|
||||
headers: { Authorization: await nip98AuthHeader(url, "GET") },
|
||||
});
|
||||
if (!response.ok) {
|
||||
throw new Error(`Transcribe status check failed: ${response.status}`);
|
||||
}
|
||||
return response.json();
|
||||
}
|
||||
|
||||
/**
|
||||
* Mint an OpenAI Realtime session and complete the WebRTC SDP exchange in a
|
||||
* single relay round-trip. The relay holds the OpenAI bearer token server-side
|
||||
* — the client never sees it. This also works correctly across multiple relay
|
||||
* replicas since no server-side session state is needed between requests.
|
||||
*/
|
||||
export async function transcribeConnect(
|
||||
sdp: string,
|
||||
): Promise<TranscribeConnectResponse> {
|
||||
const baseUrl = await getRelayHttpUrl();
|
||||
const url = `${baseUrl}/transcribe/connect`;
|
||||
const response = await fetch(url, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
Authorization: await nip98AuthHeader(url, "POST"),
|
||||
},
|
||||
body: JSON.stringify({ sdp }),
|
||||
});
|
||||
if (!response.ok) {
|
||||
const body = await response.text().catch(() => "");
|
||||
throw new Error(`Transcribe connect failed (${response.status}): ${body}`);
|
||||
}
|
||||
return response.json();
|
||||
}
|
||||
@@ -64,12 +64,23 @@ export function useComposerDictation({
|
||||
|
||||
// Auto-cancel dictation when the composer becomes disabled mid-recording
|
||||
// (e.g. channel becomes read-only, parent send state disables thread composer).
|
||||
// Without this, the WebRTC session keeps running with no way to stop it.
|
||||
// Without this, the STT session keeps running with no way to stop it.
|
||||
useEffect(() => {
|
||||
if (disabled && dictation.isRecording) {
|
||||
dictation.cancelRecording();
|
||||
}
|
||||
}, [disabled, dictation.isRecording, dictation.cancelRecording]);
|
||||
|
||||
// ⌘D global shortcut — dispatched from AppShell's keydown handler.
|
||||
useEffect(() => {
|
||||
function handleToggle() {
|
||||
dictation.toggleRecording();
|
||||
}
|
||||
window.addEventListener("buzz:toggle-dictation", handleToggle);
|
||||
return () => {
|
||||
window.removeEventListener("buzz:toggle-dictation", handleToggle);
|
||||
};
|
||||
}, [dictation.toggleRecording]);
|
||||
|
||||
return dictation;
|
||||
}
|
||||
|
||||
@@ -7,7 +7,6 @@ import {
|
||||
replaceTrailingTranscribedText,
|
||||
} from "../lib/voiceInput";
|
||||
import { useLocalDictation } from "./useLocalDictation";
|
||||
import { useRealtimeDictation } from "./useRealtimeDictation";
|
||||
|
||||
interface UseDictationOptions {
|
||||
/** Returns the current composer text (must be fresh — synced from editor). */
|
||||
@@ -71,25 +70,13 @@ export function useDictation({
|
||||
[autoSubmitPhrases, getText, onSend, isSendBlockedRef, setText],
|
||||
);
|
||||
|
||||
// Try local STT first (offline, no API key needed).
|
||||
const localDictation = useLocalDictation({
|
||||
const dictation = useLocalDictation({
|
||||
onRecordingStart: () => {
|
||||
lastTranscriptRef.current = "";
|
||||
},
|
||||
onTranscriptText: handleTranscript,
|
||||
});
|
||||
|
||||
// Fall back to cloud (OpenAI Realtime) if local is unavailable.
|
||||
const cloudDictation = useRealtimeDictation({
|
||||
disabled: localDictation.isEnabled, // Disable cloud when local is available
|
||||
onRecordingStart: () => {
|
||||
lastTranscriptRef.current = "";
|
||||
},
|
||||
onTranscriptText: handleTranscript,
|
||||
});
|
||||
|
||||
// Use whichever is enabled — local takes priority.
|
||||
const dictation = localDictation.isEnabled ? localDictation : cloudDictation;
|
||||
stopRecordingRef.current = dictation.stopRecording;
|
||||
|
||||
return dictation;
|
||||
|
||||
@@ -1,363 +0,0 @@
|
||||
import { useCallback, useEffect, useRef, useState } from "react";
|
||||
import { toast } from "sonner";
|
||||
import { getTranscribeStatus } from "../api/transcribeSession";
|
||||
import {
|
||||
type AudioBufferCapture,
|
||||
type TranscriptEvent,
|
||||
type TranscriptSegmentState,
|
||||
BUFFER_COMMITTED_EVENT,
|
||||
TRANSCRIPT_COMPLETED_EVENT,
|
||||
TRANSCRIPT_DELTA_EVENT,
|
||||
TRANSCRIPT_FAILED_EVENT,
|
||||
commitAudioBuffer,
|
||||
connectPeerConnection,
|
||||
createAudioBufferCapture,
|
||||
createPeerConnection,
|
||||
createTranscriptSegmentState,
|
||||
flushAudioBuffer,
|
||||
getTranscriptText,
|
||||
mergeTranscriptEvent,
|
||||
requiresManualCommit,
|
||||
} from "../lib/realtimeAudio";
|
||||
|
||||
interface UseRealtimeDictationOptions {
|
||||
disabled?: boolean;
|
||||
onRecordingStart?: () => void;
|
||||
onTranscriptText: (text: string) => void;
|
||||
}
|
||||
|
||||
function closeResources(resources: {
|
||||
audioCapture?: AudioBufferCapture | null;
|
||||
dataChannel?: RTCDataChannel | null;
|
||||
peerConnection?: RTCPeerConnection | null;
|
||||
stream?: MediaStream | null;
|
||||
}) {
|
||||
resources.audioCapture?.close();
|
||||
resources.dataChannel?.close();
|
||||
resources.peerConnection?.close();
|
||||
for (const track of resources.stream?.getTracks() ?? []) {
|
||||
track.stop();
|
||||
}
|
||||
}
|
||||
|
||||
export function useRealtimeDictation({
|
||||
disabled = false,
|
||||
onRecordingStart,
|
||||
onTranscriptText,
|
||||
}: UseRealtimeDictationOptions) {
|
||||
const [isRecording, setIsRecording] = useState(false);
|
||||
const [isStarting, setIsStarting] = useState(false);
|
||||
const [isTranscribing, setIsTranscribing] = useState(false);
|
||||
const [isConfigured, setIsConfigured] = useState(false);
|
||||
|
||||
const peerConnectionRef = useRef<RTCPeerConnection | null>(null);
|
||||
const dataChannelRef = useRef<RTCDataChannel | null>(null);
|
||||
const streamRef = useRef<MediaStream | null>(null);
|
||||
const audioCaptureRef = useRef<AudioBufferCapture | null>(null);
|
||||
const segmentStateRef = useRef<TranscriptSegmentState>(
|
||||
createTranscriptSegmentState(),
|
||||
);
|
||||
const activeRunIdRef = useRef(0);
|
||||
const manualCommitRef = useRef(false);
|
||||
const commitIntervalRef = useRef<ReturnType<typeof setInterval> | null>(null);
|
||||
const onRecordingStartRef = useRef(onRecordingStart);
|
||||
const onTranscriptTextRef = useRef(onTranscriptText);
|
||||
|
||||
onRecordingStartRef.current = onRecordingStart;
|
||||
onTranscriptTextRef.current = onTranscriptText;
|
||||
|
||||
const isEnabled = !disabled && isConfigured;
|
||||
|
||||
// Check if transcription is configured on mount
|
||||
useEffect(() => {
|
||||
let cancelled = false;
|
||||
getTranscribeStatus()
|
||||
.then((status) => {
|
||||
if (!cancelled) setIsConfigured(status.configured);
|
||||
})
|
||||
.catch(() => {
|
||||
if (!cancelled) setIsConfigured(false);
|
||||
});
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
}, []);
|
||||
|
||||
/**
|
||||
* Tear down recording resources.
|
||||
*
|
||||
* @param invalidateRun - When true, immediately marks the run stale so
|
||||
* late transcript events are rejected (used by send/edit-save/navigation).
|
||||
* When false (user-initiated stop), the run stays valid during the grace
|
||||
* window so the final commit's transcript is delivered to the composer.
|
||||
*/
|
||||
const cleanupResources = useCallback((invalidateRun = true) => {
|
||||
// Clear periodic commit interval if active.
|
||||
if (commitIntervalRef.current) {
|
||||
clearInterval(commitIntervalRef.current);
|
||||
commitIntervalRef.current = null;
|
||||
}
|
||||
|
||||
// For manual-commit models (no server VAD), commit any buffered audio
|
||||
// before tearing down the connection so OpenAI processes the final chunk.
|
||||
// We keep the data channel open briefly to receive the transcript response.
|
||||
const dc = dataChannelRef.current;
|
||||
const needsCommit =
|
||||
manualCommitRef.current && dc && dc.readyState === "open";
|
||||
manualCommitRef.current = false;
|
||||
|
||||
// Send final commit for manual-commit models.
|
||||
if (needsCommit && dc) {
|
||||
commitAudioBuffer(dc);
|
||||
}
|
||||
|
||||
// Determine whether we need a grace window before full teardown.
|
||||
// User-initiated stop (!invalidateRun) needs a grace period for BOTH:
|
||||
// - Manual-commit models: final commit's transcript response
|
||||
// - Server-VAD models: in-flight VAD completion events
|
||||
const needsGrace = !invalidateRun && dc && dc.readyState === "open";
|
||||
|
||||
if (needsGrace && dc) {
|
||||
// Stop the mic immediately so no new audio is sent.
|
||||
for (const track of streamRef.current?.getTracks() ?? []) {
|
||||
track.stop();
|
||||
}
|
||||
streamRef.current = null;
|
||||
audioCaptureRef.current?.close();
|
||||
audioCaptureRef.current = null;
|
||||
// Delay WebRTC teardown briefly so final transcript events arrive.
|
||||
const pc = peerConnectionRef.current;
|
||||
peerConnectionRef.current = null;
|
||||
dataChannelRef.current = null;
|
||||
const runId = activeRunIdRef.current;
|
||||
|
||||
setTimeout(() => {
|
||||
// After the grace window, invalidate the run and tear down.
|
||||
if (activeRunIdRef.current === runId) {
|
||||
activeRunIdRef.current += 1;
|
||||
}
|
||||
dc.close();
|
||||
pc?.close();
|
||||
}, 3000);
|
||||
return;
|
||||
}
|
||||
|
||||
// Immediate teardown: send/navigation cancellation, or no open channel.
|
||||
activeRunIdRef.current += 1;
|
||||
closeResources({
|
||||
audioCapture: audioCaptureRef.current,
|
||||
dataChannel: dataChannelRef.current,
|
||||
peerConnection: peerConnectionRef.current,
|
||||
stream: streamRef.current,
|
||||
});
|
||||
audioCaptureRef.current = null;
|
||||
dataChannelRef.current = null;
|
||||
peerConnectionRef.current = null;
|
||||
streamRef.current = null;
|
||||
}, []);
|
||||
|
||||
/** Full cleanup — invalidates the run (used by send/navigation). */
|
||||
const cleanup = useCallback(() => {
|
||||
cleanupResources(true);
|
||||
setIsRecording(false);
|
||||
setIsStarting(false);
|
||||
setIsTranscribing(false);
|
||||
}, [cleanupResources]);
|
||||
|
||||
/** User-initiated stop — preserves final transcript for manual-commit models. */
|
||||
const userStop = useCallback(() => {
|
||||
cleanupResources(false);
|
||||
setIsRecording(false);
|
||||
setIsStarting(false);
|
||||
// Note: isTranscribing stays true briefly while the final transcript arrives.
|
||||
}, [cleanupResources]);
|
||||
|
||||
useEffect(() => cleanupResources, [cleanupResources]);
|
||||
|
||||
const handleRealtimeEvent = useCallback(
|
||||
(runId: number, event: TranscriptEvent) => {
|
||||
// Ignore events from a stale run (e.g. user sent/stopped while
|
||||
// transcripts were still in-flight from the data channel).
|
||||
if (activeRunIdRef.current !== runId) return;
|
||||
|
||||
if (event.type === "error") {
|
||||
console.error("OpenAI realtime server error", event);
|
||||
toast.error(event.error?.message ?? "Voice input error");
|
||||
return;
|
||||
}
|
||||
|
||||
if (event.type === TRANSCRIPT_FAILED_EVENT) {
|
||||
console.warn("OpenAI transcription failed for item", event);
|
||||
setIsTranscribing(false);
|
||||
toast.error(event.error?.message ?? "Transcription failed");
|
||||
return;
|
||||
}
|
||||
|
||||
if (
|
||||
event.type !== TRANSCRIPT_DELTA_EVENT &&
|
||||
event.type !== TRANSCRIPT_COMPLETED_EVENT &&
|
||||
event.type !== BUFFER_COMMITTED_EVENT
|
||||
) {
|
||||
return;
|
||||
}
|
||||
|
||||
const prevText = getTranscriptText(segmentStateRef.current);
|
||||
const merged = mergeTranscriptEvent(segmentStateRef.current, event);
|
||||
|
||||
// Always update transcribing state — a completed event must clear the
|
||||
// indicator even when the final text matches the accumulated deltas.
|
||||
const stillTranscribing = event.type !== TRANSCRIPT_COMPLETED_EVENT;
|
||||
setIsTranscribing(stillTranscribing);
|
||||
|
||||
if (merged === prevText) return;
|
||||
onTranscriptTextRef.current(merged);
|
||||
},
|
||||
[],
|
||||
);
|
||||
|
||||
const startRecording = useCallback(async () => {
|
||||
if (!isEnabled || isStarting || isRecording) return;
|
||||
|
||||
const runId = activeRunIdRef.current + 1;
|
||||
activeRunIdRef.current = runId;
|
||||
const isStaleRun = () => activeRunIdRef.current !== runId;
|
||||
|
||||
let stream: MediaStream | null = null;
|
||||
let audioCapture: AudioBufferCapture | null = null;
|
||||
let peerConnection: RTCPeerConnection | null = null;
|
||||
let dataChannel: RTCDataChannel | null = null;
|
||||
|
||||
setIsStarting(true);
|
||||
segmentStateRef.current = createTranscriptSegmentState();
|
||||
onRecordingStartRef.current?.();
|
||||
|
||||
try {
|
||||
// 1. Capture mic immediately for instant feedback
|
||||
stream = await navigator.mediaDevices.getUserMedia({
|
||||
audio: {
|
||||
autoGainControl: true,
|
||||
echoCancellation: true,
|
||||
noiseSuppression: true,
|
||||
},
|
||||
});
|
||||
if (isStaleRun()) {
|
||||
closeResources({ stream });
|
||||
return;
|
||||
}
|
||||
streamRef.current = stream;
|
||||
setIsRecording(true);
|
||||
|
||||
// 2. Buffer PCM via AudioWorklet while network calls proceed
|
||||
audioCapture = await createAudioBufferCapture(stream);
|
||||
if (isStaleRun()) {
|
||||
closeResources({ audioCapture, stream });
|
||||
return;
|
||||
}
|
||||
audioCaptureRef.current = audioCapture;
|
||||
|
||||
// 3. Set up WebRTC peer connection
|
||||
peerConnection = createPeerConnection();
|
||||
peerConnectionRef.current = peerConnection;
|
||||
const activeStream = stream;
|
||||
stream.getAudioTracks().forEach((track) => {
|
||||
peerConnection?.addTrack(track, activeStream);
|
||||
});
|
||||
|
||||
dataChannel = peerConnection.createDataChannel("oai-events");
|
||||
dataChannelRef.current = dataChannel;
|
||||
dataChannel.addEventListener("message", (message) => {
|
||||
try {
|
||||
handleRealtimeEvent(runId, JSON.parse(String(message.data)));
|
||||
} catch {
|
||||
// Ignore non-JSON events
|
||||
}
|
||||
});
|
||||
|
||||
// Flush buffered audio once data channel opens
|
||||
const channelToFlush = dataChannel;
|
||||
const captureToFlush = audioCapture;
|
||||
dataChannel.addEventListener("open", () => {
|
||||
if (isStaleRun()) {
|
||||
captureToFlush.close();
|
||||
return;
|
||||
}
|
||||
flushAudioBuffer(channelToFlush, captureToFlush.chunks);
|
||||
if (manualCommitRef.current) {
|
||||
commitAudioBuffer(channelToFlush);
|
||||
commitIntervalRef.current = setInterval(() => {
|
||||
if (channelToFlush.readyState === "open") {
|
||||
commitAudioBuffer(channelToFlush);
|
||||
}
|
||||
}, 2000);
|
||||
}
|
||||
captureToFlush.close();
|
||||
audioCaptureRef.current = null;
|
||||
});
|
||||
|
||||
// 4. Session creation + SDP exchange in a single relay round-trip.
|
||||
// No cross-replica state needed — works in HA deployments.
|
||||
const { model } = await connectPeerConnection({ peerConnection });
|
||||
if (isStaleRun()) {
|
||||
closeResources({ audioCapture, dataChannel, peerConnection, stream });
|
||||
return;
|
||||
}
|
||||
manualCommitRef.current = requiresManualCommit(model);
|
||||
} catch (error) {
|
||||
closeResources({ audioCapture, dataChannel, peerConnection, stream });
|
||||
if (!isStaleRun()) {
|
||||
audioCaptureRef.current = null;
|
||||
dataChannelRef.current = null;
|
||||
peerConnectionRef.current = null;
|
||||
streamRef.current = null;
|
||||
setIsRecording(false);
|
||||
setIsTranscribing(false);
|
||||
|
||||
const message =
|
||||
error instanceof Error ? error.message : "Voice input failed";
|
||||
if (/not allowed|denied|permission/i.test(message)) {
|
||||
toast.error("Microphone access denied", {
|
||||
description:
|
||||
"Allow microphone access in System Settings to use dictation.",
|
||||
});
|
||||
} else if (/not found|no audio/i.test(message)) {
|
||||
toast.error("No microphone found", {
|
||||
description: "Connect a microphone and try again.",
|
||||
});
|
||||
} else {
|
||||
toast.error("Voice input failed", { description: message });
|
||||
}
|
||||
}
|
||||
} finally {
|
||||
if (!isStaleRun()) setIsStarting(false);
|
||||
}
|
||||
}, [handleRealtimeEvent, isEnabled, isRecording, isStarting]);
|
||||
|
||||
/** User-initiated stop (mic button) — preserves final transcript. */
|
||||
const stopRecording = useCallback(() => userStop(), [userStop]);
|
||||
|
||||
/**
|
||||
* Cancel recording and reject all pending transcripts. Used by send,
|
||||
* edit-save, and navigation to prevent late events from refilling the
|
||||
* composer after the content has been dispatched.
|
||||
*/
|
||||
const cancelRecording = useCallback(() => cleanup(), [cleanup]);
|
||||
|
||||
const toggleRecording = useCallback(() => {
|
||||
if (isRecording || isStarting) {
|
||||
stopRecording();
|
||||
return;
|
||||
}
|
||||
void startRecording();
|
||||
}, [isRecording, isStarting, startRecording, stopRecording]);
|
||||
|
||||
return {
|
||||
isEnabled,
|
||||
isRecording,
|
||||
isStarting,
|
||||
isTranscribing,
|
||||
startRecording,
|
||||
stopRecording,
|
||||
cancelRecording,
|
||||
toggleRecording,
|
||||
};
|
||||
}
|
||||
@@ -1,334 +0,0 @@
|
||||
import { describe, it } from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
|
||||
// Inline the logic to keep the test self-contained and avoid bundler issues.
|
||||
const TRANSCRIPT_DELTA_EVENT =
|
||||
"conversation.item.input_audio_transcription.delta";
|
||||
const TRANSCRIPT_COMPLETED_EVENT =
|
||||
"conversation.item.input_audio_transcription.completed";
|
||||
const BUFFER_COMMITTED_EVENT = "input_audio_buffer.committed";
|
||||
|
||||
function createTranscriptSegmentState() {
|
||||
return { itemOrder: [], items: new Map() };
|
||||
}
|
||||
|
||||
function getOrCreateItem(state, itemId, previousItemId) {
|
||||
let seg = state.items.get(itemId);
|
||||
if (!seg) {
|
||||
seg = { pending: "", finalized: null };
|
||||
state.items.set(itemId, seg);
|
||||
if (previousItemId) {
|
||||
const prevIndex = state.itemOrder.indexOf(previousItemId);
|
||||
if (prevIndex !== -1) {
|
||||
state.itemOrder.splice(prevIndex + 1, 0, itemId);
|
||||
} else {
|
||||
state.itemOrder.push(itemId);
|
||||
}
|
||||
} else {
|
||||
state.itemOrder.push(itemId);
|
||||
}
|
||||
}
|
||||
return seg;
|
||||
}
|
||||
|
||||
function getTranscriptText(state) {
|
||||
let result = "";
|
||||
for (const id of state.itemOrder) {
|
||||
const seg = state.items.get(id);
|
||||
if (!seg) continue;
|
||||
const text = seg.finalized ?? seg.pending;
|
||||
if (!text) continue;
|
||||
if (result && !result.endsWith(" ") && !text.startsWith(" ")) {
|
||||
result += " ";
|
||||
}
|
||||
result += text;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
function mergeTranscriptEvent(state, event) {
|
||||
const itemId = event.item_id ?? "__default__";
|
||||
|
||||
if (event.type === BUFFER_COMMITTED_EVENT) {
|
||||
getOrCreateItem(state, itemId, event.previous_item_id ?? undefined);
|
||||
} else if (event.type === TRANSCRIPT_DELTA_EVENT) {
|
||||
const seg = getOrCreateItem(state, itemId);
|
||||
const delta = event.delta ?? "";
|
||||
if (delta) {
|
||||
seg.pending += delta;
|
||||
}
|
||||
} else if (event.type === TRANSCRIPT_COMPLETED_EVENT) {
|
||||
const seg = getOrCreateItem(state, itemId);
|
||||
seg.finalized = event.transcript ?? "";
|
||||
}
|
||||
|
||||
return getTranscriptText(state);
|
||||
}
|
||||
|
||||
describe("mergeTranscriptEvent", () => {
|
||||
it("accumulates delta events for a single item", () => {
|
||||
const state = createTranscriptSegmentState();
|
||||
const r1 = mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_DELTA_EVENT,
|
||||
item_id: "item_1",
|
||||
delta: "hello ",
|
||||
});
|
||||
assert.equal(r1, "hello ");
|
||||
|
||||
const r2 = mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_DELTA_EVENT,
|
||||
item_id: "item_1",
|
||||
delta: "world",
|
||||
});
|
||||
assert.equal(r2, "hello world");
|
||||
});
|
||||
|
||||
it("replaces deltas with finalized text on completed event", () => {
|
||||
const state = createTranscriptSegmentState();
|
||||
mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_DELTA_EVENT,
|
||||
item_id: "item_1",
|
||||
delta: "hello world",
|
||||
});
|
||||
|
||||
const result = mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_COMPLETED_EVENT,
|
||||
item_id: "item_1",
|
||||
transcript: "Hello, world.",
|
||||
});
|
||||
assert.equal(result, "Hello, world.");
|
||||
});
|
||||
|
||||
it("handles multiple items in order", () => {
|
||||
const state = createTranscriptSegmentState();
|
||||
|
||||
// First item
|
||||
mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_DELTA_EVENT,
|
||||
item_id: "item_1",
|
||||
delta: "first ",
|
||||
});
|
||||
mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_COMPLETED_EVENT,
|
||||
item_id: "item_1",
|
||||
transcript: "First.",
|
||||
});
|
||||
|
||||
// Second item
|
||||
mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_DELTA_EVENT,
|
||||
item_id: "item_2",
|
||||
delta: "second",
|
||||
});
|
||||
assert.equal(state.items.get("item_2").pending, "second");
|
||||
|
||||
const result = mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_COMPLETED_EVENT,
|
||||
item_id: "item_2",
|
||||
transcript: "Second.",
|
||||
});
|
||||
assert.equal(result, "First. Second.");
|
||||
});
|
||||
|
||||
it("handles out-of-order completed events by item id", () => {
|
||||
const state = createTranscriptSegmentState();
|
||||
|
||||
// Both items start with deltas
|
||||
mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_DELTA_EVENT,
|
||||
item_id: "item_1",
|
||||
delta: "first",
|
||||
});
|
||||
mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_DELTA_EVENT,
|
||||
item_id: "item_2",
|
||||
delta: "second",
|
||||
});
|
||||
|
||||
// item_2 completes before item_1
|
||||
mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_COMPLETED_EVENT,
|
||||
item_id: "item_2",
|
||||
transcript: "Second.",
|
||||
});
|
||||
|
||||
// item_1 still shows pending
|
||||
let result = mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_DELTA_EVENT,
|
||||
item_id: "item_1",
|
||||
delta: " more",
|
||||
});
|
||||
assert.equal(result, "first more Second.");
|
||||
|
||||
// item_1 finally completes
|
||||
result = mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_COMPLETED_EVENT,
|
||||
item_id: "item_1",
|
||||
transcript: "First more.",
|
||||
});
|
||||
assert.equal(result, "First more. Second.");
|
||||
});
|
||||
|
||||
it("does not duplicate text on completed event", () => {
|
||||
const state = createTranscriptSegmentState();
|
||||
|
||||
mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_DELTA_EVENT,
|
||||
item_id: "item_1",
|
||||
delta: "hello world",
|
||||
});
|
||||
|
||||
const result = mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_COMPLETED_EVENT,
|
||||
item_id: "item_1",
|
||||
transcript: "Hello, world.",
|
||||
});
|
||||
assert.equal(result, "Hello, world.");
|
||||
});
|
||||
|
||||
it("falls back to __default__ when item_id is missing", () => {
|
||||
const state = createTranscriptSegmentState();
|
||||
mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_DELTA_EVENT,
|
||||
delta: "no id",
|
||||
});
|
||||
const result = mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_COMPLETED_EVENT,
|
||||
transcript: "No id.",
|
||||
});
|
||||
assert.equal(result, "No id.");
|
||||
});
|
||||
|
||||
it("preserves committed item order when completions arrive out of order", () => {
|
||||
const state = createTranscriptSegmentState();
|
||||
|
||||
// Server commits items in order: item_1 then item_2
|
||||
mergeTranscriptEvent(state, {
|
||||
type: BUFFER_COMMITTED_EVENT,
|
||||
item_id: "item_1",
|
||||
previous_item_id: null,
|
||||
});
|
||||
mergeTranscriptEvent(state, {
|
||||
type: BUFFER_COMMITTED_EVENT,
|
||||
item_id: "item_2",
|
||||
previous_item_id: "item_1",
|
||||
});
|
||||
|
||||
// But item_2's completion arrives first
|
||||
mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_COMPLETED_EVENT,
|
||||
item_id: "item_2",
|
||||
transcript: "Second.",
|
||||
});
|
||||
|
||||
// Then item_1's completion arrives
|
||||
const result = mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_COMPLETED_EVENT,
|
||||
item_id: "item_1",
|
||||
transcript: "First.",
|
||||
});
|
||||
|
||||
// Order must follow committed order, not arrival order
|
||||
assert.equal(result, "First. Second.");
|
||||
});
|
||||
|
||||
it("inserts item after previous_item_id even if later items already exist", () => {
|
||||
const state = createTranscriptSegmentState();
|
||||
|
||||
// Commit item_1
|
||||
mergeTranscriptEvent(state, {
|
||||
type: BUFFER_COMMITTED_EVENT,
|
||||
item_id: "item_1",
|
||||
});
|
||||
|
||||
// Commit item_3 (item_2 not yet committed)
|
||||
mergeTranscriptEvent(state, {
|
||||
type: BUFFER_COMMITTED_EVENT,
|
||||
item_id: "item_3",
|
||||
previous_item_id: "item_1",
|
||||
});
|
||||
|
||||
// Now commit item_2 between item_1 and item_3
|
||||
// (previous_item_id = item_1, so it goes after item_1)
|
||||
mergeTranscriptEvent(state, {
|
||||
type: BUFFER_COMMITTED_EVENT,
|
||||
item_id: "item_2",
|
||||
previous_item_id: "item_1",
|
||||
});
|
||||
|
||||
// Add transcripts
|
||||
mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_COMPLETED_EVENT,
|
||||
item_id: "item_1",
|
||||
transcript: "One.",
|
||||
});
|
||||
mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_COMPLETED_EVENT,
|
||||
item_id: "item_2",
|
||||
transcript: "Two.",
|
||||
});
|
||||
const result = mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_COMPLETED_EVENT,
|
||||
item_id: "item_3",
|
||||
transcript: "Three.",
|
||||
});
|
||||
|
||||
// item_2 was inserted after item_1 (before item_3)
|
||||
assert.equal(result, "One. Two. Three.");
|
||||
});
|
||||
|
||||
it("handles completion-only flow with committed order", () => {
|
||||
const state = createTranscriptSegmentState();
|
||||
|
||||
// Server commits both items
|
||||
mergeTranscriptEvent(state, {
|
||||
type: BUFFER_COMMITTED_EVENT,
|
||||
item_id: "item_A",
|
||||
});
|
||||
mergeTranscriptEvent(state, {
|
||||
type: BUFFER_COMMITTED_EVENT,
|
||||
item_id: "item_B",
|
||||
previous_item_id: "item_A",
|
||||
});
|
||||
|
||||
// Only completed events arrive (no deltas) — in reverse order
|
||||
mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_COMPLETED_EVENT,
|
||||
item_id: "item_B",
|
||||
transcript: "Bravo.",
|
||||
});
|
||||
const result = mergeTranscriptEvent(state, {
|
||||
type: TRANSCRIPT_COMPLETED_EVENT,
|
||||
item_id: "item_A",
|
||||
transcript: "Alpha.",
|
||||
});
|
||||
|
||||
assert.equal(result, "Alpha. Bravo.");
|
||||
});
|
||||
});
|
||||
|
||||
// Inline the logic to keep the test self-contained.
|
||||
function requiresManualCommit(model) {
|
||||
return model.includes("realtime-whisper");
|
||||
}
|
||||
|
||||
describe("requiresManualCommit", () => {
|
||||
it("returns true for gpt-realtime-whisper", () => {
|
||||
assert.equal(requiresManualCommit("gpt-realtime-whisper"), true);
|
||||
});
|
||||
|
||||
it("returns true for versioned realtime-whisper model", () => {
|
||||
assert.equal(
|
||||
requiresManualCommit("gpt-4o-realtime-whisper-20250512"),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
it("returns false for whisper-1", () => {
|
||||
assert.equal(requiresManualCommit("whisper-1"), false);
|
||||
});
|
||||
|
||||
it("returns false for gpt-4o-transcribe", () => {
|
||||
assert.equal(requiresManualCommit("gpt-4o-transcribe"), false);
|
||||
});
|
||||
});
|
||||
@@ -1,253 +0,0 @@
|
||||
import { transcribeConnect } from "../api/transcribeSession";
|
||||
import {
|
||||
REALTIME_BUFFER_PROCESSOR_NAME,
|
||||
createWorkletBlobUrl,
|
||||
} from "./realtimeBufferWorklet";
|
||||
|
||||
export const TRANSCRIPT_DELTA_EVENT =
|
||||
"conversation.item.input_audio_transcription.delta";
|
||||
export const TRANSCRIPT_COMPLETED_EVENT =
|
||||
"conversation.item.input_audio_transcription.completed";
|
||||
export const TRANSCRIPT_FAILED_EVENT =
|
||||
"conversation.item.input_audio_transcription.failed";
|
||||
export const BUFFER_COMMITTED_EVENT = "input_audio_buffer.committed";
|
||||
|
||||
const MAX_BUFFER_CHUNKS = 500; // ~10s at 20ms per chunk
|
||||
|
||||
export type TranscriptEvent = {
|
||||
type?: string;
|
||||
item_id?: string;
|
||||
previous_item_id?: string;
|
||||
content_index?: number;
|
||||
delta?: string;
|
||||
transcript?: string;
|
||||
message?: string;
|
||||
error?: { message?: string };
|
||||
};
|
||||
|
||||
export function createPeerConnection(): RTCPeerConnection {
|
||||
return new RTCPeerConnection();
|
||||
}
|
||||
|
||||
/**
|
||||
* Complete the WebRTC SDP exchange via the relay.
|
||||
*
|
||||
* The relay mints the OpenAI session and proxies the SDP exchange in a single
|
||||
* request — the client never sees the bearer token, and no server-side session
|
||||
* state is needed between requests (works across HA relay replicas).
|
||||
*
|
||||
* Returns the model name from the relay so the caller can configure VAD mode.
|
||||
*/
|
||||
export async function connectPeerConnection(options: {
|
||||
peerConnection: RTCPeerConnection;
|
||||
}): Promise<{ model: string }> {
|
||||
const offer = await options.peerConnection.createOffer();
|
||||
await options.peerConnection.setLocalDescription(offer);
|
||||
|
||||
const { sdp: answerSdp, model } = await transcribeConnect(offer.sdp ?? "");
|
||||
|
||||
await options.peerConnection.setRemoteDescription({
|
||||
type: "answer",
|
||||
sdp: answerSdp,
|
||||
});
|
||||
|
||||
return { model };
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-item tracking for a single transcription turn. OpenAI's Realtime API
|
||||
* sends delta and completed events tagged with an `item_id`; completed events
|
||||
* for different turns can arrive out of order, so we reconcile by item id.
|
||||
*/
|
||||
interface ItemSegment {
|
||||
/** Accumulated delta text (replaced by finalized on completion). */
|
||||
pending: string;
|
||||
/** Finalized text (set once the completed event arrives). */
|
||||
finalized: string | null;
|
||||
}
|
||||
|
||||
/**
|
||||
* State for tracking transcription segments keyed by item id.
|
||||
* Maintains insertion order so the full transcript is reconstructed
|
||||
* in the order items were first seen.
|
||||
*/
|
||||
export interface TranscriptSegmentState {
|
||||
/** Ordered item ids (insertion order = utterance order). */
|
||||
itemOrder: string[];
|
||||
/** Per-item segment data. */
|
||||
items: Map<string, ItemSegment>;
|
||||
}
|
||||
|
||||
export function createTranscriptSegmentState(): TranscriptSegmentState {
|
||||
return { itemOrder: [], items: new Map() };
|
||||
}
|
||||
|
||||
/** Get the current full transcript text from segment state. */
|
||||
export function getTranscriptText(state: TranscriptSegmentState): string {
|
||||
let result = "";
|
||||
for (const id of state.itemOrder) {
|
||||
const seg = state.items.get(id);
|
||||
if (!seg) continue;
|
||||
const text = seg.finalized ?? seg.pending;
|
||||
if (!text) continue;
|
||||
if (result && !result.endsWith(" ") && !text.startsWith(" ")) {
|
||||
result += " ";
|
||||
}
|
||||
result += text;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Internal: get or create the segment for an item.
|
||||
* When `previousItemId` is provided (from committed events), the new item is
|
||||
* inserted after that item in `itemOrder` to preserve the server's utterance
|
||||
* order — even if transcript events for a later item arrive first.
|
||||
*/
|
||||
function getOrCreateItem(
|
||||
state: TranscriptSegmentState,
|
||||
itemId: string,
|
||||
previousItemId?: string,
|
||||
): ItemSegment {
|
||||
let seg = state.items.get(itemId);
|
||||
if (!seg) {
|
||||
seg = { pending: "", finalized: null };
|
||||
state.items.set(itemId, seg);
|
||||
if (previousItemId) {
|
||||
const prevIndex = state.itemOrder.indexOf(previousItemId);
|
||||
if (prevIndex !== -1) {
|
||||
state.itemOrder.splice(prevIndex + 1, 0, itemId);
|
||||
} else {
|
||||
// Previous item not yet seen — append (best effort).
|
||||
state.itemOrder.push(itemId);
|
||||
}
|
||||
} else {
|
||||
state.itemOrder.push(itemId);
|
||||
}
|
||||
}
|
||||
return seg;
|
||||
}
|
||||
|
||||
/**
|
||||
* Merge a transcript event into the segment state, keyed by `item_id`.
|
||||
*
|
||||
* - Committed events: register the item in the correct order using
|
||||
* `previous_item_id` from the server, before any transcript arrives.
|
||||
* - Delta events: append to the item's `pending` text.
|
||||
* - Completed events: store `finalized` text, replacing accumulated deltas.
|
||||
*
|
||||
* Returns the full merged text across all items in order.
|
||||
*/
|
||||
export function mergeTranscriptEvent(
|
||||
state: TranscriptSegmentState,
|
||||
event: TranscriptEvent,
|
||||
): string {
|
||||
// Use item_id from the event; fall back to a synthetic key for events
|
||||
// that lack one (shouldn't happen in practice, but be defensive).
|
||||
const itemId = event.item_id ?? "__default__";
|
||||
|
||||
if (event.type === BUFFER_COMMITTED_EVENT) {
|
||||
// Register the item in the correct position before transcripts arrive.
|
||||
getOrCreateItem(state, itemId, event.previous_item_id ?? undefined);
|
||||
} else if (event.type === TRANSCRIPT_DELTA_EVENT) {
|
||||
const seg = getOrCreateItem(state, itemId);
|
||||
const delta = event.delta ?? "";
|
||||
if (delta) {
|
||||
seg.pending += delta;
|
||||
}
|
||||
} else if (event.type === TRANSCRIPT_COMPLETED_EVENT) {
|
||||
const seg = getOrCreateItem(state, itemId);
|
||||
seg.finalized = event.transcript ?? "";
|
||||
}
|
||||
|
||||
// Reconstruct full text from all items in order.
|
||||
return getTranscriptText(state);
|
||||
}
|
||||
|
||||
// ── Audio buffer capture ──────────────────────────────────────────────────
|
||||
|
||||
export interface AudioBufferCapture {
|
||||
chunks: Int16Array[];
|
||||
close(): void;
|
||||
}
|
||||
|
||||
export async function createAudioBufferCapture(
|
||||
stream: MediaStream,
|
||||
): Promise<AudioBufferCapture> {
|
||||
const audioContext = new AudioContext();
|
||||
const blobUrl = createWorkletBlobUrl();
|
||||
try {
|
||||
await audioContext.audioWorklet.addModule(blobUrl);
|
||||
} finally {
|
||||
URL.revokeObjectURL(blobUrl);
|
||||
}
|
||||
|
||||
const source = audioContext.createMediaStreamSource(stream);
|
||||
const worklet = new AudioWorkletNode(
|
||||
audioContext,
|
||||
REALTIME_BUFFER_PROCESSOR_NAME,
|
||||
);
|
||||
source.connect(worklet);
|
||||
worklet.connect(audioContext.destination);
|
||||
|
||||
const chunks: Int16Array[] = [];
|
||||
worklet.port.onmessage = (event: MessageEvent<ArrayBuffer>) => {
|
||||
if (chunks.length < MAX_BUFFER_CHUNKS) {
|
||||
chunks.push(new Int16Array(event.data));
|
||||
}
|
||||
};
|
||||
|
||||
return {
|
||||
chunks,
|
||||
close() {
|
||||
worklet.disconnect();
|
||||
source.disconnect();
|
||||
void audioContext.close();
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// ── Flush buffered PCM into the data channel ──────────────────────────────
|
||||
|
||||
function int16ToBase64(pcm: Int16Array): string {
|
||||
const bytes = new Uint8Array(pcm.buffer, pcm.byteOffset, pcm.byteLength);
|
||||
let binary = "";
|
||||
for (let i = 0; i < bytes.length; i++) {
|
||||
binary += String.fromCharCode(bytes[i]);
|
||||
}
|
||||
return btoa(binary);
|
||||
}
|
||||
|
||||
export function flushAudioBuffer(
|
||||
dataChannel: RTCDataChannel,
|
||||
chunks: Int16Array[],
|
||||
): void {
|
||||
for (const chunk of chunks) {
|
||||
dataChannel.send(
|
||||
JSON.stringify({
|
||||
type: "input_audio_buffer.append",
|
||||
audio: int16ToBase64(chunk),
|
||||
}),
|
||||
);
|
||||
}
|
||||
chunks.length = 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Send `input_audio_buffer.commit` to finalize buffered audio for transcription.
|
||||
*
|
||||
* Required when server VAD is disabled (e.g. `realtime-whisper` models) —
|
||||
* without a commit, appended audio is never processed. For models using
|
||||
* server VAD, the server commits automatically on speech boundaries.
|
||||
*/
|
||||
export function commitAudioBuffer(dataChannel: RTCDataChannel): void {
|
||||
dataChannel.send(JSON.stringify({ type: "input_audio_buffer.commit" }));
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether a transcription model requires manual audio commit (no server VAD).
|
||||
* Models containing "realtime-whisper" use manual commit per OpenAI guidance.
|
||||
*/
|
||||
export function requiresManualCommit(model: string): boolean {
|
||||
return model.includes("realtime-whisper");
|
||||
}
|
||||
@@ -1,53 +0,0 @@
|
||||
const TARGET_SAMPLE_RATE = 24000;
|
||||
const FRAME_SAMPLES = 480; // 20ms at 24kHz
|
||||
|
||||
export const REALTIME_BUFFER_PROCESSOR_NAME = "realtime-buffer-processor";
|
||||
|
||||
export const REALTIME_BUFFER_WORKLET_SOURCE = /* js */ `
|
||||
class RealtimeBufferProcessor extends AudioWorkletProcessor {
|
||||
constructor() {
|
||||
super();
|
||||
this._ratio = sampleRate / ${TARGET_SAMPLE_RATE};
|
||||
this._offset = 0;
|
||||
this._buf = new Float32Array(${FRAME_SAMPLES});
|
||||
this._idx = 0;
|
||||
}
|
||||
|
||||
process(inputs) {
|
||||
const input = inputs[0]?.[0];
|
||||
if (!input) return true;
|
||||
|
||||
while (this._offset < input.length) {
|
||||
const i = Math.floor(this._offset);
|
||||
const frac = this._offset - i;
|
||||
const s0 = input[i];
|
||||
const s1 = i + 1 < input.length ? input[i + 1] : s0;
|
||||
this._buf[this._idx++] = s0 + frac * (s1 - s0);
|
||||
|
||||
if (this._idx >= ${FRAME_SAMPLES}) {
|
||||
const pcm = new Int16Array(${FRAME_SAMPLES});
|
||||
for (let j = 0; j < ${FRAME_SAMPLES}; j++) {
|
||||
const s = Math.max(-1, Math.min(1, this._buf[j]));
|
||||
pcm[j] = s < 0 ? s * 0x8000 : s * 0x7fff;
|
||||
}
|
||||
this.port.postMessage(pcm.buffer, [pcm.buffer]);
|
||||
this._idx = 0;
|
||||
}
|
||||
this._offset += this._ratio;
|
||||
}
|
||||
this._offset -= input.length;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
registerProcessor('${REALTIME_BUFFER_PROCESSOR_NAME}', RealtimeBufferProcessor);
|
||||
`;
|
||||
|
||||
/** Create a blob URL that can be passed to `audioWorklet.addModule()`. */
|
||||
export function createWorkletBlobUrl(): string {
|
||||
return URL.createObjectURL(
|
||||
new Blob([REALTIME_BUFFER_WORKLET_SOURCE], {
|
||||
type: "application/javascript",
|
||||
}),
|
||||
);
|
||||
}
|
||||
@@ -2,6 +2,7 @@ import { Mic } from "lucide-react";
|
||||
import { Button } from "@/shared/ui/button";
|
||||
import { Tooltip, TooltipContent, TooltipTrigger } from "@/shared/ui/tooltip";
|
||||
import { cn } from "@/shared/lib/cn";
|
||||
import { isMacPlatform } from "@/shared/lib/platform";
|
||||
|
||||
interface DictationState {
|
||||
isEnabled: boolean;
|
||||
@@ -22,25 +23,19 @@ export function DictationButton({
|
||||
}: DictationButtonProps) {
|
||||
if (!dictation.isEnabled) return null;
|
||||
|
||||
const shortcutHint = isMacPlatform() ? "⌘D" : "Ctrl+D";
|
||||
|
||||
const tooltipText = dictation.isRecording
|
||||
? "Stop recording"
|
||||
: dictation.isTranscribing
|
||||
? "Transcribing…"
|
||||
: "Dictate message";
|
||||
|
||||
// Allow the stop action whenever the mic is live (isRecording), even if
|
||||
// the session setup is still in progress (isStarting) or the composer is
|
||||
// disabled. Only block the button when idle + disabled, or when startup
|
||||
// hasn't captured the mic yet (isStarting && !isRecording).
|
||||
const isDisabled = dictation.isRecording
|
||||
? false
|
||||
: disabled || dictation.isStarting;
|
||||
: null;
|
||||
|
||||
return (
|
||||
<Tooltip>
|
||||
<TooltipTrigger asChild>
|
||||
<Button
|
||||
aria-label={tooltipText}
|
||||
aria-label={tooltipText ?? "Voice Dictation"}
|
||||
aria-pressed={dictation.isRecording}
|
||||
className={cn(
|
||||
"rounded-full",
|
||||
@@ -48,7 +43,9 @@ export function DictationButton({
|
||||
"bg-destructive text-destructive-foreground hover:bg-destructive/90 hover:text-destructive-foreground active:bg-destructive active:text-destructive-foreground",
|
||||
dictation.isTranscribing && "animate-pulse",
|
||||
)}
|
||||
disabled={isDisabled}
|
||||
disabled={
|
||||
dictation.isRecording ? false : disabled || dictation.isStarting
|
||||
}
|
||||
onClick={dictation.toggleRecording}
|
||||
size="icon"
|
||||
type="button"
|
||||
@@ -57,7 +54,16 @@ export function DictationButton({
|
||||
<Mic />
|
||||
</Button>
|
||||
</TooltipTrigger>
|
||||
<TooltipContent>{tooltipText}</TooltipContent>
|
||||
<TooltipContent>
|
||||
{tooltipText ?? (
|
||||
<span className="flex items-center gap-1.5">
|
||||
Voice Dictation
|
||||
<kbd className="rounded border border-foreground/10 px-1 py-0.5 text-2xs font-medium text-foreground/60">
|
||||
{shortcutHint}
|
||||
</kbd>
|
||||
</span>
|
||||
)}
|
||||
</TooltipContent>
|
||||
</Tooltip>
|
||||
);
|
||||
}
|
||||
|
||||
@@ -181,6 +181,14 @@ export const KEYBOARD_SHORTCUTS: KeyboardShortcut[] = [
|
||||
keysWindows: "Ctrl+Space",
|
||||
category: "Messages",
|
||||
},
|
||||
{
|
||||
id: "voice-dictation",
|
||||
label: "Voice Dictation",
|
||||
description: "Start or stop voice dictation in the composer",
|
||||
keys: "⌘D",
|
||||
keysWindows: "Ctrl+D",
|
||||
category: "Messages",
|
||||
},
|
||||
|
||||
// Formatting
|
||||
{
|
||||
|
||||
Reference in New Issue
Block a user