fix(dictation): typed transcription session, sync editor before merge, block sends during upload

- Use OpenAI typed transcription session format (type: "transcription")
  instead of legacy realtime fields that would fail or produce no transcripts
- Sync editor content via syncContentRef before merging dictation text so
  manually typed prefixes are preserved when dictation starts
- Read send-blocked state from refs at transcript time so uploads prevent
  auto-submit from clearing the composer

Signed-off-by: klopez4212 <klopez4212@gmail.com>
This commit is contained in:
klopez4212
2026-07-11 16:18:07 +01:00
parent 71580439bb
commit caa6bbfea8
4 changed files with 33 additions and 30 deletions
+1 -2
View File
@@ -82,8 +82,7 @@ pub async fn create_transcribe_session(
.header("Content-Type", "application/json")
.json(&serde_json::json!({
"session": {
"model": "gpt-4o-mini-realtime-preview",
"modalities": ["text"],
"type": "transcription",
"input_audio_transcription": {
"model": model,
},
@@ -1,10 +1,13 @@
import type * as React from "react";
import { useRef } from "react";
import { useDictation } from "./useDictation";
interface UseComposerDictationOptions {
contentRef: React.MutableRefObject<string>;
disabled: boolean;
isSending: boolean;
/** Ref to a function that syncs contentRef from the Tiptap editor and returns it. */
syncContentRef: React.MutableRefObject<() => string>;
disabledRef: React.MutableRefObject<boolean>;
isSendingRef: React.MutableRefObject<boolean>;
isUploadingRef: React.MutableRefObject<boolean>;
/** Updates contentRef + isContentEmpty state. */
setComposerContent: (text: string) => void;
/** Ref to a function that updates the Tiptap editor document. */
@@ -14,18 +17,23 @@ interface UseComposerDictationOptions {
/**
* Thin wrapper around `useDictation` pre-wired for the MessageComposer's
* state management (contentRef, setComposerContent, editor, submitMessageRef).
* state management (syncContentRef, setComposerContent, editor, submitMessageRef).
*/
export function useComposerDictation({
contentRef,
disabled,
isSending,
syncContentRef,
disabledRef,
isSendingRef,
isUploadingRef,
setComposerContent,
setEditorContentRef,
submitMessageRef,
}: UseComposerDictationOptions) {
const isSendBlockedRef = useRef(false);
isSendBlockedRef.current =
disabledRef.current || isSendingRef.current || isUploadingRef.current;
return useDictation({
text: contentRef.current,
getText: () => syncContentRef.current(),
setText: (text) => {
setComposerContent(text);
setEditorContentRef.current(text);
@@ -38,6 +46,6 @@ export function useComposerDictation({
// holds the dictated text.
submitMessageRef.current();
},
sendDisabled: disabled || isSending,
isSendBlockedRef,
});
}
@@ -1,3 +1,4 @@
import type * as React from "react";
import { useCallback, useMemo, useRef } from "react";
import {
DEFAULT_AUTO_SUBMIT_PHRASE,
@@ -8,35 +9,33 @@ import {
import { useRealtimeDictation } from "./useRealtimeDictation";
interface UseDictationOptions {
/** Current composer text */
text: string;
/** Returns the current composer text (must be fresh — synced from editor). */
getText: () => string;
/** Set composer text */
setText: (value: string) => void;
/** Send the message */
onSend: (text: string) => void;
/** Whether sending is currently blocked */
sendDisabled?: boolean;
/** Ref that is `true` when sending is blocked (uploading, preparing mention, etc.) */
isSendBlockedRef?: React.MutableRefObject<boolean>;
}
export function useDictation({
text,
getText,
setText,
onSend,
sendDisabled = false,
isSendBlockedRef,
}: UseDictationOptions) {
const autoSubmitPhrases = useMemo(
() => parseAutoSubmitPhrases(DEFAULT_AUTO_SUBMIT_PHRASE),
[],
);
const stopRecordingRef = useRef<() => void>(() => {});
const textRef = useRef(text);
textRef.current = text;
const lastTranscriptRef = useRef("");
const handleTranscript = useCallback(
(transcript: string) => {
const previous = lastTranscriptRef.current;
const latest = textRef.current;
const latest = getText();
const merged = replaceTrailingTranscribedText(
latest,
previous,
@@ -46,7 +45,6 @@ export function useDictation({
if (!match) {
setText(merged);
textRef.current = merged;
lastTranscriptRef.current = transcript;
return;
}
@@ -60,18 +58,16 @@ export function useDictation({
stopRecordingRef.current();
if (sendDisabled) {
if (isSendBlockedRef?.current) {
setText(textWithoutPhrase);
textRef.current = textWithoutPhrase;
return;
}
onSend(textWithoutPhrase.trim());
setText("");
textRef.current = "";
lastTranscriptRef.current = "";
},
[autoSubmitPhrases, onSend, sendDisabled, setText],
[autoSubmitPhrases, getText, onSend, isSendBlockedRef, setText],
);
const dictation = useRealtimeDictation({
@@ -266,17 +266,17 @@ function MessageComposerImpl({
emojiAutocomplete.isEmojiAutocompleteOpen;
const submitMessageRef = React.useRef<() => void>(() => {});
const setEditorContentRef = React.useRef<(t: string) => void>(() => {});
const setEditorContentRef = React.useRef<(text: string) => void>(() => {});
const dictation = useComposerDictation({
contentRef,
disabled,
isSending,
syncContentRef: syncContentRefFromEditorRef,
disabledRef,
isSendingRef,
isUploadingRef,
setComposerContent,
setEditorContentRef,
submitMessageRef,
});
const composerScrollRef = React.useRef<HTMLDivElement>(null);
// Set after `useLinkEditor` exists below; the editor's link-click handler
// delegates through this ref to break the hook ordering cycle (the editor
// needs `onEditLink`, but the link editor needs the editor's `richText`).