From 5fe86fbb6d6e89fe57c3835fe0bba87f8d2d9958 Mon Sep 17 00:00:00 2001 From: "t3-code[bot]" <269035359+t3-code[bot]@users.noreply.github.com> Date: Sun, 2 Aug 2026 04:33:13 +0000 Subject: [PATCH 01/21] feat: add voice dictation beta --- .../settings/DesktopClientSettings.test.ts | 5 + apps/server/src/http.ts | 46 ++++++ apps/server/src/httpCors.ts | 3 + apps/server/src/server.ts | 2 + apps/server/src/transcription.test.ts | 66 ++++++++ apps/server/src/transcription.ts | 113 +++++++++++++ apps/web/src/components/chat/ChatComposer.tsx | 63 ++++++++ .../chat/VoiceTranscriptionPanel.tsx | 68 ++++++++ .../src/components/settings/settingsSearch.ts | 5 + apps/web/src/hooks/useVoiceTranscription.ts | 149 ++++++++++++++++++ apps/web/src/lib/voiceTranscription.ts | 47 ++++++ packages/contracts/src/settings.ts | 21 +++ scripts/build-desktop-artifact.test.ts | 3 + scripts/build-desktop-artifact.ts | 3 + 14 files changed, 594 insertions(+) create mode 100644 apps/server/src/transcription.test.ts create mode 100644 apps/server/src/transcription.ts create mode 100644 apps/web/src/components/chat/VoiceTranscriptionPanel.tsx create mode 100644 apps/web/src/hooks/useVoiceTranscription.ts create mode 100644 apps/web/src/lib/voiceTranscription.ts diff --git a/apps/desktop/src/settings/DesktopClientSettings.test.ts b/apps/desktop/src/settings/DesktopClientSettings.test.ts index 44c12cc554a..8c9266ddedc 100644 --- a/apps/desktop/src/settings/DesktopClientSettings.test.ts +++ b/apps/desktop/src/settings/DesktopClientSettings.test.ts @@ -42,6 +42,11 @@ const clientSettings: ClientSettings = { sidebarThreadPreviewCount: 6, legacySidebarEnabled: false, timestampFormat: "24-hour", + voiceTranscriptionEnabled: true, + voiceTranscriptionProvider: "local", + voiceTranscriptionBaseUrl: "http://127.0.0.1:8080/v1", + voiceTranscriptionModel: "whisper-1", + voiceTranscriptionApiKey: "", wordWrap: true, }; diff --git a/apps/server/src/http.ts b/apps/server/src/http.ts index 0da55686b92..af25dc6a943 100644 --- a/apps/server/src/http.ts +++ b/apps/server/src/http.ts @@ -39,8 +39,10 @@ import { } from "./auth/http.ts"; import * as ServerEnvironment from "./environment/ServerEnvironment.ts"; import { browserApiCorsAllowedHeaders, browserApiCorsAllowedMethods } from "./httpCors.ts"; +import { forwardWhisperTranscription, MAX_TRANSCRIPTION_AUDIO_BYTES } from "./transcription.ts"; const OTLP_TRACES_PROXY_PATH = "/api/observability/v1/traces"; +const TRANSCRIPTION_PATH = "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/api/transcription"; const LOOPBACK_HOSTNAMES = new Set(["127.0.0.1", "::1", "localhost"]); const DESKTOP_RENDERER_ORIGINS = ["t3code://app", "t3code-dev://app"]; const SVG_CONTENT_SECURITY_POLICY = "default-src 'none'; style-src 'unsafe-inline'; sandbox"; @@ -194,6 +196,50 @@ export const otlpTracesProxyRouteLayer = HttpRouter.add( ), ); +export const transcriptionRouteLayer = HttpRouter.add( + "POST", + TRANSCRIPTION_PATH, + Effect.gen(function* () { + yield* authenticateRawRouteWithScope(AuthOrchestrationOperateScope); + const request = yield* HttpServerRequest.HttpServerRequest; + const declaredLength = Number(request.headers["content-length"] ?? "0"); + if (Number.isFinite(declaredLength) && declaredLength > MAX_TRANSCRIPTION_AUDIO_BYTES) { + return HttpServerResponse.jsonUnsafe( + { error: "The recording exceeds the 25 MB limit." }, + { status: 413 }, + ); + } + + const body = yield* request.arrayBuffer; + const audio = new Uint8Array(body); + const result = yield* Effect.tryPromise(() => + forwardWhisperTranscription({ + audio, + audioMimeType: request.headers["content-type"] ?? "audio/webm", + baseUrl: request.headers["x-t3-transcription-base-url"] ?? "", + model: request.headers["x-t3-transcription-model"] ?? "", + apiKey: request.headers["x-t3-transcription-api-key"] ?? "", + }), + ).pipe( + Effect.orElseSucceed(() => ({ + ok: false as const, + status: 502, + message: "Transcription failed unexpectedly.", + })), + ); + + return result.ok + ? HttpServerResponse.jsonUnsafe({ text: result.text }) + : HttpServerResponse.jsonUnsafe({ error: result.message }, { status: result.status }); + }).pipe( + Effect.catchTags({ + EnvironmentAuthInvalidError: HttpServerRespondable.toResponse, + EnvironmentInternalError: HttpServerRespondable.toResponse, + EnvironmentScopeRequiredError: HttpServerRespondable.toResponse, + }), + ), +); + export const assetRouteLayer = HttpRouter.add( "GET", `${ASSET_ROUTE_PREFIX}/*`, diff --git a/apps/server/src/httpCors.ts b/apps/server/src/httpCors.ts index aeb8dbce5a5..b41f65e1fb1 100644 --- a/apps/server/src/httpCors.ts +++ b/apps/server/src/httpCors.ts @@ -5,6 +5,9 @@ export const browserApiCorsAllowedHeaders = [ "traceparent", "content-type", "dpop", + "x-t3-transcription-api-key", + "x-t3-transcription-base-url", + "x-t3-transcription-model", ] as const; export const browserApiCorsHeaders = { diff --git a/apps/server/src/server.ts b/apps/server/src/server.ts index 32bcaaa8b96..823c4fe3bd8 100644 --- a/apps/server/src/server.ts +++ b/apps/server/src/server.ts @@ -12,6 +12,7 @@ import * as HostPowerMonitor from "./background/HostPowerMonitor.ts"; import * as ServerConfig from "./config.ts"; import { otlpTracesProxyRouteLayer, + transcriptionRouteLayer, assetRouteLayer, serverEnvironmentHttpApiLayer, staticAndDevRouteLayer, @@ -450,6 +451,7 @@ export const makeRoutesLayer = Layer.mergeAll( Layer.provide(environmentAuthenticatedAuthLayer), ), otlpTracesProxyRouteLayer, + transcriptionRouteLayer, assetRouteLayer, staticAndDevRouteLayer, websocketRpcRouteLayer, diff --git a/apps/server/src/transcription.test.ts b/apps/server/src/transcription.test.ts new file mode 100644 index 00000000000..44dd0a135e4 --- /dev/null +++ b/apps/server/src/transcription.test.ts @@ -0,0 +1,66 @@ +import { expect, it } from "@effect/vitest"; +import { describe, vi } from "vite-plus/test"; + +import { forwardWhisperTranscription, resolveWhisperTranscriptionUrl } from "./transcription.ts"; + +describe("resolveWhisperTranscriptionUrl", () => { + it("adds the OpenAI-compatible transcription path", () => { + expect(resolveWhisperTranscriptionUrl("http://127.0.0.1:8080/v1/")?.toString()).toBe( + "http://127.0.0.1:8080/v1/audio/transcriptions", + ); + expect( + resolveWhisperTranscriptionUrl( + "https://api.groq.com/openai/v1/audio/transcriptions", + )?.toString(), + ).toBe("https://api.groq.com/openai/v1/audio/transcriptions"); + }); + + it("rejects non-http and credential-bearing URLs", () => { + expect(resolveWhisperTranscriptionUrl("file:///tmp/whisper")).toBeNull(); + expect(resolveWhisperTranscriptionUrl("https://user:pass@example.com/v1")).toBeNull(); + }); +}); + +describe("forwardWhisperTranscription", () => { + it("sends OpenAI-compatible multipart audio with BYOK authorization", async () => { + const fetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => { + expect(init?.headers).toEqual({ authorization: "Bearer groq-key" }); + expect(init?.body).toBeInstanceOf(FormData); + const form = init?.body as FormData; + expect(form.get("model")).toBe("whisper-large-v3-turbo"); + expect(form.get("file")).toBeInstanceOf(Blob); + return Response.json({ text: "hello from the microphone" }); + }); + + await expect( + forwardWhisperTranscription( + { + audio: new Uint8Array([1, 2, 3]), + audioMimeType: "audio/webm;codecs=opus", + baseUrl: "https://api.groq.com/openai/v1", + model: "whisper-large-v3-turbo", + apiKey: "groq-key", + }, + fetchImpl, + ), + ).resolves.toEqual({ ok: true, text: "hello from the microphone" }); + }); + + it("returns a safe provider error", async () => { + const fetchImpl = vi.fn(async () => + Response.json({ error: { message: "bad key" } }, { status: 401 }), + ); + await expect( + forwardWhisperTranscription( + { + audio: new Uint8Array([1]), + audioMimeType: "audio/webm", + baseUrl: "https://example.com/v1", + model: "whisper-1", + apiKey: "bad", + }, + fetchImpl, + ), + ).resolves.toEqual({ ok: false, status: 502, message: "bad key" }); + }); +}); diff --git a/apps/server/src/transcription.ts b/apps/server/src/transcription.ts new file mode 100644 index 00000000000..5d380bafa88 --- /dev/null +++ b/apps/server/src/transcription.ts @@ -0,0 +1,113 @@ +const TRANSCRIPTION_PATH = "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/audio/transcriptions"; + +export const MAX_TRANSCRIPTION_AUDIO_BYTES = 25 * 1024 * 1024; + +export interface WhisperTranscriptionInput { + readonly audio: Uint8Array; + readonly audioMimeType: string; + readonly baseUrl: string; + readonly model: string; + readonly apiKey: string; +} + +type TranscriptionFetch = (input: string | URL | Request, init?: RequestInit) => Promise; + +export type WhisperTranscriptionResult = + | { readonly ok: true; readonly text: string } + | { readonly ok: false; readonly status: number; readonly message: string }; + +export function resolveWhisperTranscriptionUrl(baseUrl: string): URL | null { + try { + const url = new URL(baseUrl.trim()); + if (url.protocol !== "http:" && url.protocol !== "https:") return null; + if (url.username || url.password) return null; + url.search = ""; + url.hash = ""; + url.pathname = url.pathname.replace(/\/$/, ""); + if (!url.pathname.endsWith(TRANSCRIPTION_PATH)) { + url.pathname += TRANSCRIPTION_PATH; + } + return url; + } catch { + return null; + } +} + +function audioFileExtension(mimeType: string): string { + if (mimeType.includes("ogg")) return "ogg"; + if (mimeType.includes("mp4") || mimeType.includes("m4a")) return "m4a"; + if (mimeType.includes("wav")) return "wav"; + return "webm"; +} + +export async function forwardWhisperTranscription( + input: WhisperTranscriptionInput, + fetchImpl: TranscriptionFetch = globalThis.fetch, +): Promise { + const endpoint = resolveWhisperTranscriptionUrl(input.baseUrl); + if (!endpoint) { + return { ok: false, status: 400, message: "Enter a valid HTTP transcription endpoint." }; + } + if (!input.model.trim()) { + return { ok: false, status: 400, message: "Enter a transcription model." }; + } + if (input.audio.byteLength === 0) { + return { ok: false, status: 400, message: "The recording was empty." }; + } + if (input.audio.byteLength > MAX_TRANSCRIPTION_AUDIO_BYTES) { + return { ok: false, status: 413, message: "The recording exceeds the 25 MB limit." }; + } + + const mimeType = input.audioMimeType.split(";", 1)[0]?.trim() || "audio/webm"; + const form = new FormData(); + form.set("model", input.model.trim()); + form.set( + "file", + new Blob([input.audio], { type: mimeType }), + `recording.${audioFileExtension(mimeType)}`, + ); + + let response: Response; + try { + response = await fetchImpl(endpoint, { + method: "POST", + headers: input.apiKey ? { authorization: `Bearer ${input.apiKey}` } : undefined, + body: form, + signal: AbortSignal.timeout(120_000), + }); + } catch { + return { ok: false, status: 502, message: "Could not reach the transcription provider." }; + } + + let payload: unknown; + try { + payload = await response.json(); + } catch { + return { ok: false, status: 502, message: "The transcription provider returned invalid JSON." }; + } + + if (!response.ok) { + const providerMessage = + typeof payload === "object" && + payload !== null && + "error" in payload && + typeof payload.error === "object" && + payload.error !== null && + "message" in payload.error && + typeof payload.error.message === "string" + ? payload.error.message + : null; + return { + ok: false, + status: 502, + message: providerMessage?.slice(0, 300) || "The transcription provider rejected the request.", + }; + } + + const text = + typeof payload === "object" && payload !== null && "text" in payload ? payload.text : undefined; + if (typeof text !== "string") { + return { ok: false, status: 502, message: "The transcription response did not contain text." }; + } + return { ok: true, text: text.trim() }; +} diff --git a/apps/web/src/components/chat/ChatComposer.tsx b/apps/web/src/components/chat/ChatComposer.tsx index e918e775868..6823101a706 100644 --- a/apps/web/src/components/chat/ChatComposer.tsx +++ b/apps/web/src/components/chat/ChatComposer.tsx @@ -106,6 +106,8 @@ import { buildExpandedImagePreview, type ExpandedImagePreview } from "./Expanded import { basenameOfPath } from "../../pierre-icons"; import { cn, randomUUID } from "~/lib/utils"; import { Separator } from "../ui/separator"; +import { useVoiceTranscription } from "../../hooks/useVoiceTranscription"; +import { VoiceTranscriptionPanel } from "./VoiceTranscriptionPanel"; type ComposerCommandMenuPosition = { bottom: number; @@ -200,6 +202,7 @@ import { LockOpenIcon, PenLineIcon, SparklesIcon, + MicIcon, XIcon, } from "lucide-react"; import { proposedPlanTitle } from "../../proposedPlan"; @@ -1254,6 +1257,30 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) [composerDraftTarget, setComposerDraftPrompt], ); + const appendVoiceTranscript = useCallback( + (transcript: string) => { + const currentPrompt = promptRef.current; + const boundary = currentPrompt.length > 0 && !/\s$/.test(currentPrompt) ? " " : ""; + const nextPrompt = `${currentPrompt}${boundary}${transcript}`; + promptRef.current = nextPrompt; + setPrompt(nextPrompt); + const nextCursor = collapseExpandedComposerCursor(nextPrompt, nextPrompt.length); + setComposerCursor(nextCursor); + setComposerTrigger(null); + scheduleComposerFocus(); + }, + [promptRef, scheduleComposerFocus, setPrompt], + ); + const voiceTranscription = useVoiceTranscription({ + config: { + provider: settings.voiceTranscriptionProvider, + baseUrl: settings.voiceTranscriptionBaseUrl, + model: settings.voiceTranscriptionModel, + apiKey: settings.voiceTranscriptionApiKey, + }, + onTranscript: appendVoiceTranscript, + }); + const addComposerImage = useCallback( (image: ComposerImageAttachment) => { addComposerDraftImage(composerDraftTarget, image); @@ -3086,6 +3113,19 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) + {settings.voiceTranscriptionEnabled && voiceTranscription.status !== "idle" ? ( + + ) : settings.voiceTranscriptionEnabled && voiceTranscription.error ? ( +

+ {voiceTranscription.error} +

+ ) : null} + {/* Bottom toolbar */} {isComposerCollapsedMobile ? null : activePendingApproval ? (
@@ -3182,6 +3222,29 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) } className="flex shrink-0 flex-nowrap items-center justify-end gap-2" > + {settings.voiceTranscriptionEnabled && + voiceTranscription.status === "idle" && + !isComposerApprovalState && + pendingUserInputs.length === 0 ? ( + + void voiceTranscription.start()} + aria-label="Start dictation" + > + + + } + /> + Start dictation + + ) : null} `voice-waveform-${index}`); + +function formatElapsed(elapsedMs: number): string { + const seconds = Math.floor(elapsedMs / 1_000); + return `${Math.floor(seconds / 60)}:${String(seconds % 60).padStart(2, "0")}`; +} + +export function VoiceTranscriptionPanel({ + status, + elapsedMs, + levels, + onStop, +}: { + readonly status: Exclude; + readonly elapsedMs: number; + readonly levels: readonly number[]; + readonly onStop: () => void; +}) { + return ( +
+ {status === "recording" ? ( + + ) : ( +
+ + transcribing with your configured provider... +
+ )} + + {formatElapsed(elapsedMs)} + + {status === "recording" ? ( + + ) : null} +
+ ); +} diff --git a/apps/web/src/components/settings/settingsSearch.ts b/apps/web/src/components/settings/settingsSearch.ts index e0fc3d2f07e..5495fdaaf57 100644 --- a/apps/web/src/components/settings/settingsSearch.ts +++ b/apps/web/src/components/settings/settingsSearch.ts @@ -123,6 +123,11 @@ export const SETTINGS_SEARCH_ITEMS = [ title: "Provider update checks", to: "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/settings/general", }, + { + id: "voice-dictation", + title: "Voice dictation", + to: "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/settings/general", + }, { id: "new-threads", title: "New threads", diff --git a/apps/web/src/hooks/useVoiceTranscription.ts b/apps/web/src/hooks/useVoiceTranscription.ts new file mode 100644 index 00000000000..061739cef37 --- /dev/null +++ b/apps/web/src/hooks/useVoiceTranscription.ts @@ -0,0 +1,149 @@ +import { useCallback, useEffect, useRef, useState } from "react"; + +import { transcribeVoiceRecording, type VoiceTranscriptionConfig } from "../lib/voiceTranscription"; + +const LEVEL_COUNT = 36; +const MAX_RECORDING_MS = 5 * 60 * 1_000; +const MIME_TYPES = ["audio/webm;codecs=opus", "audio/ogg;codecs=opus", "audio/mp4"]; + +export type VoiceTranscriptionStatus = "idle" | "recording" | "transcribing"; + +function supportedMimeType(): string | undefined { + if (typeof MediaRecorder === "undefined") return undefined; + return MIME_TYPES.find((mimeType) => MediaRecorder.isTypeSupported(mimeType)); +} + +export function useVoiceTranscription({ + config, + onTranscript, +}: { + readonly config: VoiceTranscriptionConfig; + readonly onTranscript: (text: string) => void; +}) { + const [status, setStatus] = useState("idle"); + const [elapsedMs, setElapsedMs] = useState(0); + const [levels, setLevels] = useState(() => Array(LEVEL_COUNT).fill(0.12)); + const [error, setError] = useState(null); + const recorderRef = useRef(null); + const streamRef = useRef(null); + const audioContextRef = useRef(null); + const intervalsRef = useRef([]); + const timeoutRef = useRef(null); + const startedAtRef = useRef(0); + const mountedRef = useRef(true); + const configRef = useRef(config); + const onTranscriptRef = useRef(onTranscript); + configRef.current = config; + onTranscriptRef.current = onTranscript; + + const cleanupCapture = useCallback(() => { + for (const interval of intervalsRef.current) window.clearInterval(interval); + intervalsRef.current = []; + if (timeoutRef.current !== null) window.clearTimeout(timeoutRef.current); + timeoutRef.current = null; + streamRef.current?.getTracks().forEach((track) => track.stop()); + streamRef.current = null; + void audioContextRef.current?.close(); + audioContextRef.current = null; + recorderRef.current = null; + }, []); + + useEffect(() => { + mountedRef.current = true; + return () => { + mountedRef.current = false; + const recorder = recorderRef.current; + if (recorder?.state === "recording") recorder.stop(); + cleanupCapture(); + }; + }, [cleanupCapture]); + + const stop = useCallback(() => { + const recorder = recorderRef.current; + if (recorder?.state === "recording") recorder.stop(); + }, []); + + const start = useCallback(async () => { + if (status !== "idle") return; + setError(null); + if (!navigator.mediaDevices?.getUserMedia || typeof MediaRecorder === "undefined") { + setError("Microphone recording is not supported on this device."); + return; + } + + try { + const stream = await navigator.mediaDevices.getUserMedia({ + audio: { echoCancellation: true, noiseSuppression: true, autoGainControl: true }, + }); + if (!mountedRef.current) { + stream.getTracks().forEach((track) => track.stop()); + return; + } + + const mimeType = supportedMimeType(); + const recorder = new MediaRecorder(stream, mimeType ? { mimeType } : undefined); + const chunks: Blob[] = []; + streamRef.current = stream; + recorderRef.current = recorder; + recorder.addEventListener("dataavailable", (event) => { + if (event.data.size > 0) chunks.push(event.data); + }); + recorder.addEventListener("error", () => { + if (mountedRef.current) setError("The microphone stopped unexpectedly."); + }); + recorder.addEventListener("stop", () => { + const blob = new Blob(chunks, { type: recorder.mimeType || mimeType || "audio/webm" }); + cleanupCapture(); + if (!mountedRef.current) return; + setStatus("transcribing"); + void transcribeVoiceRecording(blob, configRef.current) + .then((text) => { + if (!mountedRef.current) return; + if (text) onTranscriptRef.current(text); + setStatus("idle"); + setElapsedMs(0); + }) + .catch((cause: unknown) => { + if (!mountedRef.current) return; + setError(cause instanceof Error ? cause.message : "Voice transcription failed."); + setStatus("idle"); + }); + }); + + const audioContext = new AudioContext(); + audioContextRef.current = audioContext; + const analyser = audioContext.createAnalyser(); + analyser.fftSize = 128; + audioContext.createMediaStreamSource(stream).connect(analyser); + const samples = new Uint8Array(analyser.frequencyBinCount); + intervalsRef.current.push( + window.setInterval(() => { + analyser.getByteFrequencyData(samples); + setLevels( + Array.from({ length: LEVEL_COUNT }, (_, index) => { + const sampleIndex = Math.floor((index / LEVEL_COUNT) * samples.length); + return Math.max(0.1, (samples[sampleIndex] ?? 0) / 255); + }), + ); + }, 80), + ); + startedAtRef.current = Date.now(); + intervalsRef.current.push( + window.setInterval(() => setElapsedMs(Date.now() - startedAtRef.current), 250), + ); + timeoutRef.current = window.setTimeout(() => recorder.stop(), MAX_RECORDING_MS); + recorder.start(250); + setStatus("recording"); + } catch (cause) { + cleanupCapture(); + setStatus("idle"); + setError( + cause instanceof DOMException && cause.name === "NotAllowedError" + ? "Microphone permission was denied. Allow access and try again." + : "Could not start the microphone.", + ); + } + }, [cleanupCapture, status]); + + return { status, elapsedMs, levels, error, start, stop } as const; +} diff --git a/apps/web/src/lib/voiceTranscription.ts b/apps/web/src/lib/voiceTranscription.ts new file mode 100644 index 00000000000..05499b4c4fc --- /dev/null +++ b/apps/web/src/lib/voiceTranscription.ts @@ -0,0 +1,47 @@ +import type { VoiceTranscriptionProvider } from "@t3tools/contracts"; + +import { readDesktopPrimaryBearerToken } from "../environments/primary/desktopAuth"; +import { resolvePrimaryEnvironmentHttpUrl } from "../environments/primary/target"; + +export const GROQ_TRANSCRIPTION_BASE_URL = "https://api.groq.com/openai/v1"; +export const GROQ_TRANSCRIPTION_MODEL = "whisper-large-v3-turbo"; + +export interface VoiceTranscriptionConfig { + readonly provider: VoiceTranscriptionProvider; + readonly baseUrl: string; + readonly model: string; + readonly apiKey: string; +} + +export async function transcribeVoiceRecording( + audio: Blob, + config: VoiceTranscriptionConfig, +): Promise { + const bearerToken = await readDesktopPrimaryBearerToken(); + const response = await globalThis.fetch(resolvePrimaryEnvironmentHttpUrl("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/api/transcription"), { + method: "POST", + credentials: bearerToken ? "omit" : "include", + headers: { + ...(bearerToken ? { authorization: `Bearer ${bearerToken}` } : {}), + "content-type": audio.type || "audio/webm", + "x-t3-transcription-base-url": config.baseUrl, + "x-t3-transcription-model": config.model, + ...(config.apiKey ? { "x-t3-transcription-api-key": config.apiKey } : {}), + }, + body: audio, + }); + + const payload = (await response.json().catch(() => null)) as { + readonly text?: unknown; + readonly error?: unknown; + } | null; + if (!response.ok) { + throw new Error( + typeof payload?.error === "string" ? payload.error : "Voice transcription failed.", + ); + } + if (typeof payload?.text !== "string") { + throw new Error("The transcription response did not contain text."); + } + return payload.text.trim(); +} diff --git a/packages/contracts/src/settings.ts b/packages/contracts/src/settings.ts index ee1970639ad..bc7fff64189 100644 --- a/packages/contracts/src/settings.ts +++ b/packages/contracts/src/settings.ts @@ -111,6 +111,11 @@ export const DEFAULT_ENVIRONMENT_IDENTIFICATION_MODE: EnvironmentIdentificationM export const FontFamilyPreference = Schema.String.check(Schema.isMaxLength(200)); export type FontFamilyPreference = typeof FontFamilyPreference.Type; +export const VoiceTranscriptionProvider = Schema.Literals(["local", "groq", "custom"]); +export type VoiceTranscriptionProvider = typeof VoiceTranscriptionProvider.Type; +export const DEFAULT_VOICE_TRANSCRIPTION_BASE_URL = "http://127.0.0.1:8080/v1"; +export const DEFAULT_VOICE_TRANSCRIPTION_MODEL = "whisper-1"; + export const ClientSettingsSchema = Schema.Struct({ confirmThreadArchive: Schema.Boolean.pipe(Schema.withDecodingDefault(Effect.succeed(false))), confirmThreadDelete: Schema.Boolean.pipe(Schema.withDecodingDefault(Effect.succeed(true))), @@ -200,6 +205,17 @@ export const ClientSettingsSchema = Schema.Struct({ timestampFormat: TimestampFormat.pipe( Schema.withDecodingDefault(Effect.succeed(DEFAULT_TIMESTAMP_FORMAT)), ), + voiceTranscriptionEnabled: Schema.Boolean.pipe(Schema.withDecodingDefault(Effect.succeed(false))), + voiceTranscriptionProvider: VoiceTranscriptionProvider.pipe( + Schema.withDecodingDefault(Effect.succeed("local" as const)), + ), + voiceTranscriptionBaseUrl: TrimmedString.pipe( + Schema.withDecodingDefault(Effect.succeed(DEFAULT_VOICE_TRANSCRIPTION_BASE_URL)), + ), + voiceTranscriptionModel: TrimmedString.pipe( + Schema.withDecodingDefault(Effect.succeed(DEFAULT_VOICE_TRANSCRIPTION_MODEL)), + ), + voiceTranscriptionApiKey: Schema.String.pipe(Schema.withDecodingDefault(Effect.succeed(""))), wordWrap: Schema.Boolean.pipe(Schema.withDecodingDefault(Effect.succeed(true))), }); export type ClientSettings = typeof ClientSettingsSchema.Type; @@ -803,6 +819,11 @@ export const ClientSettingsPatch = Schema.Struct({ sidebarThreadSortOrder: Schema.optionalKey(SidebarThreadSortOrder), sidebarThreadPreviewCount: Schema.optionalKey(SidebarThreadPreviewCount), timestampFormat: Schema.optionalKey(TimestampFormat), + voiceTranscriptionEnabled: Schema.optionalKey(Schema.Boolean), + voiceTranscriptionProvider: Schema.optionalKey(VoiceTranscriptionProvider), + voiceTranscriptionBaseUrl: Schema.optionalKey(TrimmedString), + voiceTranscriptionModel: Schema.optionalKey(TrimmedString), + voiceTranscriptionApiKey: Schema.optionalKey(Schema.String), wordWrap: Schema.optionalKey(Schema.Boolean), }); export type ClientSettingsPatch = typeof ClientSettingsPatch.Type; diff --git a/scripts/build-desktop-artifact.test.ts b/scripts/build-desktop-artifact.test.ts index 2b9fd3e0258..3dd437c30e3 100644 --- a/scripts/build-desktop-artifact.test.ts +++ b/scripts/build-desktop-artifact.test.ts @@ -463,6 +463,9 @@ it.layer(NodeServices.layer)("build-desktop-artifact", (it) => { "**/node_modules/.bin", "**/node_modules/.bin/**", ]); + assert.deepStrictEqual((mac.mac as Record).extendInfo, { + NSMicrophoneUsageDescription: "T3 Code uses the microphone for voice dictation.", + }); // Linux must register the renderer schemes so the generated .desktop // entry advertises MimeType=x-scheme-handler/t3code; for OAuth deep links. assert.deepStrictEqual((linux.linux as Record).protocols, [ diff --git a/scripts/build-desktop-artifact.ts b/scripts/build-desktop-artifact.ts index 0cda9766f37..c7501a208e7 100644 --- a/scripts/build-desktop-artifact.ts +++ b/scripts/build-desktop-artifact.ts @@ -2024,6 +2024,9 @@ export const createBuildConfig = Effect.fn("createBuildConfig")(function* ( target: target === "dmg" ? [target, "zip"] : [target], icon: "icon.icns", category: "public.app-category.developer-tools", + extendInfo: { + NSMicrophoneUsageDescription: "T3 Code uses the microphone for voice dictation.", + }, protocols: [ { name: "T3 Code", From 55961eaea6d19288e972fbaf67e7ff056070e2fc Mon Sep 17 00:00:00 2001 From: "t3-code[bot]" <269035359+t3-code[bot]@users.noreply.github.com> Date: Sun, 2 Aug 2026 04:53:37 +0000 Subject: [PATCH 02/21] fix: simplify voice dictation providers --- .../settings/DesktopClientSettings.test.ts | 4 +- apps/server/src/http.ts | 52 ++-- apps/server/src/httpCors.ts | 3 +- apps/server/src/server.test.ts | 4 + apps/server/src/transcription.test.ts | 96 ++++---- apps/server/src/transcription.ts | 223 +++++++++++------- apps/web/src/components/chat/ChatComposer.tsx | 4 +- apps/web/src/hooks/useVoiceTranscription.ts | 21 +- apps/web/src/lib/voiceTranscription.ts | 13 +- packages/contracts/src/settings.ts | 18 +- 10 files changed, 249 insertions(+), 189 deletions(-) diff --git a/apps/desktop/src/settings/DesktopClientSettings.test.ts b/apps/desktop/src/settings/DesktopClientSettings.test.ts index 8c9266ddedc..e7e47d9948a 100644 --- a/apps/desktop/src/settings/DesktopClientSettings.test.ts +++ b/apps/desktop/src/settings/DesktopClientSettings.test.ts @@ -43,9 +43,7 @@ const clientSettings: ClientSettings = { legacySidebarEnabled: false, timestampFormat: "24-hour", voiceTranscriptionEnabled: true, - voiceTranscriptionProvider: "local", - voiceTranscriptionBaseUrl: "http://127.0.0.1:8080/v1", - voiceTranscriptionModel: "whisper-1", + voiceTranscriptionProvider: "openai", voiceTranscriptionApiKey: "", wordWrap: true, }; diff --git a/apps/server/src/http.ts b/apps/server/src/http.ts index af25dc6a943..bfe16f38a51 100644 --- a/apps/server/src/http.ts +++ b/apps/server/src/http.ts @@ -39,7 +39,13 @@ import { } from "./auth/http.ts"; import * as ServerEnvironment from "./environment/ServerEnvironment.ts"; import { browserApiCorsAllowedHeaders, browserApiCorsAllowedMethods } from "./httpCors.ts"; -import { forwardWhisperTranscription, MAX_TRANSCRIPTION_AUDIO_BYTES } from "./transcription.ts"; +import { + forwardVoiceTranscription, + MAX_TRANSCRIPTION_AUDIO_BYTES, + readTranscriptionAudio, + resolveTranscriptionProvider, + TranscriptionInputError, +} from "./transcription.ts"; const OTLP_TRACES_PROXY_PATH = "/api/observability/v1/traces"; const TRANSCRIPTION_PATH = "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/api/transcription"; @@ -210,32 +216,38 @@ export const transcriptionRouteLayer = HttpRouter.add( ); } - const body = yield* request.arrayBuffer; - const audio = new Uint8Array(body); - const result = yield* Effect.tryPromise(() => - forwardWhisperTranscription({ - audio, - audioMimeType: request.headers["content-type"] ?? "audio/webm", - baseUrl: request.headers["x-t3-transcription-base-url"] ?? "", - model: request.headers["x-t3-transcription-model"] ?? "", - apiKey: request.headers["x-t3-transcription-api-key"] ?? "", - }), - ).pipe( - Effect.orElseSucceed(() => ({ - ok: false as const, - status: 502, - message: "Transcription failed unexpectedly.", - })), + const provider = resolveTranscriptionProvider( + request.headers["x-t3-transcription-provider"] ?? "", ); + if (!provider) { + return yield* new TranscriptionInputError({ reason: "invalid_provider" }); + } - return result.ok - ? HttpServerResponse.jsonUnsafe({ text: result.text }) - : HttpServerResponse.jsonUnsafe({ error: result.message }, { status: result.status }); + const audio = yield* readTranscriptionAudio(request.stream); + const text = yield* forwardVoiceTranscription({ + audio, + audioMimeType: request.headers["content-type"] ?? "audio/webm", + provider, + apiKey: request.headers["x-t3-transcription-api-key"] ?? "", + }); + return HttpServerResponse.jsonUnsafe({ text }); }).pipe( Effect.catchTags({ EnvironmentAuthInvalidError: HttpServerRespondable.toResponse, EnvironmentInternalError: HttpServerRespondable.toResponse, EnvironmentScopeRequiredError: HttpServerRespondable.toResponse, + TranscriptionAudioTooLargeError: (error) => + Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 413 })), + TranscriptionBodyReadError: (error) => + Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })), + TranscriptionInputError: (error) => + Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })), + TranscriptionProviderError: (error) => + Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })), + TranscriptionRequestError: (error) => + Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })), + TranscriptionResponseError: (error) => + Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })), }), ), ); diff --git a/apps/server/src/httpCors.ts b/apps/server/src/httpCors.ts index b41f65e1fb1..a29d8f271d3 100644 --- a/apps/server/src/httpCors.ts +++ b/apps/server/src/httpCors.ts @@ -6,8 +6,7 @@ export const browserApiCorsAllowedHeaders = [ "content-type", "dpop", "x-t3-transcription-api-key", - "x-t3-transcription-base-url", - "x-t3-transcription-model", + "x-t3-transcription-provider", ] as const; export const browserApiCorsHeaders = { diff --git a/apps/server/src/server.test.ts b/apps/server/src/server.test.ts index 3f63eb4dbef..60d1984edca 100644 --- a/apps/server/src/server.test.ts +++ b/apps/server/src/server.test.ts @@ -1357,6 +1357,8 @@ const assertBrowserApiCorsPreflightHeaders = ( "content-type", "dpop", "traceparent", + "x-t3-transcription-api-key", + "x-t3-transcription-provider", ]); }; const crossOriginClientOrigin = "http://remote-client.test:3773"; @@ -4288,6 +4290,8 @@ it.layer(NodeServices.layer)("server router seam", (it) => { "content-type", "dpop", "traceparent", + "x-t3-transcription-api-key", + "x-t3-transcription-provider", ]); }).pipe(Effect.provide(NodeHttpServer.layerTest)), ); diff --git a/apps/server/src/transcription.test.ts b/apps/server/src/transcription.test.ts index 44dd0a135e4..3bbd526e0f5 100644 --- a/apps/server/src/transcription.test.ts +++ b/apps/server/src/transcription.test.ts @@ -1,66 +1,50 @@ import { expect, it } from "@effect/vitest"; -import { describe, vi } from "vite-plus/test"; +import * as Effect from "effect/Effect"; +import * as Stream from "effect/Stream"; +import { describe } from "vite-plus/test"; -import { forwardWhisperTranscription, resolveWhisperTranscriptionUrl } from "./transcription.ts"; +import { + MAX_TRANSCRIPTION_AUDIO_BYTES, + readTranscriptionAudio, + resolveTranscriptionProvider, + transcriptionProviderConfig, +} from "./transcription.ts"; -describe("resolveWhisperTranscriptionUrl", () => { - it("adds the OpenAI-compatible transcription path", () => { - expect(resolveWhisperTranscriptionUrl("http://127.0.0.1:8080/v1/")?.toString()).toBe( - "http://127.0.0.1:8080/v1/audio/transcriptions", - ); - expect( - resolveWhisperTranscriptionUrl( - "https://api.groq.com/openai/v1/audio/transcriptions", - )?.toString(), - ).toBe("https://api.groq.com/openai/v1/audio/transcriptions"); +describe("transcription providers", () => { + it("uses a fixed OpenAI endpoint and model", () => { + expect(resolveTranscriptionProvider("openai")).toBe("openai"); + expect(transcriptionProviderConfig("openai")).toEqual({ + endpoint: "https://api.openai.com/v1/audio/transcriptions", + model: "gpt-4o-mini-transcribe", + }); }); - it("rejects non-http and credential-bearing URLs", () => { - expect(resolveWhisperTranscriptionUrl("file:///tmp/whisper")).toBeNull(); - expect(resolveWhisperTranscriptionUrl("https://user:pass@example.com/v1")).toBeNull(); + it("uses a fixed Groq endpoint and rejects custom providers", () => { + expect(resolveTranscriptionProvider("groq")).toBe("groq"); + expect(transcriptionProviderConfig("groq")).toEqual({ + endpoint: "https://api.groq.com/openai/v1/audio/transcriptions", + model: "whisper-large-v3-turbo", + }); + expect(resolveTranscriptionProvider("http://127.0.0.1:8080/v1")).toBeNull(); }); }); -describe("forwardWhisperTranscription", () => { - it("sends OpenAI-compatible multipart audio with BYOK authorization", async () => { - const fetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => { - expect(init?.headers).toEqual({ authorization: "Bearer groq-key" }); - expect(init?.body).toBeInstanceOf(FormData); - const form = init?.body as FormData; - expect(form.get("model")).toBe("whisper-large-v3-turbo"); - expect(form.get("file")).toBeInstanceOf(Blob); - return Response.json({ text: "hello from the microphone" }); - }); - - await expect( - forwardWhisperTranscription( - { - audio: new Uint8Array([1, 2, 3]), - audioMimeType: "audio/webm;codecs=opus", - baseUrl: "https://api.groq.com/openai/v1", - model: "whisper-large-v3-turbo", - apiKey: "groq-key", - }, - fetchImpl, - ), - ).resolves.toEqual({ ok: true, text: "hello from the microphone" }); - }); +describe("readTranscriptionAudio", () => { + it.effect("combines streamed audio chunks", () => + Effect.gen(function* () { + const audio = yield* readTranscriptionAudio( + Stream.make(new Uint8Array([1, 2]), new Uint8Array([3, 4])), + ); + expect(Array.from(audio)).toEqual([1, 2, 3, 4]); + }), + ); - it("returns a safe provider error", async () => { - const fetchImpl = vi.fn(async () => - Response.json({ error: { message: "bad key" } }, { status: 401 }), - ); - await expect( - forwardWhisperTranscription( - { - audio: new Uint8Array([1]), - audioMimeType: "audio/webm", - baseUrl: "https://example.com/v1", - model: "whisper-1", - apiKey: "bad", - }, - fetchImpl, - ), - ).resolves.toEqual({ ok: false, status: 502, message: "bad key" }); - }); + it.effect("stops when streamed audio exceeds the limit", () => + Effect.gen(function* () { + const error = yield* readTranscriptionAudio( + Stream.make(new Uint8Array(MAX_TRANSCRIPTION_AUDIO_BYTES), new Uint8Array([1])), + ).pipe(Effect.flip); + expect(error._tag).toBe("TranscriptionAudioTooLargeError"); + }), + ); }); diff --git a/apps/server/src/transcription.ts b/apps/server/src/transcription.ts index 5d380bafa88..3fc111289f8 100644 --- a/apps/server/src/transcription.ts +++ b/apps/server/src/transcription.ts @@ -1,38 +1,127 @@ -const TRANSCRIPTION_PATH = "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/audio/transcriptions"; +import type { VoiceTranscriptionProvider } from "@t3tools/contracts"; +import * as Effect from "effect/Effect"; +import * as Schema from "effect/Schema"; +import * as Stream from "effect/Stream"; +import { HttpClient, HttpClientRequest, HttpClientResponse } from "effect/unstable/http"; export const MAX_TRANSCRIPTION_AUDIO_BYTES = 25 * 1024 * 1024; -export interface WhisperTranscriptionInput { +const PROVIDERS = { + openai: { + endpoint: "https://api.openai.com/v1/audio/transcriptions", + model: "gpt-4o-mini-transcribe", + }, + groq: { + endpoint: "https://api.groq.com/openai/v1/audio/transcriptions", + model: "whisper-large-v3-turbo", + }, +} as const satisfies Record; + +export interface VoiceTranscriptionInput { readonly audio: Uint8Array; readonly audioMimeType: string; - readonly baseUrl: string; - readonly model: string; + readonly provider: VoiceTranscriptionProvider; readonly apiKey: string; } -type TranscriptionFetch = (input: string | URL | Request, init?: RequestInit) => Promise; - -export type WhisperTranscriptionResult = - | { readonly ok: true; readonly text: string } - | { readonly ok: false; readonly status: number; readonly message: string }; - -export function resolveWhisperTranscriptionUrl(baseUrl: string): URL | null { - try { - const url = new URL(baseUrl.trim()); - if (url.protocol !== "http:" && url.protocol !== "https:") return null; - if (url.username || url.password) return null; - url.search = ""; - url.hash = ""; - url.pathname = url.pathname.replace(/\/$/, ""); - if (!url.pathname.endsWith(TRANSCRIPTION_PATH)) { - url.pathname += TRANSCRIPTION_PATH; - } - return url; - } catch { - return null; +export class TranscriptionAudioTooLargeError extends Schema.TaggedErrorClass()( + "TranscriptionAudioTooLargeError", + { receivedBytes: Schema.Number }, +) { + override get message(): string { + return "The recording exceeds the 25 MB limit."; + } +} + +export class TranscriptionBodyReadError extends Schema.TaggedErrorClass()( + "TranscriptionBodyReadError", + { cause: Schema.Defect() }, +) { + override get message(): string { + return "Could not read the recording."; + } +} + +export class TranscriptionInputError extends Schema.TaggedErrorClass()( + "TranscriptionInputError", + { reason: Schema.Literals(["empty_audio", "invalid_provider", "missing_api_key"]) }, +) { + override get message(): string { + if (this.reason === "empty_audio") return "The recording was empty."; + if (this.reason === "invalid_provider") return "Select OpenAI or Groq for transcription."; + return "Add an API key for the selected transcription provider."; + } +} + +export class TranscriptionRequestError extends Schema.TaggedErrorClass()( + "TranscriptionRequestError", + { + provider: Schema.String, + cause: Schema.Defect(), + }, +) { + override get message(): string { + return "Could not reach the transcription provider."; } } +export class TranscriptionProviderError extends Schema.TaggedErrorClass()( + "TranscriptionProviderError", + { + provider: Schema.String, + providerStatus: Schema.Int, + }, +) { + override get message(): string { + return "The transcription provider rejected the request. Check the provider and API key."; + } +} + +export class TranscriptionResponseError extends Schema.TaggedErrorClass()( + "TranscriptionResponseError", + { + provider: Schema.String, + cause: Schema.Defect(), + }, +) { + override get message(): string { + return "The transcription provider returned an invalid response."; + } +} + +const TranscriptionResponse = Schema.Struct({ text: Schema.String }); + +export function resolveTranscriptionProvider(provider: string): VoiceTranscriptionProvider | null { + return provider === "openai" || provider === "groq" ? provider : null; +} + +export function transcriptionProviderConfig(provider: VoiceTranscriptionProvider) { + return PROVIDERS[provider]; +} + +export const readTranscriptionAudio = (stream: Stream.Stream) => + stream.pipe( + Stream.mapError((cause) => new TranscriptionBodyReadError({ cause })), + Stream.runFoldEffect( + () => ({ chunks: [] as Uint8Array[], size: 0 }), + (accumulator, chunk) => { + const size = accumulator.size + chunk.byteLength; + return size > MAX_TRANSCRIPTION_AUDIO_BYTES + ? Effect.fail(new TranscriptionAudioTooLargeError({ receivedBytes: size })) + : Effect.succeed({ chunks: [...accumulator.chunks, chunk], size }); + }, + ), + Effect.map(({ chunks, size }) => { + const audio = new Uint8Array(size); + let offset = 0; + for (const chunk of chunks) { + audio.set(chunk, offset); + offset += chunk.byteLength; + } + return audio; + }), + ); + function audioFileExtension(mimeType: string): string { if (mimeType.includes("ogg")) return "ogg"; if (mimeType.includes("mp4") || mimeType.includes("m4a")) return "m4a"; @@ -40,74 +129,50 @@ function audioFileExtension(mimeType: string): string { return "webm"; } -export async function forwardWhisperTranscription( - input: WhisperTranscriptionInput, - fetchImpl: TranscriptionFetch = globalThis.fetch, -): Promise { - const endpoint = resolveWhisperTranscriptionUrl(input.baseUrl); - if (!endpoint) { - return { ok: false, status: 400, message: "Enter a valid HTTP transcription endpoint." }; - } - if (!input.model.trim()) { - return { ok: false, status: 400, message: "Enter a transcription model." }; - } +export const forwardVoiceTranscription = Effect.fn("voiceTranscription.forward")(function* ( + input: VoiceTranscriptionInput, +) { if (input.audio.byteLength === 0) { - return { ok: false, status: 400, message: "The recording was empty." }; + return yield* new TranscriptionInputError({ reason: "empty_audio" }); } if (input.audio.byteLength > MAX_TRANSCRIPTION_AUDIO_BYTES) { - return { ok: false, status: 413, message: "The recording exceeds the 25 MB limit." }; + return yield* new TranscriptionAudioTooLargeError({ + receivedBytes: input.audio.byteLength, + }); + } + const apiKey = input.apiKey.trim(); + if (!apiKey) { + return yield* new TranscriptionInputError({ reason: "missing_api_key" }); } + const providerConfig = transcriptionProviderConfig(input.provider); const mimeType = input.audioMimeType.split(";", 1)[0]?.trim() || "audio/webm"; const form = new FormData(); - form.set("model", input.model.trim()); + form.set("model", providerConfig.model); form.set( "file", new Blob([input.audio], { type: mimeType }), `recording.${audioFileExtension(mimeType)}`, ); - let response: Response; - try { - response = await fetchImpl(endpoint, { - method: "POST", - headers: input.apiKey ? { authorization: `Bearer ${input.apiKey}` } : undefined, - body: form, - signal: AbortSignal.timeout(120_000), - }); - } catch { - return { ok: false, status: 502, message: "Could not reach the transcription provider." }; - } - - let payload: unknown; - try { - payload = await response.json(); - } catch { - return { ok: false, status: 502, message: "The transcription provider returned invalid JSON." }; - } + const httpClient = yield* HttpClient.HttpClient; + const response = yield* HttpClientRequest.post(providerConfig.endpoint).pipe( + HttpClientRequest.bearerToken(apiKey), + HttpClientRequest.bodyFormData(form), + httpClient.execute, + Effect.timeout("2 minutes"), + Effect.mapError((cause) => new TranscriptionRequestError({ provider: input.provider, cause })), + ); - if (!response.ok) { - const providerMessage = - typeof payload === "object" && - payload !== null && - "error" in payload && - typeof payload.error === "object" && - payload.error !== null && - "message" in payload.error && - typeof payload.error.message === "string" - ? payload.error.message - : null; - return { - ok: false, - status: 502, - message: providerMessage?.slice(0, 300) || "The transcription provider rejected the request.", - }; + if (response.status < 200 || response.status >= 300) { + return yield* new TranscriptionProviderError({ + provider: input.provider, + providerStatus: response.status, + }); } - const text = - typeof payload === "object" && payload !== null && "text" in payload ? payload.text : undefined; - if (typeof text !== "string") { - return { ok: false, status: 502, message: "The transcription response did not contain text." }; - } - return { ok: true, text: text.trim() }; -} + const payload = yield* HttpClientResponse.schemaBodyJson(TranscriptionResponse)(response).pipe( + Effect.mapError((cause) => new TranscriptionResponseError({ provider: input.provider, cause })), + ); + return payload.text.trim(); +}); diff --git a/apps/web/src/components/chat/ChatComposer.tsx b/apps/web/src/components/chat/ChatComposer.tsx index 6823101a706..7027317b55c 100644 --- a/apps/web/src/components/chat/ChatComposer.tsx +++ b/apps/web/src/components/chat/ChatComposer.tsx @@ -1274,8 +1274,6 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) const voiceTranscription = useVoiceTranscription({ config: { provider: settings.voiceTranscriptionProvider, - baseUrl: settings.voiceTranscriptionBaseUrl, - model: settings.voiceTranscriptionModel, apiKey: settings.voiceTranscriptionApiKey, }, onTranscript: appendVoiceTranscript, @@ -3113,7 +3111,7 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps)
- {settings.voiceTranscriptionEnabled && voiceTranscription.status !== "idle" ? ( + {voiceTranscription.status !== "idle" ? ( (() => Array(LEVEL_COUNT).fill(0.12)); const [error, setError] = useState(null); const recorderRef = useRef(null); + const startingRef = useRef(false); const streamRef = useRef(null); const audioContextRef = useRef(null); const intervalsRef = useRef([]); @@ -52,6 +53,7 @@ export function useVoiceTranscription({ mountedRef.current = true; return () => { mountedRef.current = false; + startingRef.current = false; const recorder = recorderRef.current; if (recorder?.state === "recording") recorder.stop(); cleanupCapture(); @@ -64,9 +66,11 @@ export function useVoiceTranscription({ }, []); const start = useCallback(async () => { - if (status !== "idle") return; + if (startingRef.current || status !== "idle") return; + startingRef.current = true; setError(null); if (!navigator.mediaDevices?.getUserMedia || typeof MediaRecorder === "undefined") { + startingRef.current = false; setError("Microphone recording is not supported on this device."); return; } @@ -77,24 +81,31 @@ export function useVoiceTranscription({ }); if (!mountedRef.current) { stream.getTracks().forEach((track) => track.stop()); + startingRef.current = false; return; } + streamRef.current = stream; const mimeType = supportedMimeType(); const recorder = new MediaRecorder(stream, mimeType ? { mimeType } : undefined); const chunks: Blob[] = []; - streamRef.current = stream; + let recordingFailed = false; recorderRef.current = recorder; recorder.addEventListener("dataavailable", (event) => { if (event.data.size > 0) chunks.push(event.data); }); recorder.addEventListener("error", () => { - if (mountedRef.current) setError("The microphone stopped unexpectedly."); + recordingFailed = true; + cleanupCapture(); + if (mountedRef.current) { + setStatus("idle"); + setError("The microphone stopped unexpectedly."); + } }); recorder.addEventListener("stop", () => { const blob = new Blob(chunks, { type: recorder.mimeType || mimeType || "audio/webm" }); cleanupCapture(); - if (!mountedRef.current) return; + if (!mountedRef.current || recordingFailed) return; setStatus("transcribing"); void transcribeVoiceRecording(blob, configRef.current) .then((text) => { @@ -133,8 +144,10 @@ export function useVoiceTranscription({ ); timeoutRef.current = window.setTimeout(() => recorder.stop(), MAX_RECORDING_MS); recorder.start(250); + startingRef.current = false; setStatus("recording"); } catch (cause) { + startingRef.current = false; cleanupCapture(); setStatus("idle"); setError( diff --git a/apps/web/src/lib/voiceTranscription.ts b/apps/web/src/lib/voiceTranscription.ts index 05499b4c4fc..8fcee72cb17 100644 --- a/apps/web/src/lib/voiceTranscription.ts +++ b/apps/web/src/lib/voiceTranscription.ts @@ -3,13 +3,8 @@ import type { VoiceTranscriptionProvider } from "@t3tools/contracts"; import { readDesktopPrimaryBearerToken } from "../environments/primary/desktopAuth"; import { resolvePrimaryEnvironmentHttpUrl } from "../environments/primary/target"; -export const GROQ_TRANSCRIPTION_BASE_URL = "https://api.groq.com/openai/v1"; -export const GROQ_TRANSCRIPTION_MODEL = "whisper-large-v3-turbo"; - export interface VoiceTranscriptionConfig { readonly provider: VoiceTranscriptionProvider; - readonly baseUrl: string; - readonly model: string; readonly apiKey: string; } @@ -17,6 +12,9 @@ export async function transcribeVoiceRecording( audio: Blob, config: VoiceTranscriptionConfig, ): Promise { + const apiKey = config.apiKey.trim(); + if (!apiKey) throw new Error("Add an API key for the selected transcription provider."); + const bearerToken = await readDesktopPrimaryBearerToken(); const response = await globalThis.fetch(resolvePrimaryEnvironmentHttpUrl("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/api/transcription"), { method: "POST", @@ -24,9 +22,8 @@ export async function transcribeVoiceRecording( headers: { ...(bearerToken ? { authorization: `Bearer ${bearerToken}` } : {}), "content-type": audio.type || "audio/webm", - "x-t3-transcription-base-url": config.baseUrl, - "x-t3-transcription-model": config.model, - ...(config.apiKey ? { "x-t3-transcription-api-key": config.apiKey } : {}), + "x-t3-transcription-provider": config.provider, + "x-t3-transcription-api-key": apiKey, }, body: audio, }); diff --git a/packages/contracts/src/settings.ts b/packages/contracts/src/settings.ts index bc7fff64189..7fb174df0e5 100644 --- a/packages/contracts/src/settings.ts +++ b/packages/contracts/src/settings.ts @@ -111,10 +111,8 @@ export const DEFAULT_ENVIRONMENT_IDENTIFICATION_MODE: EnvironmentIdentificationM export const FontFamilyPreference = Schema.String.check(Schema.isMaxLength(200)); export type FontFamilyPreference = typeof FontFamilyPreference.Type; -export const VoiceTranscriptionProvider = Schema.Literals(["local", "groq", "custom"]); +export const VoiceTranscriptionProvider = Schema.Literals(["openai", "groq"]); export type VoiceTranscriptionProvider = typeof VoiceTranscriptionProvider.Type; -export const DEFAULT_VOICE_TRANSCRIPTION_BASE_URL = "http://127.0.0.1:8080/v1"; -export const DEFAULT_VOICE_TRANSCRIPTION_MODEL = "whisper-1"; export const ClientSettingsSchema = Schema.Struct({ confirmThreadArchive: Schema.Boolean.pipe(Schema.withDecodingDefault(Effect.succeed(false))), @@ -207,15 +205,9 @@ export const ClientSettingsSchema = Schema.Struct({ ), voiceTranscriptionEnabled: Schema.Boolean.pipe(Schema.withDecodingDefault(Effect.succeed(false))), voiceTranscriptionProvider: VoiceTranscriptionProvider.pipe( - Schema.withDecodingDefault(Effect.succeed("local" as const)), + Schema.withDecodingDefault(Effect.succeed("openai" as const)), ), - voiceTranscriptionBaseUrl: TrimmedString.pipe( - Schema.withDecodingDefault(Effect.succeed(DEFAULT_VOICE_TRANSCRIPTION_BASE_URL)), - ), - voiceTranscriptionModel: TrimmedString.pipe( - Schema.withDecodingDefault(Effect.succeed(DEFAULT_VOICE_TRANSCRIPTION_MODEL)), - ), - voiceTranscriptionApiKey: Schema.String.pipe(Schema.withDecodingDefault(Effect.succeed(""))), + voiceTranscriptionApiKey: TrimmedString.pipe(Schema.withDecodingDefault(Effect.succeed(""))), wordWrap: Schema.Boolean.pipe(Schema.withDecodingDefault(Effect.succeed(true))), }); export type ClientSettings = typeof ClientSettingsSchema.Type; @@ -821,9 +813,7 @@ export const ClientSettingsPatch = Schema.Struct({ timestampFormat: Schema.optionalKey(TimestampFormat), voiceTranscriptionEnabled: Schema.optionalKey(Schema.Boolean), voiceTranscriptionProvider: Schema.optionalKey(VoiceTranscriptionProvider), - voiceTranscriptionBaseUrl: Schema.optionalKey(TrimmedString), - voiceTranscriptionModel: Schema.optionalKey(TrimmedString), - voiceTranscriptionApiKey: Schema.optionalKey(Schema.String), + voiceTranscriptionApiKey: Schema.optionalKey(TrimmedString), wordWrap: Schema.optionalKey(Schema.Boolean), }); export type ClientSettingsPatch = typeof ClientSettingsPatch.Type; From 1f8a731730e73b1f01ef5d9d1725aeaccdb78ac2 Mon Sep 17 00:00:00 2001 From: "t3-code[bot]" <269035359+t3-code[bot]@users.noreply.github.com> Date: Sun, 2 Aug 2026 04:56:48 +0000 Subject: [PATCH 03/21] fix: split transcription input errors --- apps/server/src/http.ts | 12 +++++++++--- apps/server/src/transcription.ts | 30 +++++++++++++++++++++++------- 2 files changed, 32 insertions(+), 10 deletions(-) diff --git a/apps/server/src/http.ts b/apps/server/src/http.ts index bfe16f38a51..827941089c4 100644 --- a/apps/server/src/http.ts +++ b/apps/server/src/http.ts @@ -44,7 +44,9 @@ import { MAX_TRANSCRIPTION_AUDIO_BYTES, readTranscriptionAudio, resolveTranscriptionProvider, - TranscriptionInputError, + TranscriptionApiKeyMissingError, + TranscriptionEmptyAudioError, + TranscriptionProviderUnsupportedError, } from "./transcription.ts"; const OTLP_TRACES_PROXY_PATH = "/api/observability/v1/traces"; @@ -220,7 +222,7 @@ export const transcriptionRouteLayer = HttpRouter.add( request.headers["x-t3-transcription-provider"] ?? "", ); if (!provider) { - return yield* new TranscriptionInputError({ reason: "invalid_provider" }); + return yield* new TranscriptionProviderUnsupportedError(); } const audio = yield* readTranscriptionAudio(request.stream); @@ -238,12 +240,16 @@ export const transcriptionRouteLayer = HttpRouter.add( EnvironmentScopeRequiredError: HttpServerRespondable.toResponse, TranscriptionAudioTooLargeError: (error) => Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 413 })), + TranscriptionApiKeyMissingError: (error) => + Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })), TranscriptionBodyReadError: (error) => Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })), - TranscriptionInputError: (error) => + TranscriptionEmptyAudioError: (error) => Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })), TranscriptionProviderError: (error) => Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })), + TranscriptionProviderUnsupportedError: (error) => + Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })), TranscriptionRequestError: (error) => Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })), TranscriptionResponseError: (error) => diff --git a/apps/server/src/transcription.ts b/apps/server/src/transcription.ts index 3fc111289f8..ee488408be3 100644 --- a/apps/server/src/transcription.ts +++ b/apps/server/src/transcription.ts @@ -42,13 +42,29 @@ export class TranscriptionBodyReadError extends Schema.TaggedErrorClass()( - "TranscriptionInputError", - { reason: Schema.Literals(["empty_audio", "invalid_provider", "missing_api_key"]) }, +export class TranscriptionEmptyAudioError extends Schema.TaggedErrorClass()( + "TranscriptionEmptyAudioError", + {}, +) { + override get message(): string { + return "The recording was empty."; + } +} + +export class TranscriptionProviderUnsupportedError extends Schema.TaggedErrorClass()( + "TranscriptionProviderUnsupportedError", + {}, +) { + override get message(): string { + return "Select OpenAI or Groq for transcription."; + } +} + +export class TranscriptionApiKeyMissingError extends Schema.TaggedErrorClass()( + "TranscriptionApiKeyMissingError", + {}, ) { override get message(): string { - if (this.reason === "empty_audio") return "The recording was empty."; - if (this.reason === "invalid_provider") return "Select OpenAI or Groq for transcription."; return "Add an API key for the selected transcription provider."; } } @@ -133,7 +149,7 @@ export const forwardVoiceTranscription = Effect.fn("voiceTranscription.forward") input: VoiceTranscriptionInput, ) { if (input.audio.byteLength === 0) { - return yield* new TranscriptionInputError({ reason: "empty_audio" }); + return yield* new TranscriptionEmptyAudioError(); } if (input.audio.byteLength > MAX_TRANSCRIPTION_AUDIO_BYTES) { return yield* new TranscriptionAudioTooLargeError({ @@ -142,7 +158,7 @@ export const forwardVoiceTranscription = Effect.fn("voiceTranscription.forward") } const apiKey = input.apiKey.trim(); if (!apiKey) { - return yield* new TranscriptionInputError({ reason: "missing_api_key" }); + return yield* new TranscriptionApiKeyMissingError(); } const providerConfig = transcriptionProviderConfig(input.provider); From 64d1eba85bcf4bce5b5672f3b05ae8a245685c64 Mon Sep 17 00:00:00 2001 From: "t3-code[bot]" <269035359+t3-code[bot]@users.noreply.github.com> Date: Sun, 2 Aug 2026 05:05:42 +0000 Subject: [PATCH 04/21] fix: bound transcription response processing --- apps/server/src/transcription.ts | 41 +++++++++++++++++++------------- 1 file changed, 25 insertions(+), 16 deletions(-) diff --git a/apps/server/src/transcription.ts b/apps/server/src/transcription.ts index ee488408be3..dfdea7415a0 100644 --- a/apps/server/src/transcription.ts +++ b/apps/server/src/transcription.ts @@ -122,9 +122,11 @@ export const readTranscriptionAudio = (stream: Stream.Stream ({ chunks: [] as Uint8Array[], size: 0 }), (accumulator, chunk) => { const size = accumulator.size + chunk.byteLength; - return size > MAX_TRANSCRIPTION_AUDIO_BYTES - ? Effect.fail(new TranscriptionAudioTooLargeError({ receivedBytes: size })) - : Effect.succeed({ chunks: [...accumulator.chunks, chunk], size }); + if (size > MAX_TRANSCRIPTION_AUDIO_BYTES) { + return Effect.fail(new TranscriptionAudioTooLargeError({ receivedBytes: size })); + } + accumulator.chunks.push(chunk); + return Effect.succeed({ chunks: accumulator.chunks, size }); }, ), Effect.map(({ chunks, size }) => { @@ -172,23 +174,30 @@ export const forwardVoiceTranscription = Effect.fn("voiceTranscription.forward") ); const httpClient = yield* HttpClient.HttpClient; - const response = yield* HttpClientRequest.post(providerConfig.endpoint).pipe( + const payload = yield* HttpClientRequest.post(providerConfig.endpoint).pipe( HttpClientRequest.bearerToken(apiKey), HttpClientRequest.bodyFormData(form), httpClient.execute, - Effect.timeout("2 minutes"), Effect.mapError((cause) => new TranscriptionRequestError({ provider: input.provider, cause })), - ); - - if (response.status < 200 || response.status >= 300) { - return yield* new TranscriptionProviderError({ - provider: input.provider, - providerStatus: response.status, - }); - } - - const payload = yield* HttpClientResponse.schemaBodyJson(TranscriptionResponse)(response).pipe( - Effect.mapError((cause) => new TranscriptionResponseError({ provider: input.provider, cause })), + Effect.flatMap((response) => + Effect.gen(function* () { + if (response.status < 200 || response.status >= 300) { + return yield* new TranscriptionProviderError({ + provider: input.provider, + providerStatus: response.status, + }); + } + return yield* HttpClientResponse.schemaBodyJson(TranscriptionResponse)(response).pipe( + Effect.mapError( + (cause) => new TranscriptionResponseError({ provider: input.provider, cause }), + ), + ); + }), + ), + Effect.timeout("2 minutes"), + Effect.catchTag("TimeoutError", (cause) => + Effect.fail(new TranscriptionRequestError({ provider: input.provider, cause })), + ), ); return payload.text.trim(); }); From 521fa697feaa594454c548bf2af7ab43db356ca2 Mon Sep 17 00:00:00 2001 From: "t3-code[bot]" <269035359+t3-code[bot]@users.noreply.github.com> Date: Sun, 2 Aug 2026 05:09:45 +0000 Subject: [PATCH 05/21] fix: follow effect recovery conventions --- apps/server/src/transcription.ts | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/apps/server/src/transcription.ts b/apps/server/src/transcription.ts index dfdea7415a0..f5ecc6033e1 100644 --- a/apps/server/src/transcription.ts +++ b/apps/server/src/transcription.ts @@ -195,9 +195,10 @@ export const forwardVoiceTranscription = Effect.fn("voiceTranscription.forward") }), ), Effect.timeout("2 minutes"), - Effect.catchTag("TimeoutError", (cause) => - Effect.fail(new TranscriptionRequestError({ provider: input.provider, cause })), - ), + Effect.catchTags({ + TimeoutError: (cause) => + Effect.fail(new TranscriptionRequestError({ provider: input.provider, cause })), + }), ); return payload.text.trim(); }); From 04d18fb2164baccca9865cad3c7d22b79e98cfc5 Mon Sep 17 00:00:00 2001 From: maria-rcks Date: Sun, 2 Aug 2026 19:49:34 -0400 Subject: [PATCH 06/21] feat: improve voice dictation configuration and recording --- .../settings/DesktopClientSettings.test.ts | 1 + apps/server/src/http.ts | 71 +++++++++++- apps/server/src/httpCors.ts | 1 + apps/server/src/server.test.ts | 2 + apps/server/src/transcription.test.ts | 84 ++++++++++++-- apps/server/src/transcription.ts | 106 ++++++++++++++++-- apps/web/src/components/chat/ChatComposer.tsx | 56 ++++++--- .../chat/VoiceTranscriptionPanel.tsx | 67 +++++------ apps/web/src/hooks/useVoiceTranscription.ts | 71 +++++++++--- apps/web/src/lib/voiceTranscription.test.ts | 70 ++++++++++++ apps/web/src/lib/voiceTranscription.ts | 87 +++++++++++++- docs/README.md | 1 + docs/user/voice-dictation.md | 28 +++++ packages/contracts/src/settings.test.ts | 16 +++ packages/contracts/src/settings.ts | 2 + 15 files changed, 574 insertions(+), 89 deletions(-) create mode 100644 apps/web/src/lib/voiceTranscription.test.ts create mode 100644 docs/user/voice-dictation.md diff --git a/apps/desktop/src/settings/DesktopClientSettings.test.ts b/apps/desktop/src/settings/DesktopClientSettings.test.ts index e7e47d9948a..4d781f302da 100644 --- a/apps/desktop/src/settings/DesktopClientSettings.test.ts +++ b/apps/desktop/src/settings/DesktopClientSettings.test.ts @@ -45,6 +45,7 @@ const clientSettings: ClientSettings = { voiceTranscriptionEnabled: true, voiceTranscriptionProvider: "openai", voiceTranscriptionApiKey: "", + voiceTranscriptionModel: "", wordWrap: true, }; diff --git a/apps/server/src/http.ts b/apps/server/src/http.ts index 827941089c4..1f9eecb09ee 100644 --- a/apps/server/src/http.ts +++ b/apps/server/src/http.ts @@ -41,16 +41,17 @@ import * as ServerEnvironment from "./environment/ServerEnvironment.ts"; import { browserApiCorsAllowedHeaders, browserApiCorsAllowedMethods } from "./httpCors.ts"; import { forwardVoiceTranscription, + listVoiceTranscriptionModels, MAX_TRANSCRIPTION_AUDIO_BYTES, readTranscriptionAudio, resolveTranscriptionProvider, - TranscriptionApiKeyMissingError, - TranscriptionEmptyAudioError, + transcriptionEnvironmentApiKeyStatus, TranscriptionProviderUnsupportedError, } from "./transcription.ts"; const OTLP_TRACES_PROXY_PATH = "/api/observability/v1/traces"; const TRANSCRIPTION_PATH = "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/api/transcription"; +const TRANSCRIPTION_MODELS_PATH = "/api/transcription/models"; const LOOPBACK_HOSTNAMES = new Set(["127.0.0.1", "::1", "localhost"]); const DESKTOP_RENDERER_ORIGINS = ["t3code://app", "t3code-dev://app"]; const SVG_CONTENT_SECURITY_POLICY = "default-src 'none'; style-src 'unsafe-inline'; sandbox"; @@ -204,7 +205,26 @@ export const otlpTracesProxyRouteLayer = HttpRouter.add( ), ); -export const transcriptionRouteLayer = HttpRouter.add( +const transcriptionConfigRouteLayer = HttpRouter.add( + "GET", + TRANSCRIPTION_PATH, + Effect.gen(function* () { + yield* authenticateRawRouteWithScope(AuthOrchestrationOperateScope); + const [openai, groq] = yield* Effect.all([ + transcriptionEnvironmentApiKeyStatus("openai"), + transcriptionEnvironmentApiKeyStatus("groq"), + ]); + return HttpServerResponse.jsonUnsafe({ openai, groq }); + }).pipe( + Effect.catchTags({ + EnvironmentAuthInvalidError: HttpServerRespondable.toResponse, + EnvironmentInternalError: HttpServerRespondable.toResponse, + EnvironmentScopeRequiredError: HttpServerRespondable.toResponse, + }), + ), +); + +const transcriptionUploadRouteLayer = HttpRouter.add( "POST", TRANSCRIPTION_PATH, Effect.gen(function* () { @@ -231,6 +251,7 @@ export const transcriptionRouteLayer = HttpRouter.add( audioMimeType: request.headers["content-type"] ?? "audio/webm", provider, apiKey: request.headers["x-t3-transcription-api-key"] ?? "", + model: request.headers["x-t3-transcription-model"] ?? "", }); return HttpServerResponse.jsonUnsafe({ text }); }).pipe( @@ -246,6 +267,44 @@ export const transcriptionRouteLayer = HttpRouter.add( Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })), TranscriptionEmptyAudioError: (error) => Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })), + TranscriptionModelMissingError: (error) => + Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })), + TranscriptionProviderError: (error) => + Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })), + TranscriptionProviderUnsupportedError: (error) => + Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })), + TranscriptionRequestError: (error) => + Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })), + TranscriptionResponseError: (error) => + Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })), + }), + ), +); + +const transcriptionModelsRouteLayer = HttpRouter.add( + "GET", + TRANSCRIPTION_MODELS_PATH, + Effect.gen(function* () { + yield* authenticateRawRouteWithScope(AuthOrchestrationOperateScope); + const request = yield* HttpServerRequest.HttpServerRequest; + const provider = resolveTranscriptionProvider( + request.headers["x-t3-transcription-provider"] ?? "", + ); + if (!provider) { + return yield* new TranscriptionProviderUnsupportedError(); + } + const models = yield* listVoiceTranscriptionModels({ + provider, + apiKey: request.headers["x-t3-transcription-api-key"] ?? "", + }); + return HttpServerResponse.jsonUnsafe({ models }); + }).pipe( + Effect.catchTags({ + EnvironmentAuthInvalidError: HttpServerRespondable.toResponse, + EnvironmentInternalError: HttpServerRespondable.toResponse, + EnvironmentScopeRequiredError: HttpServerRespondable.toResponse, + TranscriptionApiKeyMissingError: (error) => + Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 400 })), TranscriptionProviderError: (error) => Effect.succeed(HttpServerResponse.jsonUnsafe({ error: error.message }, { status: 502 })), TranscriptionProviderUnsupportedError: (error) => @@ -258,6 +317,12 @@ export const transcriptionRouteLayer = HttpRouter.add( ), ); +export const transcriptionRouteLayer = Layer.mergeAll( + transcriptionConfigRouteLayer, + transcriptionModelsRouteLayer, + transcriptionUploadRouteLayer, +); + export const assetRouteLayer = HttpRouter.add( "GET", `${ASSET_ROUTE_PREFIX}/*`, diff --git a/apps/server/src/httpCors.ts b/apps/server/src/httpCors.ts index a29d8f271d3..6212a958e16 100644 --- a/apps/server/src/httpCors.ts +++ b/apps/server/src/httpCors.ts @@ -6,6 +6,7 @@ export const browserApiCorsAllowedHeaders = [ "content-type", "dpop", "x-t3-transcription-api-key", + "x-t3-transcription-model", "x-t3-transcription-provider", ] as const; diff --git a/apps/server/src/server.test.ts b/apps/server/src/server.test.ts index 60d1984edca..b53791bb29b 100644 --- a/apps/server/src/server.test.ts +++ b/apps/server/src/server.test.ts @@ -1358,6 +1358,7 @@ const assertBrowserApiCorsPreflightHeaders = ( "dpop", "traceparent", "x-t3-transcription-api-key", + "x-t3-transcription-model", "x-t3-transcription-provider", ]); }; @@ -4291,6 +4292,7 @@ it.layer(NodeServices.layer)("server router seam", (it) => { "dpop", "traceparent", "x-t3-transcription-api-key", + "x-t3-transcription-model", "x-t3-transcription-provider", ]); }).pipe(Effect.provide(NodeHttpServer.layerTest)), diff --git a/apps/server/src/transcription.test.ts b/apps/server/src/transcription.test.ts index 3bbd526e0f5..85e954eec89 100644 --- a/apps/server/src/transcription.test.ts +++ b/apps/server/src/transcription.test.ts @@ -1,32 +1,102 @@ import { expect, it } from "@effect/vitest"; +import * as ConfigProvider from "effect/ConfigProvider"; import * as Effect from "effect/Effect"; import * as Stream from "effect/Stream"; +import { HttpClient, HttpClientRequest, HttpClientResponse } from "effect/unstable/http"; import { describe } from "vite-plus/test"; import { + forwardVoiceTranscription, + listVoiceTranscriptionModels, MAX_TRANSCRIPTION_AUDIO_BYTES, readTranscriptionAudio, resolveTranscriptionProvider, + transcriptionEnvironmentApiKeyStatus, transcriptionProviderConfig, } from "./transcription.ts"; describe("transcription providers", () => { - it("uses a fixed OpenAI endpoint and model", () => { + it("uses fixed OpenAI and Groq configurations", () => { expect(resolveTranscriptionProvider("openai")).toBe("openai"); expect(transcriptionProviderConfig("openai")).toEqual({ endpoint: "https://api.openai.com/v1/audio/transcriptions", - model: "gpt-4o-mini-transcribe", + modelsEndpoint: "https://api.openai.com/v1/models", + apiKeyEnvironmentVariable: "OPENAI_API_KEY", }); - }); - - it("uses a fixed Groq endpoint and rejects custom providers", () => { expect(resolveTranscriptionProvider("groq")).toBe("groq"); expect(transcriptionProviderConfig("groq")).toEqual({ endpoint: "https://api.groq.com/openai/v1/audio/transcriptions", - model: "whisper-large-v3-turbo", + modelsEndpoint: "https://api.groq.com/openai/v1/models", + apiKeyEnvironmentVariable: "GROQ_API_KEY", }); - expect(resolveTranscriptionProvider("http://127.0.0.1:8080/v1")).toBeNull(); + expect(resolveTranscriptionProvider("custom")).toBeNull(); }); + + it.effect("loads accessible transcription models with the provider API key", () => + Effect.gen(function* () { + let capturedRequest: HttpClientRequest.HttpClientRequest | undefined; + const client = HttpClient.make((request) => + Effect.sync(() => { + capturedRequest = request; + return HttpClientResponse.fromWeb( + request, + Response.json({ + data: [ + { id: "gpt-4o" }, + { id: " whisper-1 " }, + { id: "gpt-4o-mini-transcribe" }, + { id: "gpt-4o-mini-transcribe" }, + ], + }), + ); + }), + ); + + const models = yield* listVoiceTranscriptionModels({ + provider: "openai", + apiKey: "client-openai-key", + }).pipe(Effect.provideService(HttpClient.HttpClient, client)); + + expect(models).toEqual(["gpt-4o-mini-transcribe", "whisper-1"]); + expect(capturedRequest?.url).toBe("https://api.openai.com/v1/models"); + expect(capturedRequest?.headers.authorization).toBe("Bearer client-openai-key"); + }), + ); + + it.effect("uses a provider API key from the server environment", () => + Effect.gen(function* () { + let capturedRequest: HttpClientRequest.HttpClientRequest | undefined; + const client = HttpClient.make((request) => + Effect.sync(() => { + capturedRequest = request; + return HttpClientResponse.fromWeb(request, Response.json({ text: " groq transcript " })); + }), + ); + + const transcript = yield* forwardVoiceTranscription({ + audio: new Uint8Array([1, 2, 3]), + audioMimeType: "audio/webm;codecs=opus", + provider: "groq", + apiKey: "", + model: "whisper-large-v3", + }).pipe(Effect.provideService(HttpClient.HttpClient, client)); + + expect(yield* transcriptionEnvironmentApiKeyStatus("groq")).toBe(true); + expect(yield* transcriptionEnvironmentApiKeyStatus("openai")).toBe(false); + expect(transcript).toBe("groq transcript"); + expect(capturedRequest?.url).toBe("https://api.groq.com/openai/v1/audio/transcriptions"); + expect(capturedRequest?.headers.authorization).toBe("Bearer env-groq-key"); + expect(capturedRequest?.body._tag).toBe("FormData"); + if (capturedRequest?.body._tag === "FormData") { + expect(capturedRequest.body.formData.get("model")).toBe("whisper-large-v3"); + expect(capturedRequest.body.formData.get("file")).toBeInstanceOf(Blob); + } + }).pipe( + Effect.provide( + ConfigProvider.layer(ConfigProvider.fromEnv({ env: { GROQ_API_KEY: "env-groq-key" } })), + ), + ), + ); }); describe("readTranscriptionAudio", () => { diff --git a/apps/server/src/transcription.ts b/apps/server/src/transcription.ts index f5ecc6033e1..1b167c18123 100644 --- a/apps/server/src/transcription.ts +++ b/apps/server/src/transcription.ts @@ -1,4 +1,5 @@ import type { VoiceTranscriptionProvider } from "@t3tools/contracts"; +import * as Config from "effect/Config"; import * as Effect from "effect/Effect"; import * as Schema from "effect/Schema"; import * as Stream from "effect/Stream"; @@ -9,19 +10,30 @@ export const MAX_TRANSCRIPTION_AUDIO_BYTES = 25 * 1024 * 1024; const PROVIDERS = { openai: { endpoint: "https://api.openai.com/v1/audio/transcriptions", - model: "gpt-4o-mini-transcribe", + modelsEndpoint: "https://api.openai.com/v1/models", + apiKeyEnvironmentVariable: "OPENAI_API_KEY", }, groq: { endpoint: "https://api.groq.com/openai/v1/audio/transcriptions", - model: "whisper-large-v3-turbo", + modelsEndpoint: "https://api.groq.com/openai/v1/models", + apiKeyEnvironmentVariable: "GROQ_API_KEY", }, -} as const satisfies Record; +} as const satisfies Record< + VoiceTranscriptionProvider, + { endpoint: string; modelsEndpoint: string; apiKeyEnvironmentVariable: string } +>; export interface VoiceTranscriptionInput { readonly audio: Uint8Array; readonly audioMimeType: string; readonly provider: VoiceTranscriptionProvider; readonly apiKey: string; + readonly model: string; +} + +export interface VoiceTranscriptionModelsInput { + readonly provider: VoiceTranscriptionProvider; + readonly apiKey: string; } export class TranscriptionAudioTooLargeError extends Schema.TaggedErrorClass()( @@ -65,7 +77,16 @@ export class TranscriptionApiKeyMissingError extends Schema.TaggedErrorClass()( + "TranscriptionModelMissingError", + {}, +) { + override get message(): string { + return "Select a transcription model."; } } @@ -106,6 +127,10 @@ export class TranscriptionResponseError extends Schema.TaggedErrorClass 0; +}); + +const resolveTranscriptionApiKey = Effect.fn("voiceTranscription.resolveApiKey")(function* ( + input: VoiceTranscriptionModelsInput, +) { + const providerConfig = transcriptionProviderConfig(input.provider); + const apiKey = + input.apiKey.trim() || + (yield* Config.string(providerConfig.apiKeyEnvironmentVariable).pipe( + Config.withDefault(""), + )).trim(); + if (!apiKey) { + return yield* new TranscriptionApiKeyMissingError(); + } + return apiKey; +}); + +export const listVoiceTranscriptionModels = Effect.fn("voiceTranscription.listModels")(function* ( + input: VoiceTranscriptionModelsInput, +) { + const providerConfig = transcriptionProviderConfig(input.provider); + const apiKey = yield* resolveTranscriptionApiKey(input); + const httpClient = yield* HttpClient.HttpClient; + const payload = yield* HttpClientRequest.get(providerConfig.modelsEndpoint).pipe( + HttpClientRequest.bearerToken(apiKey), + httpClient.execute, + Effect.mapError((cause) => new TranscriptionRequestError({ provider: input.provider, cause })), + Effect.flatMap((response) => + Effect.gen(function* () { + if (response.status < 200 || response.status >= 300) { + return yield* new TranscriptionProviderError({ + provider: input.provider, + providerStatus: response.status, + }); + } + return yield* HttpClientResponse.schemaBodyJson(ModelsResponse)(response).pipe( + Effect.mapError( + (cause) => new TranscriptionResponseError({ provider: input.provider, cause }), + ), + ); + }), + ), + Effect.timeout("30 seconds"), + Effect.catchTags({ + TimeoutError: (cause) => + Effect.fail(new TranscriptionRequestError({ provider: input.provider, cause })), + }), + ); + + const models = [...new Set(payload.data.map(({ id }) => id.trim()).filter(Boolean))].sort(); + const transcriptionModels = models.filter((model) => TRANSCRIPTION_MODEL_ID_PATTERN.test(model)); + return transcriptionModels.length > 0 ? transcriptionModels : models; +}); + export const readTranscriptionAudio = (stream: Stream.Stream) => stream.pipe( Stream.mapError((cause) => new TranscriptionBodyReadError({ cause })), @@ -158,15 +245,16 @@ export const forwardVoiceTranscription = Effect.fn("voiceTranscription.forward") receivedBytes: input.audio.byteLength, }); } - const apiKey = input.apiKey.trim(); - if (!apiKey) { - return yield* new TranscriptionApiKeyMissingError(); + const providerConfig = transcriptionProviderConfig(input.provider); + const model = input.model.trim(); + if (!model) { + return yield* new TranscriptionModelMissingError(); } + const apiKey = yield* resolveTranscriptionApiKey(input); - const providerConfig = transcriptionProviderConfig(input.provider); const mimeType = input.audioMimeType.split(";", 1)[0]?.trim() || "audio/webm"; const form = new FormData(); - form.set("model", providerConfig.model); + form.set("model", model); form.set( "file", new Blob([input.audio], { type: mimeType }), diff --git a/apps/web/src/components/chat/ChatComposer.tsx b/apps/web/src/components/chat/ChatComposer.tsx index 7027317b55c..8631283c643 100644 --- a/apps/web/src/components/chat/ChatComposer.tsx +++ b/apps/web/src/components/chat/ChatComposer.tsx @@ -203,6 +203,7 @@ import { PenLineIcon, SparklesIcon, MicIcon, + SquareIcon, XIcon, } from "lucide-react"; import { proposedPlanTitle } from "../../proposedPlan"; @@ -1275,6 +1276,7 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) config: { provider: settings.voiceTranscriptionProvider, apiKey: settings.voiceTranscriptionApiKey, + model: settings.voiceTranscriptionModel, }, onTranscript: appendVoiceTranscript, }); @@ -3111,14 +3113,9 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) - {voiceTranscription.status !== "idle" ? ( - - ) : settings.voiceTranscriptionEnabled && voiceTranscription.error ? ( + {voiceTranscription.status === "idle" && + settings.voiceTranscriptionEnabled && + voiceTranscription.error ? (

{voiceTranscription.error}

@@ -3144,7 +3141,20 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) showMobilePendingAnswerActions && "hidden sm:flex", )} > -
+ {voiceTranscription.status !== "idle" ? ( + + ) : null} +
{noProviderAvailable ? ( } /> - Start dictation + + {voiceTranscription.status === "recording" + ? "Stop dictation" + : "Start dictation"} + ) : null} `voice-waveform-${index}`); function formatElapsed(elapsedMs: number): string { const seconds = Math.floor(elapsedMs / 1_000); @@ -14,55 +11,61 @@ export function VoiceTranscriptionPanel({ status, elapsedMs, levels, - onStop, }: { readonly status: Exclude; readonly elapsedMs: number; readonly levels: readonly number[]; - readonly onStop: () => void; }) { + const waveformPath = levels + .map((level, index) => { + if (level <= 0.01) return ""; + const amplitude = Math.max(1.75, level * 14); + return `M ${index + 0.5} ${16 - amplitude} V ${16 + amplitude}`; + }) + .join(" "); + return (
{status === "recording" ? ( - + + + ) : (
- transcribing with your configured provider... + Transcribing…
)} {formatElapsed(elapsedMs)} - {status === "recording" ? ( - - ) : null}
); } diff --git a/apps/web/src/hooks/useVoiceTranscription.ts b/apps/web/src/hooks/useVoiceTranscription.ts index 4902b46399f..22e776dafc7 100644 --- a/apps/web/src/hooks/useVoiceTranscription.ts +++ b/apps/web/src/hooks/useVoiceTranscription.ts @@ -2,9 +2,10 @@ import { useCallback, useEffect, useRef, useState } from "react"; import { transcribeVoiceRecording, type VoiceTranscriptionConfig } from "../lib/voiceTranscription"; -const LEVEL_COUNT = 36; +const LEVEL_COUNT = 160; const MAX_RECORDING_MS = 5 * 60 * 1_000; const MIME_TYPES = ["audio/webm;codecs=opus", "audio/ogg;codecs=opus", "audio/mp4"]; +const FLAT_LEVELS = Array(LEVEL_COUNT).fill(0); export type VoiceTranscriptionStatus = "idle" | "recording" | "transcribing"; @@ -22,10 +23,11 @@ export function useVoiceTranscription({ }) { const [status, setStatus] = useState("idle"); const [elapsedMs, setElapsedMs] = useState(0); - const [levels, setLevels] = useState(() => Array(LEVEL_COUNT).fill(0.12)); + const [levels, setLevels] = useState(FLAT_LEVELS); const [error, setError] = useState(null); const recorderRef = useRef(null); const startingRef = useRef(false); + const cancelStartingRef = useRef(false); const streamRef = useRef(null); const audioContextRef = useRef(null); const intervalsRef = useRef([]); @@ -54,6 +56,7 @@ export function useVoiceTranscription({ return () => { mountedRef.current = false; startingRef.current = false; + cancelStartingRef.current = true; const recorder = recorderRef.current; if (recorder?.state === "recording") recorder.stop(); cleanupCapture(); @@ -61,27 +64,51 @@ export function useVoiceTranscription({ }, [cleanupCapture]); const stop = useCallback(() => { + if (startingRef.current) { + cancelStartingRef.current = true; + cleanupCapture(); + if (mountedRef.current) { + setStatus("idle"); + setElapsedMs(0); + } + return; + } const recorder = recorderRef.current; if (recorder?.state === "recording") recorder.stop(); - }, []); + }, [cleanupCapture]); const start = useCallback(async () => { if (startingRef.current || status !== "idle") return; startingRef.current = true; setError(null); + setElapsedMs(0); + setLevels(FLAT_LEVELS); + if (!configRef.current.model.trim()) { + startingRef.current = false; + setError("Select a transcription model in Settings."); + return; + } if (!navigator.mediaDevices?.getUserMedia || typeof MediaRecorder === "undefined") { startingRef.current = false; setError("Microphone recording is not supported on this device."); return; } + cancelStartingRef.current = false; + setStatus("recording"); try { + const audioContext = new AudioContext(); + audioContextRef.current = audioContext; + if (audioContext.state === "suspended") { + void audioContext.resume().catch(() => undefined); + } const stream = await navigator.mediaDevices.getUserMedia({ audio: { echoCancellation: true, noiseSuppression: true, autoGainControl: true }, }); - if (!mountedRef.current) { + if (!mountedRef.current || cancelStartingRef.current) { stream.getTracks().forEach((track) => track.stop()); startingRef.current = false; + cancelStartingRef.current = false; return; } @@ -121,22 +148,29 @@ export function useVoiceTranscription({ }); }); - const audioContext = new AudioContext(); - audioContextRef.current = audioContext; const analyser = audioContext.createAnalyser(); - analyser.fftSize = 128; + analyser.fftSize = 256; + const silentOutput = audioContext.createGain(); + silentOutput.gain.value = 0; audioContext.createMediaStreamSource(stream).connect(analyser); - const samples = new Uint8Array(analyser.frequencyBinCount); + analyser.connect(silentOutput); + silentOutput.connect(audioContext.destination); + if (audioContext.state === "suspended") { + void audioContext.resume().catch(() => undefined); + } + const samples = new Uint8Array(analyser.fftSize); intervalsRef.current.push( window.setInterval(() => { - analyser.getByteFrequencyData(samples); - setLevels( - Array.from({ length: LEVEL_COUNT }, (_, index) => { - const sampleIndex = Math.floor((index / LEVEL_COUNT) * samples.length); - return Math.max(0.1, (samples[sampleIndex] ?? 0) / 255); - }), - ); - }, 80), + analyser.getByteTimeDomainData(samples); + let squaredAmplitude = 0; + for (const sample of samples) { + const amplitude = (sample - 128) / 128; + squaredAmplitude += amplitude * amplitude; + } + const rootMeanSquare = Math.sqrt(squaredAmplitude / samples.length); + const nextLevel = Math.min(1, Math.max(0, (rootMeanSquare - 0.008) * 9)); + setLevels((current) => [...current.slice(1), nextLevel]); + }, 50), ); startedAtRef.current = Date.now(); intervalsRef.current.push( @@ -145,10 +179,13 @@ export function useVoiceTranscription({ timeoutRef.current = window.setTimeout(() => recorder.stop(), MAX_RECORDING_MS); recorder.start(250); startingRef.current = false; - setStatus("recording"); } catch (cause) { startingRef.current = false; cleanupCapture(); + if (cancelStartingRef.current) { + cancelStartingRef.current = false; + return; + } setStatus("idle"); setError( cause instanceof DOMException && cause.name === "NotAllowedError" diff --git a/apps/web/src/lib/voiceTranscription.test.ts b/apps/web/src/lib/voiceTranscription.test.ts new file mode 100644 index 00000000000..b82a1962f21 --- /dev/null +++ b/apps/web/src/lib/voiceTranscription.test.ts @@ -0,0 +1,70 @@ +import { afterEach, describe, expect, it, vi } from "vite-plus/test"; + +import { + listVoiceTranscriptionModels, + voiceTranscriptionRequestHeaders, +} from "./voiceTranscription"; + +afterEach(() => { + vi.unstubAllGlobals(); +}); + +describe("voiceTranscriptionRequestHeaders", () => { + it("sends a client-provided OpenAI key", () => { + expect( + voiceTranscriptionRequestHeaders("audio/webm", { + provider: "openai", + apiKey: " openai-key ", + model: " gpt-4o-mini-transcribe ", + }), + ).toEqual({ + "content-type": "audio/webm", + "x-t3-transcription-provider": "openai", + "x-t3-transcription-model": "gpt-4o-mini-transcribe", + "x-t3-transcription-api-key": "openai-key", + }); + }); + + it("omits an empty key so the server can use the provider environment variable", () => { + expect( + voiceTranscriptionRequestHeaders("audio/mp4", { + provider: "groq", + apiKey: "", + model: "whisper-large-v3", + }), + ).toEqual({ + "content-type": "audio/mp4", + "x-t3-transcription-provider": "groq", + "x-t3-transcription-model": "whisper-large-v3", + }); + }); +}); + +describe("listVoiceTranscriptionModels", () => { + it("loads the provider models through the connected T3 server", async () => { + const fetchMock = vi.fn().mockResolvedValue( + Response.json({ + models: ["gpt-4o-mini-transcribe", "whisper-1"], + }), + ); + vi.stubGlobal("window", { + location: { + href: "http://localhost:3773/settings", + origin: "http://localhost:3773", + }, + }); + vi.stubGlobal("fetch", fetchMock); + + await expect( + listVoiceTranscriptionModels({ provider: "openai", apiKey: "client-key" }), + ).resolves.toEqual(["gpt-4o-mini-transcribe", "whisper-1"]); + expect(fetchMock).toHaveBeenCalledWith("http://localhost:3773/api/transcription/models", { + method: "GET", + credentials: "include", + headers: { + "x-t3-transcription-api-key": "client-key", + "x-t3-transcription-provider": "openai", + }, + }); + }); +}); diff --git a/apps/web/src/lib/voiceTranscription.ts b/apps/web/src/lib/voiceTranscription.ts index 8fcee72cb17..99a19cb783e 100644 --- a/apps/web/src/lib/voiceTranscription.ts +++ b/apps/web/src/lib/voiceTranscription.ts @@ -6,24 +6,99 @@ import { resolvePrimaryEnvironmentHttpUrl } from "../environments/primary/target export interface VoiceTranscriptionConfig { readonly provider: VoiceTranscriptionProvider; readonly apiKey: string; + readonly model: string; +} + +export type VoiceTranscriptionProviderConfig = Omit; + +export interface VoiceTranscriptionEnvironmentStatus { + readonly openai: boolean; + readonly groq: boolean; +} + +export function voiceTranscriptionRequestHeaders( + contentType: string, + config: VoiceTranscriptionConfig, +): Record { + const apiKey = config.apiKey.trim(); + return { + "content-type": contentType, + "x-t3-transcription-provider": config.provider, + "x-t3-transcription-model": config.model.trim(), + ...(apiKey ? { "x-t3-transcription-api-key": apiKey } : {}), + }; +} + +function voiceTranscriptionProviderHeaders( + config: VoiceTranscriptionProviderConfig, +): Record { + const apiKey = config.apiKey.trim(); + return { + "x-t3-transcription-provider": config.provider, + ...(apiKey ? { "x-t3-transcription-api-key": apiKey } : {}), + }; +} + +export async function readVoiceTranscriptionEnvironmentStatus(): Promise { + const bearerToken = await readDesktopPrimaryBearerToken(); + const response = await globalThis.fetch(resolvePrimaryEnvironmentHttpUrl("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/api/transcription"), { + method: "GET", + credentials: bearerToken ? "omit" : "include", + ...(bearerToken ? { headers: { authorization: `Bearer ${bearerToken}` } } : {}), + }); + if (!response.ok) throw new Error("Could not read transcription provider settings."); + + const payload = (await response.json()) as { readonly openai?: unknown; readonly groq?: unknown }; + return { + openai: payload.openai === true, + groq: payload.groq === true, + }; +} + +export async function listVoiceTranscriptionModels( + config: VoiceTranscriptionProviderConfig, +): Promise { + const bearerToken = await readDesktopPrimaryBearerToken(); + const response = await globalThis.fetch( + resolvePrimaryEnvironmentHttpUrl("/api/transcription/models"), + { + method: "GET", + credentials: bearerToken ? "omit" : "include", + headers: { + ...(bearerToken ? { authorization: `Bearer ${bearerToken}` } : {}), + ...voiceTranscriptionProviderHeaders(config), + }, + }, + ); + const payload = (await response.json().catch(() => null)) as { + readonly models?: unknown; + readonly error?: unknown; + } | null; + if (!response.ok) { + throw new Error( + typeof payload?.error === "string" ? payload.error : "Could not load transcription models.", + ); + } + if ( + !Array.isArray(payload?.models) || + !payload.models.every((model) => typeof model === "string") + ) { + throw new Error("The transcription model response was invalid."); + } + return payload.models; } export async function transcribeVoiceRecording( audio: Blob, config: VoiceTranscriptionConfig, ): Promise { - const apiKey = config.apiKey.trim(); - if (!apiKey) throw new Error("Add an API key for the selected transcription provider."); - const bearerToken = await readDesktopPrimaryBearerToken(); const response = await globalThis.fetch(resolvePrimaryEnvironmentHttpUrl("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/api/transcription"), { method: "POST", credentials: bearerToken ? "omit" : "include", headers: { ...(bearerToken ? { authorization: `Bearer ${bearerToken}` } : {}), - "content-type": audio.type || "audio/webm", - "x-t3-transcription-provider": config.provider, - "x-t3-transcription-api-key": apiKey, + ...voiceTranscriptionRequestHeaders(audio.type || "audio/webm", config), }, body: audio, }); diff --git a/docs/README.md b/docs/README.md index 30653e7d503..869d69ef846 100644 --- a/docs/README.md +++ b/docs/README.md @@ -8,6 +8,7 @@ - [Organizing threads](./user/thread-sidebar.md) - [Review usage](./user/usage.md) - [Customize a project icon](./user/project-settings.md) +- [Voice dictation](./user/voice-dictation.md) - [Remote access](./user/remote-access.md) - [Keeping app and server in sync](./user/updating.md) - [Source control integrations](./user/source-control.md) diff --git a/docs/user/voice-dictation.md b/docs/user/voice-dictation.md new file mode 100644 index 00000000000..e34e51a6bdc --- /dev/null +++ b/docs/user/voice-dictation.md @@ -0,0 +1,28 @@ +# Voice Dictation + +Voice dictation records from the message composer and inserts the transcription into your draft. +It is available in the web and desktop clients on browsers that support microphone recording. + +Enable it in **Settings** → **Beta features** → **Voice dictation**. The composer then shows a +microphone action. Select it to start recording and stop it when you are finished. Recordings stop +automatically after five minutes. + +## Providers and API Keys + +Choose **OpenAI** or **Groq**. T3 Code supplies the provider's transcription endpoint. After an +API key is available, T3 Code loads the models that key can access from the provider and lets you +select the transcription model. Model IDs are not bundled into T3 Code, so newly available models +can appear without an app update. + +You can enter a key in the client, or configure it in the environment that runs the connected T3 +Code server: + +- OpenAI uses `OPENAI_API_KEY` +- Groq uses `GROQ_API_KEY` + +When an environment key is available, the API key input indicates that it is already configured. +A key entered in the input overrides the environment key and stays in that client's local +settings. Environment key values are never sent to the client. + +Recordings are sent through the connected T3 Code server to the selected provider. Recordings +larger than 25 MB are rejected. diff --git a/packages/contracts/src/settings.test.ts b/packages/contracts/src/settings.test.ts index 570157292b5..290163a272f 100644 --- a/packages/contracts/src/settings.test.ts +++ b/packages/contracts/src/settings.test.ts @@ -113,6 +113,22 @@ describe("ClientSettings sidebar", () => { }); }); +describe("ClientSettings voice transcription", () => { + it("defaults to OpenAI and accepts Groq", () => { + const settings = decodeClientSettings({}); + + expect(settings.voiceTranscriptionProvider).toBe("openai"); + expect(settings.voiceTranscriptionModel).toBe(""); + expect( + decodeClientSettingsPatch({ voiceTranscriptionProvider: "groq" }).voiceTranscriptionProvider, + ).toBe("groq"); + expect( + decodeClientSettingsPatch({ voiceTranscriptionModel: " whisper-large-v3 " }) + .voiceTranscriptionModel, + ).toBe("whisper-large-v3"); + }); +}); + describe("ServerSettings.providerInstances (slice-2 invariant)", () => { it("defaults text generation to Luna at low reasoning effort", () => { expect(DEFAULT_SERVER_SETTINGS.textGenerationModelSelection).toEqual({ diff --git a/packages/contracts/src/settings.ts b/packages/contracts/src/settings.ts index 7fb174df0e5..a53283f33cd 100644 --- a/packages/contracts/src/settings.ts +++ b/packages/contracts/src/settings.ts @@ -208,6 +208,7 @@ export const ClientSettingsSchema = Schema.Struct({ Schema.withDecodingDefault(Effect.succeed("openai" as const)), ), voiceTranscriptionApiKey: TrimmedString.pipe(Schema.withDecodingDefault(Effect.succeed(""))), + voiceTranscriptionModel: TrimmedString.pipe(Schema.withDecodingDefault(Effect.succeed(""))), wordWrap: Schema.Boolean.pipe(Schema.withDecodingDefault(Effect.succeed(true))), }); export type ClientSettings = typeof ClientSettingsSchema.Type; @@ -814,6 +815,7 @@ export const ClientSettingsPatch = Schema.Struct({ voiceTranscriptionEnabled: Schema.optionalKey(Schema.Boolean), voiceTranscriptionProvider: Schema.optionalKey(VoiceTranscriptionProvider), voiceTranscriptionApiKey: Schema.optionalKey(TrimmedString), + voiceTranscriptionModel: Schema.optionalKey(TrimmedString), wordWrap: Schema.optionalKey(Schema.Boolean), }); export type ClientSettingsPatch = typeof ClientSettingsPatch.Type; From 0ff9fca68480cf3a03870d896b2815f03cb7e175 Mon Sep 17 00:00:00 2001 From: Yaroslav Kachur <268710642+KachurPro@users.noreply.github.com> Date: Fri, 14 Aug 2026 14:42:16 +0100 Subject: [PATCH 07/21] feat: match Codex voice dictation on web and mobile --- apps/mobile/app.config.ts | 7 + apps/mobile/package.json | 1 + apps/mobile/src/Stack.tsx | 8 + apps/mobile/src/components/AppSymbol.tsx | 2 + .../features/settings/SettingsRouteScreen.tsx | 12 + .../SettingsVoiceDictationRouteScreen.tsx | 90 +++++ .../components/settings-sheet-targets.ts | 1 + .../src/features/threads/ThreadComposer.tsx | 314 +++++++++++------- .../MobileVoiceTranscriptionPanel.tsx | 68 ++++ .../mobileVoiceTranscription.test.ts | 48 +++ .../mobileVoiceTranscription.ts | 49 +++ .../useMobileVoiceTranscription.ts | 222 +++++++++++++ .../voiceTranscriptionSettings.ts | 89 +++++ apps/web/src/components/chat/ChatComposer.tsx | 185 ++++++----- .../chat/VoiceTranscriptionPanel.tsx | 117 +++++-- .../components/settings/SettingsPanels.tsx | 183 ++++++++++ apps/web/src/hooks/useVoiceTranscription.ts | 68 +++- docs/user/voice-dictation.md | 26 +- packages/shared/package.json | 4 + .../shared/src/voiceTranscription.test.ts | 36 ++ packages/shared/src/voiceTranscription.ts | 26 ++ pnpm-lock.yaml | 20 ++ 22 files changed, 1325 insertions(+), 251 deletions(-) create mode 100644 apps/mobile/src/features/settings/SettingsVoiceDictationRouteScreen.tsx create mode 100644 apps/mobile/src/features/voice-dictation/MobileVoiceTranscriptionPanel.tsx create mode 100644 apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.test.ts create mode 100644 apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.ts create mode 100644 apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts create mode 100644 apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.ts create mode 100644 packages/shared/src/voiceTranscription.test.ts create mode 100644 packages/shared/src/voiceTranscription.ts diff --git a/apps/mobile/app.config.ts b/apps/mobile/app.config.ts index 9a51725478e..6c80e8cd001 100644 --- a/apps/mobile/app.config.ts +++ b/apps/mobile/app.config.ts @@ -235,6 +235,13 @@ const config: ExpoConfig = { }, plugins: [ "expo-asset", + [ + "expo-audio", + { + microphonePermission: "Allow T3 Code to access your microphone for voice dictation.", + enableBackgroundRecording: false, + }, + ], [ "expo-font", { diff --git a/apps/mobile/package.json b/apps/mobile/package.json index de53a37c995..6711f740e92 100644 --- a/apps/mobile/package.json +++ b/apps/mobile/package.json @@ -73,6 +73,7 @@ "effect": "catalog:", "expo": "~56.0.12", "expo-asset": "~56.0.17", + "expo-audio": "~56.0.13", "expo-auth-session": "~56.0.14", "expo-blur": "~56.0.3", "expo-build-properties": "~56.0.19", diff --git a/apps/mobile/src/Stack.tsx b/apps/mobile/src/Stack.tsx index 20bba1f6062..de54874d58c 100644 --- a/apps/mobile/src/Stack.tsx +++ b/apps/mobile/src/Stack.tsx @@ -56,6 +56,7 @@ import { SettingsAuthRouteScreen } from "./features/settings/SettingsAuthRouteSc import { SettingsEnvironmentsRouteScreen } from "./features/settings/SettingsEnvironmentsRouteScreen"; import { SettingsLegalRouteScreen } from "./features/settings/SettingsLegalRouteScreen"; import { SettingsProjectGroupingRouteScreen } from "./features/settings/SettingsProjectGroupingRouteScreen"; +import { SettingsVoiceDictationRouteScreen } from "./features/settings/SettingsVoiceDictationRouteScreen"; import { UsageRouteScreen } from "./features/usage/UsageRouteScreen"; import { SettingsRouteScreen } from "./features/settings/SettingsRouteScreen"; import { ShowcaseCaptureCoordinator } from "./features/showcase/ShowcaseCaptureCoordinator"; @@ -198,6 +199,13 @@ const SettingsContentStack = createNativeStackNavigator({ title: "Project Grouping", }, }), + SettingsVoiceDictation: createNativeStackScreen({ + screen: SettingsVoiceDictationRouteScreen, + linking: "voice-dictation", + options: { + title: "Voice Dictation", + }, + }), SettingsClientStorage: createNativeStackScreen({ screen: SettingsClientStorageRouteScreen, linking: "client-storage", diff --git a/apps/mobile/src/components/AppSymbol.tsx b/apps/mobile/src/components/AppSymbol.tsx index 89dc0cc045b..031c65df946 100644 --- a/apps/mobile/src/components/AppSymbol.tsx +++ b/apps/mobile/src/components/AppSymbol.tsx @@ -48,6 +48,7 @@ import { IconLetterSpacing, IconLink, IconMessage, + IconMicrophone, IconMinus, IconNetwork, IconPalette, @@ -123,6 +124,7 @@ const ANDROID_ICON_BY_SF_SYMBOL: Partial> = { "line.3.horizontal.decrease.circle": IconFilter, "line.3.horizontal.decrease.circle.fill": IconFilter, magnifyingglass: IconSearch, + mic: IconMicrophone, paintbrush: IconPalette, "person.crop.circle": IconUserCircle, pin: IconPin, diff --git a/apps/mobile/src/features/settings/SettingsRouteScreen.tsx b/apps/mobile/src/features/settings/SettingsRouteScreen.tsx index bcf2ce386d9..d87de1555a1 100644 --- a/apps/mobile/src/features/settings/SettingsRouteScreen.tsx +++ b/apps/mobile/src/features/settings/SettingsRouteScreen.tsx @@ -46,6 +46,7 @@ import { useSavedRemoteConnections } from "../../state/use-remote-environment-re import { SettingsRow } from "./components/SettingsRow"; import { SettingsSection } from "./components/SettingsSection"; import { SettingsSwitchRow } from "./components/SettingsSwitchRow"; +import { useMobileVoiceTranscriptionSettings } from "../voice-dictation/voiceTranscriptionSettings"; type NotificationStatus = "checking" | "enabled" | "disabled" | "unsupported"; type LiveActivityStatus = "checking" | "enabled" | "disabled" | "signed-out" | "linking"; @@ -527,10 +528,21 @@ function GeneralSettingsSection() { const autoSettleOnMerge = !AsyncResult.isSuccess(preferencesResult) || preferencesResult.value.autoSettleOnMerge !== false; + const voiceTranscription = useMobileVoiceTranscriptionSettings(); return ( + 0 + ? "Enabled" + : "Set up" + } + target="SettingsVoiceDictation" + /> { + if (!settings.loaded || draftInitialized) return; + setDraftApiKey(settings.apiKey); + setDraftInitialized(true); + }, [draftInitialized, settings.apiKey, settings.loaded]); + + const save = async () => { + try { + await saveMobileVoiceTranscriptionApiKey(draftApiKey); + Alert.alert( + draftApiKey.trim() ? "Voice dictation enabled" : "Voice dictation disabled", + draftApiKey.trim() + ? "The microphone is now available in the message composer." + : "The saved API key was removed.", + ); + } catch (cause) { + Alert.alert("Could not save API key", cause instanceof Error ? cause.message : "Try again."); + } + }; + + return ( + + + + + + void save()} + className="h-11 items-center justify-center rounded-full bg-primary disabled:opacity-50" + > + {settings.saving ? ( + + ) : ( + Save + )} + + + + + The key stays in the iPhone secure store. Once it is saved, a microphone appears in the + composer. Audio is sent directly to OpenAI using gpt-4o-mini-transcribe. + + {settings.error ? ( + {settings.error} + ) : null} + + + ); +} diff --git a/apps/mobile/src/features/settings/components/settings-sheet-targets.ts b/apps/mobile/src/features/settings/components/settings-sheet-targets.ts index 7189fdc2ebe..e49bbd5fc39 100644 --- a/apps/mobile/src/features/settings/components/settings-sheet-targets.ts +++ b/apps/mobile/src/features/settings/components/settings-sheet-targets.ts @@ -3,6 +3,7 @@ export type SettingsSheetTarget = | "SettingsArchive" | "SettingsAppearance" | "SettingsProjectGrouping" + | "SettingsVoiceDictation" | "SettingsClientStorage" | "SettingsUsage"; diff --git a/apps/mobile/src/features/threads/ThreadComposer.tsx b/apps/mobile/src/features/threads/ThreadComposer.tsx index 55a4eed568d..d8dc682b36f 100644 --- a/apps/mobile/src/features/threads/ThreadComposer.tsx +++ b/apps/mobile/src/features/threads/ThreadComposer.tsx @@ -13,6 +13,7 @@ import { serializeComposerFileLink, type ComposerTrigger, } from "@t3tools/shared/composerTrigger"; +import { appendVoiceTranscript as appendVoiceTranscriptText } from "@t3tools/shared/voiceTranscription"; import { StackActions, useFocusEffect, useNavigation } from "@react-navigation/native"; import type { ReactNode } from "react"; import { memo, useCallback, useEffect, useMemo, useRef, useState, type RefObject } from "react"; @@ -74,6 +75,9 @@ import { useThreadSettingsSheetPresentation, type NavigationWithFinishTransitioning, } from "./use-thread-settings-sheet-presentation"; +import { MobileVoiceTranscriptionPanel } from "../voice-dictation/MobileVoiceTranscriptionPanel"; +import { useMobileVoiceTranscription } from "../voice-dictation/useMobileVoiceTranscription"; +import { useMobileVoiceTranscriptionSettings } from "../voice-dictation/voiceTranscriptionSettings"; /** * Height of the collapsed composer (pill + vertical padding, excluding safe-area inset). @@ -527,7 +531,7 @@ export const ThreadComposer = memo(function ThreadComposer(props: ThreadComposer // ── Handle command selection ────────────────────────────── const { onChangeDraftMessage, onUpdateInteractionMode, draftMessage, onSendMessage } = props; - const handleSend = useCallback(async () => { + const sendCurrentDraft = useCallback(async () => { const threadKey = scopedThreadKey(props.environmentId, props.selectedThread.id); if (inFlightThreadIdsRef.current.has(threadKey)) return; inFlightThreadIdsRef.current.add(threadKey); @@ -552,6 +556,34 @@ export const ThreadComposer = memo(function ThreadComposer(props: ThreadComposer props.selectedThread.id, props.selectedThread.title, ]); + const voiceTranscriptionSettings = useMobileVoiceTranscriptionSettings(); + const appendVoiceTranscriptToDraft = useCallback( + (transcript: string) => { + const nextDraft = appendVoiceTranscriptText(draftMessage, transcript); + if (nextDraft === draftMessage) return false; + setComposerSelection({ start: nextDraft.length, end: nextDraft.length }); + onChangeDraftMessage(nextDraft); + return true; + }, + [draftMessage, onChangeDraftMessage], + ); + const voiceTranscription = useMobileVoiceTranscription({ + apiKey: voiceTranscriptionSettings.apiKey, + onTranscriptInsert: appendVoiceTranscriptToDraft, + onTranscriptSend: (transcript) => { + if (appendVoiceTranscriptToDraft(transcript)) void sendCurrentDraft(); + }, + }); + const voiceTranscriptionReady = + voiceTranscriptionSettings.loaded && voiceTranscriptionSettings.apiKey.trim().length > 0; + const handleSend = useCallback(async () => { + if (voiceTranscription.status === "recording") { + await voiceTranscription.stop("send"); + return; + } + if (voiceTranscription.status === "transcribing") return; + await sendCurrentDraft(); + }, [sendCurrentDraft, voiceTranscription.status, voiceTranscription.stop]); const handleCommandSelect = useCallback( (item: ComposerCommandItem) => { if (!composerTrigger) return; @@ -756,137 +788,175 @@ export const ThreadComposer = memo(function ThreadComposer(props: ThreadComposer } } > - {/* Attachment strip — inside the card, above the text input */} - {isExpanded ? ( - 0 ? "pb-2.5" : undefined} - entering={FadeIn.duration(160)} - exiting={FadeOut.duration(120)} - > - - - ) : null} - - - void props.onNativePasteImages(uris)} - placeholder={props.placeholder} - onFocus={handleFocus} - onBlur={handleBlur} - onSubmit={handleSend} - scrollEnabled={isExpanded} - // Android: collapsed single line centers natively (gravity) in - // a pill-height box matching the send button; iOS keeps insets. - singleLineCentered={!isExpanded} - contentInsetVertical={isExpanded || Platform.OS === "android" ? 0 : 6} - style={ - isExpanded - ? { - minHeight: 72, - maxHeight: 160, - paddingHorizontal: 4, - paddingVertical: 4, - } - : { - height: 36, - } - } - textStyle={{ - ...bodyText, - color: foregroundColor, - }} + {voiceTranscription.status !== "idle" ? ( + void voiceTranscription.cancel()} + onStop={() => void voiceTranscription.stop("insert")} + onSend={() => void voiceTranscription.stop("send")} /> - - {!isExpanded && props.draftAttachments.length > 0 ? ( - - {props.draftAttachments.slice(0, 3).map((image) => ( - onPressImage(image.previewUri)}> - + {/* Attachment strip — inside the card, above the text input */} + {isExpanded ? ( + 0 ? "pb-2.5" : undefined} + entering={FadeIn.duration(160)} + exiting={FadeOut.duration(120)} + > + - - ))} - {props.draftAttachments.length > 3 ? ( - - - +{props.draftAttachments.length - 3} - - + ) : null} - - ) : null} - {!isExpanded ? ( - - {showStopAction ? ( - - ) : ( - - )} - - ) : null} - {isExpanded ? ( - - - void props.onPickDraftImages()} - showChevron={false} - /> - + + + void props.onNativePasteImages(uris)} + placeholder={props.placeholder} + onFocus={handleFocus} + onBlur={handleBlur} + onSubmit={handleSend} + scrollEnabled={isExpanded} + // Android: collapsed single line centers natively (gravity) in + // a pill-height box matching the send button; iOS keeps insets. + singleLineCentered={!isExpanded} + contentInsetVertical={isExpanded || Platform.OS === "android" ? 0 : 6} + style={ + isExpanded + ? { + minHeight: 72, + maxHeight: 160, + paddingHorizontal: 4, + paddingVertical: 4, + } + : { + height: 36, + } } - label={currentModelOption?.label ?? currentModelSelection.model} - maxWidth={152} - onPress={openSettings} + textStyle={{ + ...bodyText, + color: foregroundColor, + }} /> - {showStopAction ? ( + + {!isExpanded && props.draftAttachments.length > 0 ? ( + + {props.draftAttachments.slice(0, 3).map((image) => ( + onPressImage(image.previewUri)}> + + + ))} + {props.draftAttachments.length > 3 ? ( + + + +{props.draftAttachments.length - 3} + + + ) : null} + + ) : null} + {!isExpanded ? ( + + + {voiceTranscriptionReady ? ( + void voiceTranscription.start()} + /> + ) : null} + {showStopAction ? ( + + ) : ( + + )} + + + ) : null} + {isExpanded ? ( + + + void props.onPickDraftImages()} + showChevron={false} + /> + {voiceTranscriptionReady ? ( + void voiceTranscription.start()} + showChevron={false} + /> + ) : null} + + } + label={currentModelOption?.label ?? currentModelSelection.model} + maxWidth={152} + onPress={openSettings} + /> + {showStopAction ? ( + + ) : null} + - ) : null} - - - - ) : null} + + ) : null} + + )} + {voiceTranscription.status === "idle" && voiceTranscription.error ? ( + + {voiceTranscription.error} + + ) : null} + {/* Queue count */} {props.queueCount > 0 ? ( diff --git a/apps/mobile/src/features/voice-dictation/MobileVoiceTranscriptionPanel.tsx b/apps/mobile/src/features/voice-dictation/MobileVoiceTranscriptionPanel.tsx new file mode 100644 index 00000000000..73f7b49f967 --- /dev/null +++ b/apps/mobile/src/features/voice-dictation/MobileVoiceTranscriptionPanel.tsx @@ -0,0 +1,68 @@ +import { ActivityIndicator, View } from "react-native"; + +import { AppText as Text } from "../../components/AppText"; +import { ControlPill } from "../../components/ControlPill"; +import type { MobileVoiceTranscriptionStatus } from "./useMobileVoiceTranscription"; + +const WAVEFORM_BAR_IDS = Array.from({ length: 24 }, (_, index) => `voice-waveform-${index}`); + +function formatElapsed(elapsedMs: number): string { + const seconds = Math.floor(elapsedMs / 1_000); + return `${Math.floor(seconds / 60)}:${String(seconds % 60).padStart(2, "0")}`; +} + +export function MobileVoiceTranscriptionPanel(props: { + readonly status: Exclude; + readonly elapsedMs: number; + readonly levels: readonly number[]; + readonly sendDisabled: boolean; + readonly onCancel: () => void; + readonly onStop: () => void; + readonly onSend: () => void; +}) { + if (props.status === "transcribing") { + return ( + + + Processing recording… + + ); + } + + return ( + + + + {WAVEFORM_BAR_IDS.map((barId, index) => ( + + ))} + + + {formatElapsed(props.elapsedMs)} + + + + + ); +} diff --git a/apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.test.ts b/apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.test.ts new file mode 100644 index 00000000000..3ddb5647f8c --- /dev/null +++ b/apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.test.ts @@ -0,0 +1,48 @@ +import { afterEach, describe, expect, it, vi } from "vite-plus/test"; + +import { + MOBILE_TRANSCRIPTION_MODEL, + transcribeMobileVoiceRecording, +} from "./mobileVoiceTranscription"; + +afterEach(() => { + vi.unstubAllGlobals(); +}); + +describe("transcribeMobileVoiceRecording", () => { + it("sends an iPhone recording directly to OpenAI", async () => { + const entries: Array<[string, unknown]> = []; + vi.stubGlobal( + "FormData", + class { + append(name: string, value: unknown) { + entries.push([name, value]); + } + }, + ); + const fetchMock = vi.fn().mockResolvedValue(Response.json({ text: " hello " })); + + await expect( + transcribeMobileVoiceRecording("file:///recording.m4a", " secret-key ", fetchMock), + ).resolves.toBe("hello"); + + expect(fetchMock).toHaveBeenCalledWith( + "https://api.openai.com/v1/audio/transcriptions", + expect.objectContaining({ + method: "POST", + headers: { authorization: "Bearer secret-key" }, + }), + ); + expect(entries).toEqual([ + ["model", MOBILE_TRANSCRIPTION_MODEL], + [ + "file", + { + uri: "file:///recording.m4a", + name: "recording.m4a", + type: "audio/mp4", + }, + ], + ]); + }); +}); diff --git a/apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.ts b/apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.ts new file mode 100644 index 00000000000..b2de78bca3b --- /dev/null +++ b/apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.ts @@ -0,0 +1,49 @@ +const OPENAI_TRANSCRIPTION_URL = "https://api.openai.com/v1/audio/transcriptions"; +export const MOBILE_TRANSCRIPTION_MODEL = "gpt-4o-mini-transcribe"; + +export async function transcribeMobileVoiceRecording( + uri: string, + apiKey: string, + fetchFn: typeof globalThis.fetch = globalThis.fetch, +): Promise { + const form = new FormData(); + form.append("model", MOBILE_TRANSCRIPTION_MODEL); + form.append("file", { + uri, + name: "recording.m4a", + type: "audio/mp4", + } as unknown as Blob); + + const controller = new AbortController(); + const timeout = setTimeout(() => controller.abort(), 2 * 60 * 1_000); + try { + const response = await fetchFn(OPENAI_TRANSCRIPTION_URL, { + method: "POST", + headers: { authorization: `Bearer ${apiKey.trim()}` }, + body: form, + signal: controller.signal, + }); + const payload = (await response.json().catch(() => null)) as { + readonly text?: unknown; + readonly error?: { readonly message?: unknown }; + } | null; + if (!response.ok) { + throw new Error( + typeof payload?.error?.message === "string" + ? payload.error.message + : "OpenAI rejected the transcription request.", + ); + } + if (typeof payload?.text !== "string") { + throw new Error("OpenAI returned an invalid transcription response."); + } + return payload.text.trim(); + } catch (cause) { + if (cause instanceof Error && cause.name === "AbortError") { + throw new Error("Voice transcription timed out.", { cause }); + } + throw cause; + } finally { + clearTimeout(timeout); + } +} diff --git a/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts b/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts new file mode 100644 index 00000000000..47a59a67f78 --- /dev/null +++ b/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts @@ -0,0 +1,222 @@ +import { + resolveVoiceTranscriptionAction, + type VoiceTranscriptionAction, +} from "@t3tools/shared/voiceTranscription"; +import { + AudioModule, + RecordingPresets, + setAudioModeAsync, + useAudioRecorder, + useAudioRecorderState, +} from "expo-audio"; +import { useCallback, useEffect, useRef, useState } from "react"; + +import { transcribeMobileVoiceRecording } from "./mobileVoiceTranscription"; + +const LEVEL_COUNT = 36; +const FLAT_LEVELS = Array(LEVEL_COUNT).fill(0); +const MIN_RECORDING_MS = 250; +const MAX_RECORDING_MS = 5 * 60 * 1_000; + +export type MobileVoiceTranscriptionStatus = "idle" | "recording" | "transcribing"; + +export function useMobileVoiceTranscription(input: { + readonly apiKey: string; + readonly onTranscriptInsert: (text: string) => void; + readonly onTranscriptSend: (text: string) => void; +}) { + const recorder = useAudioRecorder({ + ...RecordingPresets.HIGH_QUALITY, + isMeteringEnabled: true, + numberOfChannels: 1, + }); + const recorderState = useAudioRecorderState(recorder, 50); + const [status, setStatus] = useState("idle"); + const [levels, setLevels] = useState(FLAT_LEVELS); + const [error, setError] = useState(null); + const statusRef = useRef(status); + const startingRef = useRef(false); + const cancelStartingRef = useRef(false); + const stopInFlightRef = useRef(false); + const terminalActionRef = useRef(null); + const timeoutRef = useRef | null>(null); + const mountedRef = useRef(true); + const apiKeyRef = useRef(input.apiKey); + const onTranscriptInsertRef = useRef(input.onTranscriptInsert); + const onTranscriptSendRef = useRef(input.onTranscriptSend); + const stopRef = useRef<(action?: "insert" | "send") => Promise>(async () => undefined); + statusRef.current = status; + apiKeyRef.current = input.apiKey; + onTranscriptInsertRef.current = input.onTranscriptInsert; + onTranscriptSendRef.current = input.onTranscriptSend; + + const clearRecordingTimeout = useCallback(() => { + if (timeoutRef.current !== null) clearTimeout(timeoutRef.current); + timeoutRef.current = null; + }, []); + + const resetAudioMode = useCallback(async () => { + await setAudioModeAsync({ allowsRecording: false, playsInSilentMode: true }).catch( + () => undefined, + ); + }, []); + + const stop = useCallback( + async (action: "insert" | "send" = "insert") => { + terminalActionRef.current = resolveVoiceTranscriptionAction( + terminalActionRef.current, + action, + ); + if (startingRef.current) { + cancelStartingRef.current = true; + clearRecordingTimeout(); + if (mountedRef.current) setStatus("idle"); + return; + } + if (statusRef.current !== "recording" || stopInFlightRef.current) return; + stopInFlightRef.current = true; + clearRecordingTimeout(); + try { + const beforeStop = await recorder.getStatus(); + await recorder.stop(); + const terminalAction = terminalActionRef.current ?? "insert"; + const uri = recorder.uri; + terminalActionRef.current = null; + await resetAudioMode(); + if (terminalAction === "abort" || beforeStop.durationMillis < MIN_RECORDING_MS || !uri) { + if (mountedRef.current) { + setStatus("idle"); + setLevels(FLAT_LEVELS); + } + return; + } + + if (mountedRef.current) setStatus("transcribing"); + const text = await transcribeMobileVoiceRecording(uri, apiKeyRef.current); + if (!mountedRef.current) return; + if (text) { + if (terminalAction === "send") onTranscriptSendRef.current(text); + else onTranscriptInsertRef.current(text); + } + setStatus("idle"); + setLevels(FLAT_LEVELS); + } catch (cause) { + await resetAudioMode(); + if (!mountedRef.current) return; + setError(cause instanceof Error ? cause.message : "Voice transcription failed."); + setStatus("idle"); + setLevels(FLAT_LEVELS); + } finally { + stopInFlightRef.current = false; + terminalActionRef.current = null; + } + }, + [clearRecordingTimeout, recorder, resetAudioMode], + ); + stopRef.current = stop; + + const cancel = useCallback(async () => { + terminalActionRef.current = "abort"; + if (startingRef.current) { + cancelStartingRef.current = true; + clearRecordingTimeout(); + if (mountedRef.current) setStatus("idle"); + return; + } + if (statusRef.current !== "recording" || stopInFlightRef.current) return; + stopInFlightRef.current = true; + clearRecordingTimeout(); + try { + await recorder.stop(); + } catch { + // Cancellation is best-effort; the audio is discarded either way. + } finally { + await resetAudioMode(); + stopInFlightRef.current = false; + terminalActionRef.current = null; + if (mountedRef.current) { + setStatus("idle"); + setLevels(FLAT_LEVELS); + } + } + }, [clearRecordingTimeout, recorder, resetAudioMode]); + + const start = useCallback(async () => { + if (startingRef.current || statusRef.current !== "idle") return; + if (!apiKeyRef.current.trim()) { + setError("Save an OpenAI API key in Settings first."); + return; + } + + startingRef.current = true; + cancelStartingRef.current = false; + terminalActionRef.current = null; + setError(null); + setLevels(FLAT_LEVELS); + setStatus("recording"); + try { + const permission = await AudioModule.requestRecordingPermissionsAsync(); + if (!permission.granted) { + throw new Error("Microphone permission was denied."); + } + if (cancelStartingRef.current || !mountedRef.current) { + startingRef.current = false; + cancelStartingRef.current = false; + if (mountedRef.current) setStatus("idle"); + return; + } + await setAudioModeAsync({ allowsRecording: true, playsInSilentMode: true }); + await recorder.prepareToRecordAsync(); + if (cancelStartingRef.current || !mountedRef.current) { + startingRef.current = false; + cancelStartingRef.current = false; + await resetAudioMode(); + if (mountedRef.current) setStatus("idle"); + return; + } + recorder.record(); + startingRef.current = false; + timeoutRef.current = setTimeout(() => { + void stopRef.current("insert"); + }, MAX_RECORDING_MS); + } catch (cause) { + startingRef.current = false; + await resetAudioMode(); + if (cancelStartingRef.current) { + cancelStartingRef.current = false; + return; + } + if (!mountedRef.current) return; + setError(cause instanceof Error ? cause.message : "Could not start the microphone."); + setStatus("idle"); + } + }, [recorder, resetAudioMode]); + + useEffect(() => { + if (status !== "recording") return; + const metering = recorderState.metering ?? -60; + const level = Math.min(1, Math.max(0, (metering + 60) / 60)); + setLevels((current) => [...current.slice(1), level]); + }, [recorderState.metering, status]); + + useEffect(() => { + mountedRef.current = true; + return () => { + mountedRef.current = false; + terminalActionRef.current = "abort"; + clearRecordingTimeout(); + if (recorder.isRecording) void recorder.stop().catch(() => undefined); + void resetAudioMode(); + }; + }, [clearRecordingTimeout, recorder, resetAudioMode]); + + return { + status, + levels, + elapsedMs: recorderState.durationMillis, + error, + start, + stop, + cancel, + } as const; +} diff --git a/apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.ts b/apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.ts new file mode 100644 index 00000000000..8ebc261c3b4 --- /dev/null +++ b/apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.ts @@ -0,0 +1,89 @@ +import * as SecureStore from "expo-secure-store"; +import { useEffect, useSyncExternalStore } from "react"; + +const OPENAI_API_KEY_STORAGE_KEY = "t3code.voice-transcription.openai-api-key"; + +export interface MobileVoiceTranscriptionSettingsSnapshot { + readonly apiKey: string; + readonly error: string | null; + readonly loaded: boolean; + readonly saving: boolean; +} + +let snapshot: MobileVoiceTranscriptionSettingsSnapshot = { + apiKey: "", + error: null, + loaded: false, + saving: false, +}; +let loadPromise: Promise | null = null; +let revision = 0; +const listeners = new Set<() => void>(); + +function publish(next: MobileVoiceTranscriptionSettingsSnapshot) { + snapshot = next; + for (const listener of listeners) listener(); +} + +function subscribe(listener: () => void) { + listeners.add(listener); + return () => listeners.delete(listener); +} + +export function loadMobileVoiceTranscriptionSettings(): Promise { + if (snapshot.loaded) return Promise.resolve(); + if (loadPromise) return loadPromise; + const loadRevision = revision; + loadPromise = SecureStore.getItemAsync(OPENAI_API_KEY_STORAGE_KEY) + .then((apiKey) => { + if (revision !== loadRevision) return; + publish({ apiKey: apiKey ?? "", error: null, loaded: true, saving: false }); + }) + .catch(() => { + if (revision !== loadRevision) return; + publish({ + apiKey: "", + error: "Could not read the saved API key.", + loaded: true, + saving: false, + }); + }) + .finally(() => { + loadPromise = null; + }); + return loadPromise; +} + +export async function saveMobileVoiceTranscriptionApiKey(apiKey: string): Promise { + const normalized = apiKey.trim(); + revision += 1; + publish({ ...snapshot, error: null, saving: true }); + try { + if (normalized) { + await SecureStore.setItemAsync(OPENAI_API_KEY_STORAGE_KEY, normalized); + } else { + await SecureStore.deleteItemAsync(OPENAI_API_KEY_STORAGE_KEY); + } + publish({ apiKey: normalized, error: null, loaded: true, saving: false }); + } catch { + publish({ + ...snapshot, + error: "Could not save the API key.", + loaded: true, + saving: false, + }); + throw new Error("Could not save the API key."); + } +} + +export function useMobileVoiceTranscriptionSettings() { + const current = useSyncExternalStore( + subscribe, + () => snapshot, + () => snapshot, + ); + useEffect(() => { + void loadMobileVoiceTranscriptionSettings(); + }, []); + return current; +} diff --git a/apps/web/src/components/chat/ChatComposer.tsx b/apps/web/src/components/chat/ChatComposer.tsx index 8631283c643..738f49c71d0 100644 --- a/apps/web/src/components/chat/ChatComposer.tsx +++ b/apps/web/src/components/chat/ChatComposer.tsx @@ -20,6 +20,7 @@ import { import type { EnvironmentConnectionPresentation } from "@t3tools/client-runtime/connection"; import { serializeComposerFileLink } from "@t3tools/shared/composerTrigger"; import { createModelSelection, normalizeModelSlug } from "@t3tools/shared/model"; +import { appendVoiceTranscript as appendVoiceTranscriptText } from "@t3tools/shared/voiceTranscription"; import { memo, type ReactNode, @@ -203,7 +204,6 @@ import { PenLineIcon, SparklesIcon, MicIcon, - SquareIcon, XIcon, } from "lucide-react"; import { proposedPlanTitle } from "../../proposedPlan"; @@ -978,6 +978,7 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) const dragDepthRef = useRef(0); const stashPulseKeyRef = useRef(0); const stashPulseTimeoutRef = useRef(null); + const submitComposerRef = useRef<(event?: { preventDefault: () => void }) => void>(() => {}); /** * Snapshots currently being encoded, keyed by target+prompt+image ids. * Keyed rather than boolean so a genuinely different prompt (or a different @@ -1244,6 +1245,14 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) projectSelectionRequired || environmentUnavailable !== null || !composerSendState.hasSendableContent; + const voiceTranscriptionSendDisabled = + phase === "running" || + isSendBusy || + isSendDisabled || + isConnecting || + noProviderAvailable || + projectSelectionRequired || + environmentUnavailable !== null; const collapsedComposerPrimaryActionLabel = "Send message"; const showMobilePendingAnswerActions = isMobileViewport && !isComposerCollapsedMobile && pendingPrimaryAction !== null; @@ -1258,17 +1267,18 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) [composerDraftTarget, setComposerDraftPrompt], ); - const appendVoiceTranscript = useCallback( + const appendVoiceTranscriptToPrompt = useCallback( (transcript: string) => { const currentPrompt = promptRef.current; - const boundary = currentPrompt.length > 0 && !/\s$/.test(currentPrompt) ? " " : ""; - const nextPrompt = `${currentPrompt}${boundary}${transcript}`; + const nextPrompt = appendVoiceTranscriptText(currentPrompt, transcript); + if (nextPrompt === currentPrompt) return false; promptRef.current = nextPrompt; setPrompt(nextPrompt); const nextCursor = collapseExpandedComposerCursor(nextPrompt, nextPrompt.length); setComposerCursor(nextCursor); setComposerTrigger(null); scheduleComposerFocus(); + return true; }, [promptRef, scheduleComposerFocus, setPrompt], ); @@ -1278,8 +1288,24 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) apiKey: settings.voiceTranscriptionApiKey, model: settings.voiceTranscriptionModel, }, - onTranscript: appendVoiceTranscript, + onTranscriptInsert: appendVoiceTranscriptToPrompt, + onTranscriptSend: (transcript) => { + if (appendVoiceTranscriptToPrompt(transcript)) submitComposerRef.current(); + }, }); + const voiceTranscriptionReady = + settings.voiceTranscriptionEnabled && settings.voiceTranscriptionModel.trim().length > 0; + + useEffect(() => { + if (voiceTranscription.status !== "recording") return; + const cancelOnEscape = (event: KeyboardEvent) => { + if (event.key !== "Escape") return; + event.preventDefault(); + voiceTranscription.cancel(); + }; + window.addEventListener("keydown", cancelOnEscape); + return () => window.removeEventListener("keydown", cancelOnEscape); + }, [voiceTranscription.cancel, voiceTranscription.status]); const addComposerImage = useCallback( (image: ComposerImageAttachment) => { @@ -1832,7 +1858,7 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) showPlanFollowUpPrompt, ]); - const submitComposer = useCallback( + const submitComposerAfterTranscription = useCallback( (event?: { preventDefault: () => void }) => { if (noProviderAvailable || isSendDisabled) { event?.preventDefault(); @@ -1865,6 +1891,22 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) shouldBlurMobileComposerOnSubmit, ], ); + const submitComposer = useCallback( + (event?: { preventDefault: () => void }) => { + if (voiceTranscription.status === "recording") { + event?.preventDefault(); + voiceTranscription.stop("send"); + return; + } + if (voiceTranscription.status === "transcribing") { + event?.preventDefault(); + return; + } + submitComposerAfterTranscription(event); + }, + [submitComposerAfterTranscription, voiceTranscription.status, voiceTranscription.stop], + ); + submitComposerRef.current = submitComposerAfterTranscription; const expandMobileComposer = useCallback(() => { if (composerBlurFrameRef.current !== null) { window.cancelAnimationFrame(composerBlurFrameRef.current); @@ -3114,7 +3156,7 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps)
{voiceTranscription.status === "idle" && - settings.voiceTranscriptionEnabled && + voiceTranscriptionReady && voiceTranscription.error ? (

{voiceTranscription.error} @@ -3146,6 +3188,10 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) status={voiceTranscription.status} elapsedMs={voiceTranscription.elapsedMs} levels={voiceTranscription.levels} + sendDisabled={voiceTranscriptionSendDisabled} + onCancel={voiceTranscription.cancel} + onStop={() => voiceTranscription.stop("insert")} + onSend={() => voiceTranscription.stop("send")} /> ) : null}

{/* Right side: send / stop button */} -
- {settings.voiceTranscriptionEnabled && - !isComposerApprovalState && - pendingUserInputs.length === 0 && - voiceTranscription.status !== "transcribing" ? ( - - - voiceTranscription.status === "recording" - ? voiceTranscription.stop() - : void voiceTranscription.start() - } - aria-label={ - voiceTranscription.status === "recording" - ? "Stop dictation" - : "Start dictation" - } - > - {voiceTranscription.status === "recording" ? ( - - ) : ( - - )} - - } - /> - - {voiceTranscription.status === "recording" - ? "Stop dictation" - : "Start dictation"} - - - ) : null} - 0} - isSendBusy={isSendBusy} - sendDisabledReason={sendDisabledReason} - isConnecting={isConnecting} - isEnvironmentUnavailable={ - environmentUnavailable !== null || - noProviderAvailable || - projectSelectionRequired + {voiceTranscription.status === "idle" ? ( +
-
+ className="flex shrink-0 flex-nowrap items-center justify-end gap-2" + > + {voiceTranscriptionReady && + !isComposerApprovalState && + pendingUserInputs.length === 0 ? ( + + void voiceTranscription.start()} + aria-label="Start dictation" + > + + + } + /> + Start dictation + + ) : null} + 0} + isSendBusy={isSendBusy} + sendDisabledReason={sendDisabledReason} + isConnecting={isConnecting} + isEnvironmentUnavailable={ + environmentUnavailable !== null || + noProviderAvailable || + projectSelectionRequired + } + isPreparingWorktree={isPreparingWorktree} + hasSendableContent={composerSendState.hasSendableContent} + preserveComposerFocusOnPointerDown={isMobileViewport} + onPreviousPendingQuestion={onPreviousActivePendingUserInputQuestion} + onInterrupt={handleInterruptPrimaryAction} + onImplementPlanInNewThread={handleImplementPlanInNewThreadPrimaryAction} + /> +
+ ) : null}
)}
diff --git a/apps/web/src/components/chat/VoiceTranscriptionPanel.tsx b/apps/web/src/components/chat/VoiceTranscriptionPanel.tsx index 43d39185a2c..e90bedceb3c 100644 --- a/apps/web/src/components/chat/VoiceTranscriptionPanel.tsx +++ b/apps/web/src/components/chat/VoiceTranscriptionPanel.tsx @@ -1,6 +1,7 @@ -import { LoaderCircleIcon } from "lucide-react"; +import { ArrowUpIcon, LoaderCircleIcon, SquareIcon, XIcon } from "lucide-react"; import type { VoiceTranscriptionStatus } from "../../hooks/useVoiceTranscription"; +import { Button } from "../ui/button"; function formatElapsed(elapsedMs: number): string { const seconds = Math.floor(elapsedMs / 1_000); @@ -11,10 +12,18 @@ export function VoiceTranscriptionPanel({ status, elapsedMs, levels, + sendDisabled, + onCancel, + onStop, + onSend, }: { readonly status: Exclude; readonly elapsedMs: number; readonly levels: readonly number[]; + readonly sendDisabled: boolean; + readonly onCancel: () => void; + readonly onStop: () => void; + readonly onSend: () => void; }) { const waveformPath = levels .map((level, index) => { @@ -24,48 +33,84 @@ export function VoiceTranscriptionPanel({ }) .join(" "); + if (status === "transcribing") { + return ( +
+ + Processing recording… +
+ ); + } + return (
- {status === "recording" ? ( - - ) : ( -
- - Transcribing… -
- )} + + {formatElapsed(elapsedMs)} + +
); } diff --git a/apps/web/src/components/settings/SettingsPanels.tsx b/apps/web/src/components/settings/SettingsPanels.tsx index 9df7f88ab1d..0da85a732b3 100644 --- a/apps/web/src/components/settings/SettingsPanels.tsx +++ b/apps/web/src/components/settings/SettingsPanels.tsx @@ -9,6 +9,7 @@ import { ProviderDriverKind, type ScopedThreadRef, type SidebarProjectGroupingMode, + type VoiceTranscriptionProvider, } from "@t3tools/contracts"; import { scopeThreadRef } from "@t3tools/client-runtime/environment"; import { @@ -142,6 +143,10 @@ import { } from "./settingsLayout"; import { searchableSetting } from "./settingsSearch"; import { ProjectFavicon } from "../ProjectFavicon"; +import { + listVoiceTranscriptionModels, + readVoiceTranscriptionEnvironmentStatus, +} from "../../lib/voiceTranscription"; const ENVIRONMENT_IDENTIFICATION_LABELS: Record = { artwork: "Artwork", @@ -1595,6 +1600,10 @@ function FontFamilySettingsRow({ } const AUTO_SETTLE_DEFAULT_DAYS = DEFAULT_UNIFIED_SETTINGS.sidebarAutoSettleAfterDays ?? 3; +const TRANSCRIPTION_API_KEY_ENV = { + openai: "OPENAI_API_KEY", + groq: "GROQ_API_KEY", +} as const; function AutoSettleDaysInput({ value, @@ -1637,6 +1646,178 @@ function AutoSettleDaysInput({ ); } +function VoiceDictationSettingsSection() { + const settings = usePrimarySettings(); + const updateSettings = useUpdatePrimarySettings(); + const [environmentApiKeys, setEnvironmentApiKeys] = useState({ openai: false, groq: false }); + const [models, setModels] = useState([]); + const [modelsLoading, setModelsLoading] = useState(false); + const [modelsError, setModelsError] = useState(null); + const provider = settings.voiceTranscriptionProvider; + const apiKey = settings.voiceTranscriptionApiKey; + const model = settings.voiceTranscriptionModel; + + useEffect(() => { + let active = true; + void readVoiceTranscriptionEnvironmentStatus() + .then((status) => { + if (active) setEnvironmentApiKeys(status); + }) + .catch(() => undefined); + return () => { + active = false; + }; + }, []); + + const providerLabel = provider === "openai" ? "OpenAI" : "Groq"; + const environmentVariable = TRANSCRIPTION_API_KEY_ENV[provider]; + const hasEnvironmentApiKey = environmentApiKeys[provider]; + const hasApiKey = apiKey.trim().length > 0 || hasEnvironmentApiKey; + + useEffect(() => { + if (!hasApiKey) { + setModels([]); + setModelsLoading(false); + setModelsError(null); + return; + } + + let active = true; + setModelsLoading(true); + setModelsError(null); + const timeout = window.setTimeout( + () => { + void listVoiceTranscriptionModels({ provider, apiKey }) + .then((nextModels) => { + if (!active) return; + setModels(nextModels); + setModelsLoading(false); + }) + .catch((cause: unknown) => { + if (!active) return; + setModels([]); + setModelsLoading(false); + setModelsError( + cause instanceof Error ? cause.message : "Could not load transcription models.", + ); + }); + }, + apiKey.trim() ? 400 : 0, + ); + + return () => { + active = false; + window.clearTimeout(timeout); + }; + }, [apiKey, hasApiKey, provider]); + + useEffect(() => { + if (models.length > 0 && model && !models.includes(model)) { + updateSettings({ voiceTranscriptionEnabled: false, voiceTranscriptionModel: "" }); + } + }, [model, models, updateSettings]); + + const modelDescription = !hasApiKey + ? `Save an API key to load ${providerLabel} transcription models.` + : modelsLoading + ? `Loading models available from ${providerLabel}…` + : modelsError + ? modelsError + : models.length === 0 + ? `${providerLabel} did not return any models.` + : "Choose a model. The microphone appears in the composer after that."; + + return ( + + + updateSettings({ + voiceTranscriptionProvider: value as VoiceTranscriptionProvider, + voiceTranscriptionApiKey: "", + voiceTranscriptionModel: "", + voiceTranscriptionEnabled: false, + }) + } + > + + {providerLabel} + + + + OpenAI + + + Groq + + + + } + /> + + updateSettings({ + voiceTranscriptionApiKey: event.target.value, + voiceTranscriptionModel: "", + voiceTranscriptionEnabled: false, + }) + } + placeholder={hasEnvironmentApiKey ? `Using ${environmentVariable}` : "Required"} + aria-label={`${providerLabel} transcription API key`} + /> + } + /> + { + if (value !== null) { + updateSettings({ + voiceTranscriptionModel: value, + voiceTranscriptionEnabled: true, + }); + } + }} + > + + + {model || (modelsLoading ? "Loading models…" : "Select model")} + + + + {models.map((availableModel) => ( + + {availableModel} + + ))} + + + } + /> + + ); +} + // The legacy rows sit behind the fold, so a settings-search jump has to // expand the section before its target can mount and scroll. const LEGACY_FEATURE_TARGET_IDS: ReadonlySet = new Set([ @@ -2309,6 +2490,8 @@ export function GeneralSettingsPanel() { /> + + {isElectron || HOSTED_APP_CHANNEL ? ( diff --git a/apps/web/src/hooks/useVoiceTranscription.ts b/apps/web/src/hooks/useVoiceTranscription.ts index 22e776dafc7..628865fa484 100644 --- a/apps/web/src/hooks/useVoiceTranscription.ts +++ b/apps/web/src/hooks/useVoiceTranscription.ts @@ -1,8 +1,13 @@ +import { + resolveVoiceTranscriptionAction, + type VoiceTranscriptionAction, +} from "@t3tools/shared/voiceTranscription"; import { useCallback, useEffect, useRef, useState } from "react"; import { transcribeVoiceRecording, type VoiceTranscriptionConfig } from "../lib/voiceTranscription"; const LEVEL_COUNT = 160; +const MIN_RECORDING_MS = 250; const MAX_RECORDING_MS = 5 * 60 * 1_000; const MIME_TYPES = ["audio/webm;codecs=opus", "audio/ogg;codecs=opus", "audio/mp4"]; const FLAT_LEVELS = Array(LEVEL_COUNT).fill(0); @@ -16,10 +21,12 @@ function supportedMimeType(): string | undefined { export function useVoiceTranscription({ config, - onTranscript, + onTranscriptInsert, + onTranscriptSend, }: { readonly config: VoiceTranscriptionConfig; - readonly onTranscript: (text: string) => void; + readonly onTranscriptInsert: (text: string) => void; + readonly onTranscriptSend: (text: string) => void; }) { const [status, setStatus] = useState("idle"); const [elapsedMs, setElapsedMs] = useState(0); @@ -33,11 +40,14 @@ export function useVoiceTranscription({ const intervalsRef = useRef([]); const timeoutRef = useRef(null); const startedAtRef = useRef(0); + const terminalActionRef = useRef(null); const mountedRef = useRef(true); const configRef = useRef(config); - const onTranscriptRef = useRef(onTranscript); + const onTranscriptInsertRef = useRef(onTranscriptInsert); + const onTranscriptSendRef = useRef(onTranscriptSend); configRef.current = config; - onTranscriptRef.current = onTranscript; + onTranscriptInsertRef.current = onTranscriptInsert; + onTranscriptSendRef.current = onTranscriptSend; const cleanupCapture = useCallback(() => { for (const interval of intervalsRef.current) window.clearInterval(interval); @@ -57,13 +67,36 @@ export function useVoiceTranscription({ mountedRef.current = false; startingRef.current = false; cancelStartingRef.current = true; + terminalActionRef.current = "abort"; const recorder = recorderRef.current; if (recorder?.state === "recording") recorder.stop(); cleanupCapture(); }; }, [cleanupCapture]); - const stop = useCallback(() => { + const stop = useCallback( + (action: Exclude = "insert") => { + terminalActionRef.current = resolveVoiceTranscriptionAction( + terminalActionRef.current, + action, + ); + if (startingRef.current) { + cancelStartingRef.current = true; + cleanupCapture(); + if (mountedRef.current) { + setStatus("idle"); + setElapsedMs(0); + } + return; + } + const recorder = recorderRef.current; + if (recorder?.state === "recording") recorder.stop(); + }, + [cleanupCapture], + ); + + const cancel = useCallback(() => { + terminalActionRef.current = "abort"; if (startingRef.current) { cancelStartingRef.current = true; cleanupCapture(); @@ -80,6 +113,7 @@ export function useVoiceTranscription({ const start = useCallback(async () => { if (startingRef.current || status !== "idle") return; startingRef.current = true; + terminalActionRef.current = null; setError(null); setElapsedMs(0); setLevels(FLAT_LEVELS); @@ -109,6 +143,7 @@ export function useVoiceTranscription({ stream.getTracks().forEach((track) => track.stop()); startingRef.current = false; cancelStartingRef.current = false; + terminalActionRef.current = null; return; } @@ -124,20 +159,33 @@ export function useVoiceTranscription({ recorder.addEventListener("error", () => { recordingFailed = true; cleanupCapture(); + terminalActionRef.current = null; if (mountedRef.current) { setStatus("idle"); setError("The microphone stopped unexpectedly."); } }); recorder.addEventListener("stop", () => { + const durationMs = Date.now() - startedAtRef.current; + const action = terminalActionRef.current ?? "insert"; + terminalActionRef.current = null; const blob = new Blob(chunks, { type: recorder.mimeType || mimeType || "audio/webm" }); cleanupCapture(); if (!mountedRef.current || recordingFailed) return; + if (action === "abort" || durationMs < MIN_RECORDING_MS || blob.size === 0) { + setStatus("idle"); + setElapsedMs(0); + return; + } + setStatus("transcribing"); void transcribeVoiceRecording(blob, configRef.current) .then((text) => { if (!mountedRef.current) return; - if (text) onTranscriptRef.current(text); + if (text) { + if (action === "send") onTranscriptSendRef.current(text); + else onTranscriptInsertRef.current(text); + } setStatus("idle"); setElapsedMs(0); }) @@ -176,12 +224,16 @@ export function useVoiceTranscription({ intervalsRef.current.push( window.setInterval(() => setElapsedMs(Date.now() - startedAtRef.current), 250), ); - timeoutRef.current = window.setTimeout(() => recorder.stop(), MAX_RECORDING_MS); + timeoutRef.current = window.setTimeout(() => { + terminalActionRef.current = terminalActionRef.current ?? "insert"; + if (recorder.state === "recording") recorder.stop(); + }, MAX_RECORDING_MS); recorder.start(250); startingRef.current = false; } catch (cause) { startingRef.current = false; cleanupCapture(); + terminalActionRef.current = null; if (cancelStartingRef.current) { cancelStartingRef.current = false; return; @@ -195,5 +247,5 @@ export function useVoiceTranscription({ } }, [cleanupCapture, status]); - return { status, elapsedMs, levels, error, start, stop } as const; + return { status, elapsedMs, levels, error, start, stop, cancel } as const; } diff --git a/docs/user/voice-dictation.md b/docs/user/voice-dictation.md index e34e51a6bdc..6cf69ffb7d3 100644 --- a/docs/user/voice-dictation.md +++ b/docs/user/voice-dictation.md @@ -1,18 +1,20 @@ # Voice Dictation -Voice dictation records from the message composer and inserts the transcription into your draft. -It is available in the web and desktop clients on browsers that support microphone recording. +Voice dictation follows the Codex composer flow on web, desktop, and the native mobile app: -Enable it in **Settings** → **Beta features** → **Voice dictation**. The composer then shows a -microphone action. Select it to start recording and stop it when you are finished. Recordings stop -automatically after five minutes. +- **X** cancels and discards the recording. +- **Stop** transcribes and appends the text to the end of the current draft. +- **Send** transcribes, appends the text, and uses the normal message-send path. + +Empty and very short recordings are discarded, and recordings stop automatically after five +minutes. A transcript never replaces text that is already in the composer. ## Providers and API Keys -Choose **OpenAI** or **Groq**. T3 Code supplies the provider's transcription endpoint. After an -API key is available, T3 Code loads the models that key can access from the provider and lets you -select the transcription model. Model IDs are not bundled into T3 Code, so newly available models -can appear without an app update. +On web and desktop, open **Settings** → **General** → **Voice dictation**, choose **OpenAI** or +**Groq**, save an API key, and select a transcription model. The microphone appears after the +configuration is complete. T3 Code loads the models available to that key, so new models can +appear without an app update. You can enter a key in the client, or configure it in the environment that runs the connected T3 Code server: @@ -26,3 +28,9 @@ settings. Environment key values are never sent to the client. Recordings are sent through the connected T3 Code server to the selected provider. Recordings larger than 25 MB are rejected. + +## iPhone and Android + +Open **Settings** → **Voice Dictation** and save an OpenAI API key. The key is kept in the device's +secure store. Native recordings use `gpt-4o-mini-transcribe` and are sent directly from the mobile +app to OpenAI. Clearing the saved key removes the microphone from the composer. diff --git a/packages/shared/package.json b/packages/shared/package.json index f669bd0a452..e895bdb89bd 100644 --- a/packages/shared/package.json +++ b/packages/shared/package.json @@ -167,6 +167,10 @@ "types": "./src/composerInlineTokens.ts", "import": "./src/composerInlineTokens.ts" }, + "./voiceTranscription": { + "types": "./src/voiceTranscription.ts", + "import": "./src/voiceTranscription.ts" + }, "./terminalLabels": { "types": "./src/terminalLabels.ts", "import": "./src/terminalLabels.ts" diff --git a/packages/shared/src/voiceTranscription.test.ts b/packages/shared/src/voiceTranscription.test.ts new file mode 100644 index 00000000000..5ebe1daf0f1 --- /dev/null +++ b/packages/shared/src/voiceTranscription.test.ts @@ -0,0 +1,36 @@ +import { describe, expect, it } from "vite-plus/test"; + +import { appendVoiceTranscript, resolveVoiceTranscriptionAction } from "./voiceTranscription.js"; + +describe("appendVoiceTranscript", () => { + it("appends trimmed speech to an empty draft", () => { + expect(appendVoiceTranscript("", " hello ")).toBe("hello"); + }); + + it("adds one boundary space after existing text", () => { + expect(appendVoiceTranscript("existing", "speech")).toBe("existing speech"); + }); + + it("preserves an existing whitespace boundary", () => { + expect(appendVoiceTranscript("existing\n", " speech ")).toBe("existing\nspeech"); + }); + + it("ignores an empty transcript", () => { + expect(appendVoiceTranscript("existing", " ")).toBe("existing"); + }); +}); + +describe("resolveVoiceTranscriptionAction", () => { + it("upgrades insert to send", () => { + expect(resolveVoiceTranscriptionAction("insert", "send")).toBe("send"); + }); + + it("does not downgrade send to insert", () => { + expect(resolveVoiceTranscriptionAction("send", "insert")).toBe("send"); + }); + + it("lets cancellation win", () => { + expect(resolveVoiceTranscriptionAction("send", "abort")).toBe("abort"); + expect(resolveVoiceTranscriptionAction("abort", "send")).toBe("abort"); + }); +}); diff --git a/packages/shared/src/voiceTranscription.ts b/packages/shared/src/voiceTranscription.ts new file mode 100644 index 00000000000..64990c3b0ab --- /dev/null +++ b/packages/shared/src/voiceTranscription.ts @@ -0,0 +1,26 @@ +/** + * Mirrors the Codex composer boundary rule: trim the transcript, append it to + * the end, and add exactly one space only when the existing draft needs one. + */ +export function appendVoiceTranscript(existing: string, transcript: string): string { + const normalized = transcript.trim(); + if (normalized.length === 0) return existing; + if (existing.length === 0 || /\s$/.test(existing)) return `${existing}${normalized}`; + return `${existing} ${normalized}`; +} + +export type VoiceTranscriptionAction = "insert" | "send" | "abort"; + +/** + * Keeps a single terminal outcome for a recording. Cancellation always wins; + * pressing Send can upgrade an already-requested insert while transcription is + * being finalized. + */ +export function resolveVoiceTranscriptionAction( + current: VoiceTranscriptionAction | null, + next: VoiceTranscriptionAction, +): VoiceTranscriptionAction { + if (current === "abort" || next === "abort") return "abort"; + if (current === "send" || next === "send") return "send"; + return "insert"; +} diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 2c79aea36a0..9fbe93994c8 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -286,6 +286,9 @@ importers: expo-asset: specifier: ~56.0.17 version: 56.0.17(expo@56.0.12)(react-native@0.85.3(@babel/core@7.29.7)(@react-native/metro-config@0.85.3(@babel/core@7.29.7)(bufferutil@4.1.0)(utf-8-validate@6.0.6))(@types/react@19.2.16)(bufferutil@4.1.0)(react@19.2.3)(utf-8-validate@6.0.6))(react@19.2.3)(typescript@6.0.3) + expo-audio: + specifier: ~56.0.13 + version: 56.0.13(expo-asset@56.0.17(expo@56.0.12)(react-native@0.85.3(@babel/core@7.29.7)(@react-native/metro-config@0.85.3(@babel/core@7.29.7)(bufferutil@4.1.0)(utf-8-validate@6.0.6))(@types/react@19.2.16)(bufferutil@4.1.0)(react@19.2.3)(utf-8-validate@6.0.6))(react@19.2.3)(typescript@6.0.3))(expo@56.0.12)(react-native@0.85.3(@babel/core@7.29.7)(@react-native/metro-config@0.85.3(@babel/core@7.29.7)(bufferutil@4.1.0)(utf-8-validate@6.0.6))(@types/react@19.2.16)(bufferutil@4.1.0)(react@19.2.3)(utf-8-validate@6.0.6))(react@19.2.3) expo-auth-session: specifier: ~56.0.14 version: 56.0.14(expo@56.0.12)(react-native@0.85.3(@babel/core@7.29.7)(@react-native/metro-config@0.85.3(@babel/core@7.29.7)(bufferutil@4.1.0)(utf-8-validate@6.0.6))(@types/react@19.2.16)(bufferutil@4.1.0)(react@19.2.3)(utf-8-validate@6.0.6))(react@19.2.3) @@ -5184,10 +5187,12 @@ packages: '@xmldom/xmldom@0.8.13': resolution: {integrity: sha512-KRYzxepc14G/CEpEGc3Yn+JKaAeT63smlDr+vjB8jRfgTBBI9wRj/nkQEO+ucV8p8I9bfKLWp37uHgFrbntPvw==} engines: {node: '>=10.0.0'} + deprecated: this version has critical issues, please update to the latest version '@xmldom/xmldom@0.9.10': resolution: {integrity: sha512-A9gOqLdi6cV4ibazAjcQufGj0B1y/vDqYrcuP6d/6x8P27gRS8643Dj9o1dEKtB6O7fwxb2FgBmJS2mX7gpvdw==} engines: {node: '>=14.6'} + deprecated: this version has critical issues, please update to the latest version '@yuuang/ffi-rs-android-arm64@1.3.2': resolution: {integrity: sha512-eDYLT0kVBkp7e2BwdRDmt6N1rkeDPUHDefk3ZX0/nok+GLsqfy1WBoSL3Yg7HVXN1EyW8OBVc2uK8Zq8HbmaSA==} @@ -6540,6 +6545,14 @@ packages: react: '*' react-native: '*' + expo-audio@56.0.13: + resolution: {integrity: sha512-pfBmT/8OYbhrb37ECk+fC1EstLBFxMTioF4uZrAEeOj3uCgjproj1qYmkpsQLJIoEOMPSIw1iSfBEMl10v1DDA==} + peerDependencies: + expo: '*' + expo-asset: '*' + react: '*' + react-native: '*' + expo-auth-session@56.0.14: resolution: {integrity: sha512-b6URDBKXVWBjHwypnbCPW6A3PrwYyFqzLXtTrrpTGpmDlsxk7xuz6wIA77sBNziz3hMxt11Nu70iY0fJZYT4jA==} peerDependencies: @@ -16819,6 +16832,13 @@ snapshots: - typescript optional: true + expo-audio@56.0.13(expo-asset@56.0.17(expo@56.0.12)(react-native@0.85.3(@babel/core@7.29.7)(@react-native/metro-config@0.85.3(@babel/core@7.29.7)(bufferutil@4.1.0)(utf-8-validate@6.0.6))(@types/react@19.2.16)(bufferutil@4.1.0)(react@19.2.3)(utf-8-validate@6.0.6))(react@19.2.3)(typescript@6.0.3))(expo@56.0.12)(react-native@0.85.3(@babel/core@7.29.7)(@react-native/metro-config@0.85.3(@babel/core@7.29.7)(bufferutil@4.1.0)(utf-8-validate@6.0.6))(@types/react@19.2.16)(bufferutil@4.1.0)(react@19.2.3)(utf-8-validate@6.0.6))(react@19.2.3): + dependencies: + expo: 56.0.12(8895228379997a2a064f9644cda56ed0) + expo-asset: 56.0.17(expo@56.0.12)(react-native@0.85.3(@babel/core@7.29.7)(@react-native/metro-config@0.85.3(@babel/core@7.29.7)(bufferutil@4.1.0)(utf-8-validate@6.0.6))(@types/react@19.2.16)(bufferutil@4.1.0)(react@19.2.3)(utf-8-validate@6.0.6))(react@19.2.3)(typescript@6.0.3) + react: 19.2.3 + react-native: 0.85.3(@babel/core@7.29.7)(@react-native/metro-config@0.85.3(@babel/core@7.29.7)(bufferutil@4.1.0)(utf-8-validate@6.0.6))(@types/react@19.2.16)(bufferutil@4.1.0)(react@19.2.3)(utf-8-validate@6.0.6) + expo-auth-session@56.0.14(expo@56.0.12)(react-native@0.85.3(@babel/core@7.29.7)(@react-native/metro-config@0.85.3(@babel/core@7.29.7)(bufferutil@4.1.0)(utf-8-validate@6.0.6))(@types/react@19.2.16)(bufferutil@4.1.0)(react@19.2.3)(utf-8-validate@6.0.6))(react@19.2.3): dependencies: expo-application: 56.0.3(expo@56.0.12) From 143c11e5bd991eb199b16624c4853d795a5fed61 Mon Sep 17 00:00:00 2001 From: Yaroslav Kachur <268710642+KachurPro@users.noreply.github.com> Date: Fri, 14 Aug 2026 21:09:13 +0100 Subject: [PATCH 08/21] fix: harden voice transcription state transitions --- .../src/features/threads/ThreadComposer.tsx | 33 +++++++++++++-- .../useMobileVoiceTranscription.ts | 27 +++++++++---- .../voiceTranscriptionSettings.ts | 2 +- apps/web/src/components/chat/ChatComposer.tsx | 40 +++++++++++++++++-- .../components/settings/SettingsPanels.tsx | 11 +++++ apps/web/src/hooks/useVoiceTranscription.ts | 28 ++++++++----- 6 files changed, 114 insertions(+), 27 deletions(-) diff --git a/apps/mobile/src/features/threads/ThreadComposer.tsx b/apps/mobile/src/features/threads/ThreadComposer.tsx index d8dc682b36f..b3451e27db8 100644 --- a/apps/mobile/src/features/threads/ThreadComposer.tsx +++ b/apps/mobile/src/features/threads/ThreadComposer.tsx @@ -530,6 +530,10 @@ export const ThreadComposer = memo(function ThreadComposer(props: ThreadComposer // ── Handle command selection ────────────────────────────── const { onChangeDraftMessage, onUpdateInteractionMode, draftMessage, onSendMessage } = props; + const voiceTranscriptionTargetKey = scopedThreadKey(props.environmentId, props.selectedThread.id); + const voiceTranscriptionTargetKeyRef = useRef(voiceTranscriptionTargetKey); + const voiceTranscriptionOriginTargetKeyRef = useRef(null); + voiceTranscriptionTargetKeyRef.current = voiceTranscriptionTargetKey; const sendCurrentDraft = useCallback(async () => { const threadKey = scopedThreadKey(props.environmentId, props.selectedThread.id); @@ -559,10 +563,17 @@ export const ThreadComposer = memo(function ThreadComposer(props: ThreadComposer const voiceTranscriptionSettings = useMobileVoiceTranscriptionSettings(); const appendVoiceTranscriptToDraft = useCallback( (transcript: string) => { + if ( + voiceTranscriptionOriginTargetKeyRef.current === null || + voiceTranscriptionOriginTargetKeyRef.current !== voiceTranscriptionTargetKeyRef.current + ) { + return false; + } const nextDraft = appendVoiceTranscriptText(draftMessage, transcript); if (nextDraft === draftMessage) return false; setComposerSelection({ start: nextDraft.length, end: nextDraft.length }); onChangeDraftMessage(nextDraft); + voiceTranscriptionOriginTargetKeyRef.current = null; return true; }, [draftMessage, onChangeDraftMessage], @@ -576,6 +587,22 @@ export const ThreadComposer = memo(function ThreadComposer(props: ThreadComposer }); const voiceTranscriptionReady = voiceTranscriptionSettings.loaded && voiceTranscriptionSettings.apiKey.trim().length > 0; + const startVoiceTranscription = useCallback(async () => { + voiceTranscriptionOriginTargetKeyRef.current = voiceTranscriptionTargetKeyRef.current; + await voiceTranscription.start(); + }, [voiceTranscription.start]); + const cancelVoiceTranscription = useCallback(async () => { + voiceTranscriptionOriginTargetKeyRef.current = null; + await voiceTranscription.cancel(); + }, [voiceTranscription.cancel]); + useEffect(() => { + const activeTargetKey = voiceTranscriptionTargetKey; + return () => { + if (voiceTranscriptionOriginTargetKeyRef.current !== activeTargetKey) return; + voiceTranscriptionOriginTargetKeyRef.current = null; + void voiceTranscription.cancel(); + }; + }, [voiceTranscription.cancel, voiceTranscriptionTargetKey]); const handleSend = useCallback(async () => { if (voiceTranscription.status === "recording") { await voiceTranscription.stop("send"); @@ -794,7 +821,7 @@ export const ThreadComposer = memo(function ThreadComposer(props: ThreadComposer elapsedMs={voiceTranscription.elapsedMs} levels={voiceTranscription.levels} sendDisabled={false} - onCancel={() => void voiceTranscription.cancel()} + onCancel={() => void cancelVoiceTranscription()} onStop={() => void voiceTranscription.stop("insert")} onSend={() => void voiceTranscription.stop("send")} /> @@ -880,7 +907,7 @@ export const ThreadComposer = memo(function ThreadComposer(props: ThreadComposer accessibilityLabel="Start dictation" className="h-9 w-9" icon="mic" - onPress={() => void voiceTranscription.start()} + onPress={() => void startVoiceTranscription()} /> ) : null} {showStopAction ? ( @@ -913,7 +940,7 @@ export const ThreadComposer = memo(function ThreadComposer(props: ThreadComposer void voiceTranscription.start()} + onPress={() => void startVoiceTranscription()} showChevron={false} /> ) : null} diff --git a/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts b/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts index 47a59a67f78..18cd213e0ab 100644 --- a/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts +++ b/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts @@ -68,9 +68,8 @@ export function useMobileVoiceTranscription(input: { action, ); if (startingRef.current) { - cancelStartingRef.current = true; - clearRecordingTimeout(); - if (mountedRef.current) setStatus("idle"); + // Permission and recorder preparation are already in flight. Keep the + // requested terminal action and apply it as soon as recording starts. return; } if (statusRef.current !== "recording" || stopInFlightRef.current) return; @@ -79,11 +78,13 @@ export function useMobileVoiceTranscription(input: { try { const beforeStop = await recorder.getStatus(); await recorder.stop(); - const terminalAction = terminalActionRef.current ?? "insert"; const uri = recorder.uri; - terminalActionRef.current = null; await resetAudioMode(); - if (terminalAction === "abort" || beforeStop.durationMillis < MIN_RECORDING_MS || !uri) { + if ( + terminalActionRef.current === "abort" || + beforeStop.durationMillis < MIN_RECORDING_MS || + !uri + ) { if (mountedRef.current) { setStatus("idle"); setLevels(FLAT_LEVELS); @@ -94,8 +95,9 @@ export function useMobileVoiceTranscription(input: { if (mountedRef.current) setStatus("transcribing"); const text = await transcribeMobileVoiceRecording(uri, apiKeyRef.current); if (!mountedRef.current) return; - if (text) { - if (terminalAction === "send") onTranscriptSendRef.current(text); + const finalAction = resolveVoiceTranscriptionAction(terminalActionRef.current, "insert"); + if (text && finalAction !== "abort") { + if (finalAction === "send") onTranscriptSendRef.current(text); else onTranscriptInsertRef.current(text); } setStatus("idle"); @@ -120,6 +122,7 @@ export function useMobileVoiceTranscription(input: { if (startingRef.current) { cancelStartingRef.current = true; clearRecordingTimeout(); + statusRef.current = "idle"; if (mountedRef.current) setStatus("idle"); return; } @@ -153,6 +156,7 @@ export function useMobileVoiceTranscription(input: { terminalActionRef.current = null; setError(null); setLevels(FLAT_LEVELS); + statusRef.current = "recording"; setStatus("recording"); try { const permission = await AudioModule.requestRecordingPermissionsAsync(); @@ -162,6 +166,7 @@ export function useMobileVoiceTranscription(input: { if (cancelStartingRef.current || !mountedRef.current) { startingRef.current = false; cancelStartingRef.current = false; + statusRef.current = "idle"; if (mountedRef.current) setStatus("idle"); return; } @@ -171,14 +176,19 @@ export function useMobileVoiceTranscription(input: { startingRef.current = false; cancelStartingRef.current = false; await resetAudioMode(); + statusRef.current = "idle"; if (mountedRef.current) setStatus("idle"); return; } recorder.record(); startingRef.current = false; + const pendingAction = terminalActionRef.current; timeoutRef.current = setTimeout(() => { void stopRef.current("insert"); }, MAX_RECORDING_MS); + if (pendingAction === "insert" || pendingAction === "send") { + void stopRef.current(pendingAction); + } } catch (cause) { startingRef.current = false; await resetAudioMode(); @@ -188,6 +198,7 @@ export function useMobileVoiceTranscription(input: { } if (!mountedRef.current) return; setError(cause instanceof Error ? cause.message : "Could not start the microphone."); + statusRef.current = "idle"; setStatus("idle"); } }, [recorder, resetAudioMode]); diff --git a/apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.ts b/apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.ts index 8ebc261c3b4..b720a5fabbe 100644 --- a/apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.ts +++ b/apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.ts @@ -44,7 +44,7 @@ export function loadMobileVoiceTranscriptionSettings(): Promise { publish({ apiKey: "", error: "Could not read the saved API key.", - loaded: true, + loaded: false, saving: false, }); }) diff --git a/apps/web/src/components/chat/ChatComposer.tsx b/apps/web/src/components/chat/ChatComposer.tsx index 738f49c71d0..fdbc888b9e4 100644 --- a/apps/web/src/components/chat/ChatComposer.tsx +++ b/apps/web/src/components/chat/ChatComposer.tsx @@ -18,6 +18,7 @@ import { PROVIDER_SEND_TURN_MAX_IMAGE_BYTES, } from "@t3tools/contracts"; import type { EnvironmentConnectionPresentation } from "@t3tools/client-runtime/connection"; +import { scopedThreadKey } from "@t3tools/client-runtime/environment"; import { serializeComposerFileLink } from "@t3tools/shared/composerTrigger"; import { createModelSelection, normalizeModelSlug } from "@t3tools/shared/model"; import { appendVoiceTranscript as appendVoiceTranscriptText } from "@t3tools/shared/voiceTranscription"; @@ -671,6 +672,10 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) onExpandImage, } = props; const isSendDisabled = sendDisabledReason !== null; + const voiceTranscriptionTargetKey = + typeof composerDraftTarget === "string" + ? composerDraftTarget + : scopedThreadKey(composerDraftTarget); // ------------------------------------------------------------------ // Store subscriptions (prompt / images / terminal contexts) @@ -979,6 +984,9 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) const stashPulseKeyRef = useRef(0); const stashPulseTimeoutRef = useRef(null); const submitComposerRef = useRef<(event?: { preventDefault: () => void }) => void>(() => {}); + const voiceTranscriptionTargetKeyRef = useRef(voiceTranscriptionTargetKey); + const voiceTranscriptionOriginTargetKeyRef = useRef(null); + voiceTranscriptionTargetKeyRef.current = voiceTranscriptionTargetKey; /** * Snapshots currently being encoded, keyed by target+prompt+image ids. * Keyed rather than boolean so a genuinely different prompt (or a different @@ -1269,6 +1277,12 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) const appendVoiceTranscriptToPrompt = useCallback( (transcript: string) => { + if ( + voiceTranscriptionOriginTargetKeyRef.current === null || + voiceTranscriptionOriginTargetKeyRef.current !== voiceTranscriptionTargetKeyRef.current + ) { + return false; + } const currentPrompt = promptRef.current; const nextPrompt = appendVoiceTranscriptText(currentPrompt, transcript); if (nextPrompt === currentPrompt) return false; @@ -1278,6 +1292,7 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) setComposerCursor(nextCursor); setComposerTrigger(null); scheduleComposerFocus(); + voiceTranscriptionOriginTargetKeyRef.current = null; return true; }, [promptRef, scheduleComposerFocus, setPrompt], @@ -1295,17 +1310,34 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) }); const voiceTranscriptionReady = settings.voiceTranscriptionEnabled && settings.voiceTranscriptionModel.trim().length > 0; + const startVoiceTranscription = useCallback(async () => { + voiceTranscriptionOriginTargetKeyRef.current = voiceTranscriptionTargetKeyRef.current; + await voiceTranscription.start(); + }, [voiceTranscription.start]); + const cancelVoiceTranscription = useCallback(() => { + voiceTranscriptionOriginTargetKeyRef.current = null; + voiceTranscription.cancel(); + }, [voiceTranscription.cancel]); + + useEffect(() => { + const activeTargetKey = voiceTranscriptionTargetKey; + return () => { + if (voiceTranscriptionOriginTargetKeyRef.current !== activeTargetKey) return; + voiceTranscriptionOriginTargetKeyRef.current = null; + voiceTranscription.cancel(); + }; + }, [voiceTranscription.cancel, voiceTranscriptionTargetKey]); useEffect(() => { if (voiceTranscription.status !== "recording") return; const cancelOnEscape = (event: KeyboardEvent) => { if (event.key !== "Escape") return; event.preventDefault(); - voiceTranscription.cancel(); + cancelVoiceTranscription(); }; window.addEventListener("keydown", cancelOnEscape); return () => window.removeEventListener("keydown", cancelOnEscape); - }, [voiceTranscription.cancel, voiceTranscription.status]); + }, [cancelVoiceTranscription, voiceTranscription.status]); const addComposerImage = useCallback( (image: ComposerImageAttachment) => { @@ -3189,7 +3221,7 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) elapsedMs={voiceTranscription.elapsedMs} levels={voiceTranscription.levels} sendDisabled={voiceTranscriptionSendDisabled} - onCancel={voiceTranscription.cancel} + onCancel={cancelVoiceTranscription} onStop={() => voiceTranscription.stop("insert")} onSend={() => voiceTranscription.stop("send")} /> @@ -3289,7 +3321,7 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) variant="ghost" disabled={isConnecting || projectSelectionRequired} className="rounded-full text-muted-foreground" - onClick={() => void voiceTranscription.start()} + onClick={() => void startVoiceTranscription()} aria-label="Start dictation" > diff --git a/apps/web/src/components/settings/SettingsPanels.tsx b/apps/web/src/components/settings/SettingsPanels.tsx index 0da85a732b3..e2c4cc57a19 100644 --- a/apps/web/src/components/settings/SettingsPanels.tsx +++ b/apps/web/src/components/settings/SettingsPanels.tsx @@ -473,6 +473,11 @@ export function useSettingsRestore(onRestored?: () => void) { DEFAULT_UNIFIED_SETTINGS.textGenerationModelSelection ?? null, ); const isBackgroundActivityDirty = hasChangedBackgroundActivitySettings(settings); + const isVoiceTranscriptionDirty = + settings.voiceTranscriptionEnabled !== DEFAULT_UNIFIED_SETTINGS.voiceTranscriptionEnabled || + settings.voiceTranscriptionProvider !== DEFAULT_UNIFIED_SETTINGS.voiceTranscriptionProvider || + settings.voiceTranscriptionApiKey !== DEFAULT_UNIFIED_SETTINGS.voiceTranscriptionApiKey || + settings.voiceTranscriptionModel !== DEFAULT_UNIFIED_SETTINGS.voiceTranscriptionModel; const changedSettingLabels = useMemo( () => [ @@ -532,10 +537,12 @@ export function useSettingsRestore(onRestored?: () => void) { ? ["Delete confirmation"] : []), ...(isTextGenerationModelDirty ? ["Text generation model"] : []), + ...(isVoiceTranscriptionDirty ? ["Voice dictation"] : []), ], [ isTextGenerationModelDirty, isBackgroundActivityDirty, + isVoiceTranscriptionDirty, settings.confirmThreadArchive, settings.confirmThreadDelete, settings.addProjectBaseDirectory, @@ -650,6 +657,10 @@ export function useSettingsRestore(onRestored?: () => void) { confirmThreadArchive: DEFAULT_UNIFIED_SETTINGS.confirmThreadArchive, confirmThreadDelete: DEFAULT_UNIFIED_SETTINGS.confirmThreadDelete, textGenerationModelSelection: DEFAULT_UNIFIED_SETTINGS.textGenerationModelSelection, + voiceTranscriptionEnabled: DEFAULT_UNIFIED_SETTINGS.voiceTranscriptionEnabled, + voiceTranscriptionProvider: DEFAULT_UNIFIED_SETTINGS.voiceTranscriptionProvider, + voiceTranscriptionApiKey: DEFAULT_UNIFIED_SETTINGS.voiceTranscriptionApiKey, + voiceTranscriptionModel: DEFAULT_UNIFIED_SETTINGS.voiceTranscriptionModel, fontFamilySans: DEFAULT_UNIFIED_SETTINGS.fontFamilySans, fontFamilyComposer: DEFAULT_UNIFIED_SETTINGS.fontFamilyComposer, fontFamilyCode: DEFAULT_UNIFIED_SETTINGS.fontFamilyCode, diff --git a/apps/web/src/hooks/useVoiceTranscription.ts b/apps/web/src/hooks/useVoiceTranscription.ts index 628865fa484..b0b3e30210d 100644 --- a/apps/web/src/hooks/useVoiceTranscription.ts +++ b/apps/web/src/hooks/useVoiceTranscription.ts @@ -81,12 +81,8 @@ export function useVoiceTranscription({ action, ); if (startingRef.current) { - cancelStartingRef.current = true; - cleanupCapture(); - if (mountedRef.current) { - setStatus("idle"); - setElapsedMs(0); - } + // Permission acquisition is already in flight. Preserve the requested + // action and apply it as soon as MediaRecorder starts. return; } const recorder = recorderRef.current; @@ -167,12 +163,12 @@ export function useVoiceTranscription({ }); recorder.addEventListener("stop", () => { const durationMs = Date.now() - startedAtRef.current; - const action = terminalActionRef.current ?? "insert"; - terminalActionRef.current = null; + const requestedAction = terminalActionRef.current ?? "insert"; const blob = new Blob(chunks, { type: recorder.mimeType || mimeType || "audio/webm" }); cleanupCapture(); if (!mountedRef.current || recordingFailed) return; - if (action === "abort" || durationMs < MIN_RECORDING_MS || blob.size === 0) { + if (requestedAction === "abort" || durationMs < MIN_RECORDING_MS || blob.size === 0) { + terminalActionRef.current = null; setStatus("idle"); setElapsedMs(0); return; @@ -182,15 +178,18 @@ export function useVoiceTranscription({ void transcribeVoiceRecording(blob, configRef.current) .then((text) => { if (!mountedRef.current) return; - if (text) { - if (action === "send") onTranscriptSendRef.current(text); + const finalAction = terminalActionRef.current ?? requestedAction; + if (text && finalAction !== "abort") { + if (finalAction === "send") onTranscriptSendRef.current(text); else onTranscriptInsertRef.current(text); } + terminalActionRef.current = null; setStatus("idle"); setElapsedMs(0); }) .catch((cause: unknown) => { if (!mountedRef.current) return; + terminalActionRef.current = null; setError(cause instanceof Error ? cause.message : "Voice transcription failed."); setStatus("idle"); }); @@ -230,6 +229,13 @@ export function useVoiceTranscription({ }, MAX_RECORDING_MS); recorder.start(250); startingRef.current = false; + const pendingAction = terminalActionRef.current; + if ( + (pendingAction === "insert" || pendingAction === "send") && + recorder.state === "recording" + ) { + recorder.stop(); + } } catch (cause) { startingRef.current = false; cleanupCapture(); From edb4c35342656b613ae52ea77f2799f359215d6a Mon Sep 17 00:00:00 2001 From: Yaroslav Kachur <268710642+KachurPro@users.noreply.github.com> Date: Fri, 14 Aug 2026 21:24:19 +0100 Subject: [PATCH 09/21] fix: recover voice setup after transient cancellation --- .../useMobileVoiceTranscription.ts | 28 ++++++- .../components/settings/SettingsPanels.tsx | 76 +++++++++++++------ apps/web/src/hooks/useVoiceTranscription.ts | 27 ++++++- 3 files changed, 106 insertions(+), 25 deletions(-) diff --git a/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts b/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts index 18cd213e0ab..fcad3dff07c 100644 --- a/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts +++ b/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts @@ -37,6 +37,7 @@ export function useMobileVoiceTranscription(input: { const statusRef = useRef(status); const startingRef = useRef(false); const cancelStartingRef = useRef(false); + const restartAfterCancellationRef = useRef(false); const stopInFlightRef = useRef(false); const terminalActionRef = useRef(null); const timeoutRef = useRef | null>(null); @@ -44,6 +45,7 @@ export function useMobileVoiceTranscription(input: { const apiKeyRef = useRef(input.apiKey); const onTranscriptInsertRef = useRef(input.onTranscriptInsert); const onTranscriptSendRef = useRef(input.onTranscriptSend); + const startRef = useRef<() => Promise>(async () => undefined); const stopRef = useRef<(action?: "insert" | "send") => Promise>(async () => undefined); statusRef.current = status; apiKeyRef.current = input.apiKey; @@ -145,7 +147,11 @@ export function useMobileVoiceTranscription(input: { }, [clearRecordingTimeout, recorder, resetAudioMode]); const start = useCallback(async () => { - if (startingRef.current || statusRef.current !== "idle") return; + if (startingRef.current) { + if (cancelStartingRef.current) restartAfterCancellationRef.current = true; + return; + } + if (statusRef.current !== "idle") return; if (!apiKeyRef.current.trim()) { setError("Save an OpenAI API key in Settings first."); return; @@ -153,6 +159,7 @@ export function useMobileVoiceTranscription(input: { startingRef.current = true; cancelStartingRef.current = false; + restartAfterCancellationRef.current = false; terminalActionRef.current = null; setError(null); setLevels(FLAT_LEVELS); @@ -168,6 +175,11 @@ export function useMobileVoiceTranscription(input: { cancelStartingRef.current = false; statusRef.current = "idle"; if (mountedRef.current) setStatus("idle"); + const shouldRestart = restartAfterCancellationRef.current; + restartAfterCancellationRef.current = false; + if (shouldRestart && mountedRef.current) { + queueMicrotask(() => void startRef.current()); + } return; } await setAudioModeAsync({ allowsRecording: true, playsInSilentMode: true }); @@ -178,6 +190,11 @@ export function useMobileVoiceTranscription(input: { await resetAudioMode(); statusRef.current = "idle"; if (mountedRef.current) setStatus("idle"); + const shouldRestart = restartAfterCancellationRef.current; + restartAfterCancellationRef.current = false; + if (shouldRestart && mountedRef.current) { + queueMicrotask(() => void startRef.current()); + } return; } recorder.record(); @@ -194,6 +211,13 @@ export function useMobileVoiceTranscription(input: { await resetAudioMode(); if (cancelStartingRef.current) { cancelStartingRef.current = false; + statusRef.current = "idle"; + if (mountedRef.current) setStatus("idle"); + const shouldRestart = restartAfterCancellationRef.current; + restartAfterCancellationRef.current = false; + if (shouldRestart && mountedRef.current) { + queueMicrotask(() => void startRef.current()); + } return; } if (!mountedRef.current) return; @@ -202,6 +226,7 @@ export function useMobileVoiceTranscription(input: { setStatus("idle"); } }, [recorder, resetAudioMode]); + startRef.current = start; useEffect(() => { if (status !== "recording") return; @@ -215,6 +240,7 @@ export function useMobileVoiceTranscription(input: { return () => { mountedRef.current = false; terminalActionRef.current = "abort"; + restartAfterCancellationRef.current = false; clearRecordingTimeout(); if (recorder.isRecording) void recorder.stop().catch(() => undefined); void resetAudioMode(); diff --git a/apps/web/src/components/settings/SettingsPanels.tsx b/apps/web/src/components/settings/SettingsPanels.tsx index e2c4cc57a19..20fefdfe0d4 100644 --- a/apps/web/src/components/settings/SettingsPanels.tsx +++ b/apps/web/src/components/settings/SettingsPanels.tsx @@ -1661,6 +1661,9 @@ function VoiceDictationSettingsSection() { const settings = usePrimarySettings(); const updateSettings = useUpdatePrimarySettings(); const [environmentApiKeys, setEnvironmentApiKeys] = useState({ openai: false, groq: false }); + const [environmentStatusLoading, setEnvironmentStatusLoading] = useState(true); + const [environmentStatusError, setEnvironmentStatusError] = useState(null); + const [environmentStatusAttempt, setEnvironmentStatusAttempt] = useState(0); const [models, setModels] = useState([]); const [modelsLoading, setModelsLoading] = useState(false); const [modelsError, setModelsError] = useState(null); @@ -1670,15 +1673,25 @@ function VoiceDictationSettingsSection() { useEffect(() => { let active = true; + setEnvironmentStatusLoading(true); + setEnvironmentStatusError(null); void readVoiceTranscriptionEnvironmentStatus() .then((status) => { - if (active) setEnvironmentApiKeys(status); + if (!active) return; + setEnvironmentApiKeys(status); + setEnvironmentStatusLoading(false); }) - .catch(() => undefined); + .catch((cause: unknown) => { + if (!active) return; + setEnvironmentStatusLoading(false); + setEnvironmentStatusError( + cause instanceof Error ? cause.message : "Could not check server transcription keys.", + ); + }); return () => { active = false; }; - }, []); + }, [environmentStatusAttempt]); const providerLabel = provider === "openai" ? "OpenAI" : "Groq"; const environmentVariable = TRANSCRIPTION_API_KEY_ENV[provider]; @@ -1729,7 +1742,11 @@ function VoiceDictationSettingsSection() { }, [model, models, updateSettings]); const modelDescription = !hasApiKey - ? `Save an API key to load ${providerLabel} transcription models.` + ? environmentStatusLoading + ? `Checking the connected server for ${environmentVariable}…` + : environmentStatusError + ? `${environmentStatusError} Add a client key or retry the server check.` + : `Save an API key to load ${providerLabel} transcription models.` : modelsLoading ? `Loading models available from ${providerLabel}…` : modelsError @@ -1772,26 +1789,41 @@ function VoiceDictationSettingsSection() { - updateSettings({ - voiceTranscriptionApiKey: event.target.value, - voiceTranscriptionModel: "", - voiceTranscriptionEnabled: false, - }) - } - placeholder={hasEnvironmentApiKey ? `Using ${environmentVariable}` : "Required"} - aria-label={`${providerLabel} transcription API key`} - /> +
+ + updateSettings({ + voiceTranscriptionApiKey: event.target.value, + voiceTranscriptionModel: "", + voiceTranscriptionEnabled: false, + }) + } + placeholder={hasEnvironmentApiKey ? `Using ${environmentVariable}` : "Required"} + aria-label={`${providerLabel} transcription API key`} + /> + {environmentStatusError ? ( + + ) : null} +
} /> (FLAT_LEVELS); const [error, setError] = useState(null); const recorderRef = useRef(null); + const statusRef = useRef(status); const startingRef = useRef(false); const cancelStartingRef = useRef(false); + const restartAfterCancellationRef = useRef(false); const streamRef = useRef(null); const audioContextRef = useRef(null); const intervalsRef = useRef([]); @@ -45,6 +47,8 @@ export function useVoiceTranscription({ const configRef = useRef(config); const onTranscriptInsertRef = useRef(onTranscriptInsert); const onTranscriptSendRef = useRef(onTranscriptSend); + const startRef = useRef<() => Promise>(async () => undefined); + statusRef.current = status; configRef.current = config; onTranscriptInsertRef.current = onTranscriptInsert; onTranscriptSendRef.current = onTranscriptSend; @@ -67,6 +71,7 @@ export function useVoiceTranscription({ mountedRef.current = false; startingRef.current = false; cancelStartingRef.current = true; + restartAfterCancellationRef.current = false; terminalActionRef.current = "abort"; const recorder = recorderRef.current; if (recorder?.state === "recording") recorder.stop(); @@ -97,6 +102,7 @@ export function useVoiceTranscription({ cancelStartingRef.current = true; cleanupCapture(); if (mountedRef.current) { + statusRef.current = "idle"; setStatus("idle"); setElapsedMs(0); } @@ -107,7 +113,11 @@ export function useVoiceTranscription({ }, [cleanupCapture]); const start = useCallback(async () => { - if (startingRef.current || status !== "idle") return; + if (startingRef.current) { + if (cancelStartingRef.current) restartAfterCancellationRef.current = true; + return; + } + if (statusRef.current !== "idle") return; startingRef.current = true; terminalActionRef.current = null; setError(null); @@ -125,6 +135,8 @@ export function useVoiceTranscription({ } cancelStartingRef.current = false; + restartAfterCancellationRef.current = false; + statusRef.current = "recording"; setStatus("recording"); try { const audioContext = new AudioContext(); @@ -140,6 +152,11 @@ export function useVoiceTranscription({ startingRef.current = false; cancelStartingRef.current = false; terminalActionRef.current = null; + const shouldRestart = restartAfterCancellationRef.current; + restartAfterCancellationRef.current = false; + if (shouldRestart && mountedRef.current) { + queueMicrotask(() => void startRef.current()); + } return; } @@ -242,6 +259,11 @@ export function useVoiceTranscription({ terminalActionRef.current = null; if (cancelStartingRef.current) { cancelStartingRef.current = false; + const shouldRestart = restartAfterCancellationRef.current; + restartAfterCancellationRef.current = false; + if (shouldRestart && mountedRef.current) { + queueMicrotask(() => void startRef.current()); + } return; } setStatus("idle"); @@ -251,7 +273,8 @@ export function useVoiceTranscription({ : "Could not start the microphone.", ); } - }, [cleanupCapture, status]); + }, [cleanupCapture]); + startRef.current = start; return { status, elapsedMs, levels, error, start, stop, cancel } as const; } From b93fde1f1d30851e2658e6492c6032c3d28a5843 Mon Sep 17 00:00:00 2001 From: Yaroslav Kachur <268710642+KachurPro@users.noreply.github.com> Date: Fri, 14 Aug 2026 21:34:15 +0100 Subject: [PATCH 10/21] fix: cancel in-flight voice transcription --- .../MobileVoiceTranscriptionPanel.tsx | 6 +++ .../useMobileVoiceTranscription.ts | 38 +++++++++++++++++-- .../chat/VoiceTranscriptionPanel.tsx | 10 +++++ apps/web/src/hooks/useVoiceTranscription.ts | 23 ++++++++++- 4 files changed, 72 insertions(+), 5 deletions(-) diff --git a/apps/mobile/src/features/voice-dictation/MobileVoiceTranscriptionPanel.tsx b/apps/mobile/src/features/voice-dictation/MobileVoiceTranscriptionPanel.tsx index 73f7b49f967..c4342dd8d31 100644 --- a/apps/mobile/src/features/voice-dictation/MobileVoiceTranscriptionPanel.tsx +++ b/apps/mobile/src/features/voice-dictation/MobileVoiceTranscriptionPanel.tsx @@ -23,6 +23,12 @@ export function MobileVoiceTranscriptionPanel(props: { if (props.status === "transcribing") { return ( + Processing recording… diff --git a/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts b/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts index fcad3dff07c..96a7db915ff 100644 --- a/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts +++ b/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts @@ -40,6 +40,7 @@ export function useMobileVoiceTranscription(input: { const restartAfterCancellationRef = useRef(false); const stopInFlightRef = useRef(false); const terminalActionRef = useRef(null); + const transcriptionAttemptRef = useRef(0); const timeoutRef = useRef | null>(null); const mountedRef = useRef(true); const apiKeyRef = useRef(input.apiKey); @@ -65,6 +66,7 @@ export function useMobileVoiceTranscription(input: { const stop = useCallback( async (action: "insert" | "send" = "insert") => { + let transcriptionAttempt: number | null = null; terminalActionRef.current = resolveVoiceTranscriptionAction( terminalActionRef.current, action, @@ -88,6 +90,7 @@ export function useMobileVoiceTranscription(input: { !uri ) { if (mountedRef.current) { + statusRef.current = "idle"; setStatus("idle"); setLevels(FLAT_LEVELS); } @@ -95,24 +98,41 @@ export function useMobileVoiceTranscription(input: { } if (mountedRef.current) setStatus("transcribing"); + statusRef.current = "transcribing"; + transcriptionAttempt = ++transcriptionAttemptRef.current; const text = await transcribeMobileVoiceRecording(uri, apiKeyRef.current); - if (!mountedRef.current) return; + if (!mountedRef.current || transcriptionAttempt !== transcriptionAttemptRef.current) { + return; + } const finalAction = resolveVoiceTranscriptionAction(terminalActionRef.current, "insert"); if (text && finalAction !== "abort") { if (finalAction === "send") onTranscriptSendRef.current(text); else onTranscriptInsertRef.current(text); } + statusRef.current = "idle"; setStatus("idle"); setLevels(FLAT_LEVELS); } catch (cause) { + if ( + transcriptionAttempt !== null && + transcriptionAttempt !== transcriptionAttemptRef.current + ) { + return; + } await resetAudioMode(); if (!mountedRef.current) return; setError(cause instanceof Error ? cause.message : "Voice transcription failed."); + statusRef.current = "idle"; setStatus("idle"); setLevels(FLAT_LEVELS); } finally { - stopInFlightRef.current = false; - terminalActionRef.current = null; + if ( + transcriptionAttempt === null || + transcriptionAttempt === transcriptionAttemptRef.current + ) { + stopInFlightRef.current = false; + terminalActionRef.current = null; + } } }, [clearRecordingTimeout, recorder, resetAudioMode], @@ -128,6 +148,17 @@ export function useMobileVoiceTranscription(input: { if (mountedRef.current) setStatus("idle"); return; } + if (statusRef.current === "transcribing") { + transcriptionAttemptRef.current += 1; + stopInFlightRef.current = false; + terminalActionRef.current = null; + statusRef.current = "idle"; + if (mountedRef.current) { + setStatus("idle"); + setLevels(FLAT_LEVELS); + } + return; + } if (statusRef.current !== "recording" || stopInFlightRef.current) return; stopInFlightRef.current = true; clearRecordingTimeout(); @@ -241,6 +272,7 @@ export function useMobileVoiceTranscription(input: { mountedRef.current = false; terminalActionRef.current = "abort"; restartAfterCancellationRef.current = false; + transcriptionAttemptRef.current += 1; clearRecordingTimeout(); if (recorder.isRecording) void recorder.stop().catch(() => undefined); void resetAudioMode(); diff --git a/apps/web/src/components/chat/VoiceTranscriptionPanel.tsx b/apps/web/src/components/chat/VoiceTranscriptionPanel.tsx index e90bedceb3c..073ff1cfcd4 100644 --- a/apps/web/src/components/chat/VoiceTranscriptionPanel.tsx +++ b/apps/web/src/components/chat/VoiceTranscriptionPanel.tsx @@ -40,6 +40,16 @@ export function VoiceTranscriptionPanel({ role="status" aria-label="Processing recording" > + Processing recording… diff --git a/apps/web/src/hooks/useVoiceTranscription.ts b/apps/web/src/hooks/useVoiceTranscription.ts index 4e62e698658..9ec18ca338c 100644 --- a/apps/web/src/hooks/useVoiceTranscription.ts +++ b/apps/web/src/hooks/useVoiceTranscription.ts @@ -42,6 +42,7 @@ export function useVoiceTranscription({ const intervalsRef = useRef([]); const timeoutRef = useRef(null); const startedAtRef = useRef(0); + const transcriptionAttemptRef = useRef(0); const terminalActionRef = useRef(null); const mountedRef = useRef(true); const configRef = useRef(config); @@ -72,6 +73,7 @@ export function useVoiceTranscription({ startingRef.current = false; cancelStartingRef.current = true; restartAfterCancellationRef.current = false; + transcriptionAttemptRef.current += 1; terminalActionRef.current = "abort"; const recorder = recorderRef.current; if (recorder?.state === "recording") recorder.stop(); @@ -108,6 +110,14 @@ export function useVoiceTranscription({ } return; } + if (statusRef.current === "transcribing") { + transcriptionAttemptRef.current += 1; + terminalActionRef.current = null; + statusRef.current = "idle"; + setStatus("idle"); + setElapsedMs(0); + return; + } const recorder = recorderRef.current; if (recorder?.state === "recording") recorder.stop(); }, [cleanupCapture]); @@ -186,28 +196,37 @@ export function useVoiceTranscription({ if (!mountedRef.current || recordingFailed) return; if (requestedAction === "abort" || durationMs < MIN_RECORDING_MS || blob.size === 0) { terminalActionRef.current = null; + statusRef.current = "idle"; setStatus("idle"); setElapsedMs(0); return; } + const transcriptionAttempt = ++transcriptionAttemptRef.current; + statusRef.current = "transcribing"; setStatus("transcribing"); void transcribeVoiceRecording(blob, configRef.current) .then((text) => { - if (!mountedRef.current) return; + if (!mountedRef.current || transcriptionAttempt !== transcriptionAttemptRef.current) { + return; + } const finalAction = terminalActionRef.current ?? requestedAction; if (text && finalAction !== "abort") { if (finalAction === "send") onTranscriptSendRef.current(text); else onTranscriptInsertRef.current(text); } terminalActionRef.current = null; + statusRef.current = "idle"; setStatus("idle"); setElapsedMs(0); }) .catch((cause: unknown) => { - if (!mountedRef.current) return; + if (!mountedRef.current || transcriptionAttempt !== transcriptionAttemptRef.current) { + return; + } terminalActionRef.current = null; setError(cause instanceof Error ? cause.message : "Voice transcription failed."); + statusRef.current = "idle"; setStatus("idle"); }); }); From cc2c006b8e9e2c27e30711be2389b3cf40f549a7 Mon Sep 17 00:00:00 2001 From: Yaroslav Kachur <268710642+KachurPro@users.noreply.github.com> Date: Fri, 14 Aug 2026 21:45:39 +0100 Subject: [PATCH 11/21] fix: keep web voice cancellation retryable --- apps/web/src/components/chat/ChatComposer.tsx | 2 +- apps/web/src/hooks/useVoiceTranscription.ts | 1 + 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/apps/web/src/components/chat/ChatComposer.tsx b/apps/web/src/components/chat/ChatComposer.tsx index fdbc888b9e4..eecf6f176da 100644 --- a/apps/web/src/components/chat/ChatComposer.tsx +++ b/apps/web/src/components/chat/ChatComposer.tsx @@ -1329,7 +1329,7 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) }, [voiceTranscription.cancel, voiceTranscriptionTargetKey]); useEffect(() => { - if (voiceTranscription.status !== "recording") return; + if (voiceTranscription.status === "idle") return; const cancelOnEscape = (event: KeyboardEvent) => { if (event.key !== "Escape") return; event.preventDefault(); diff --git a/apps/web/src/hooks/useVoiceTranscription.ts b/apps/web/src/hooks/useVoiceTranscription.ts index 9ec18ca338c..28caae793c8 100644 --- a/apps/web/src/hooks/useVoiceTranscription.ts +++ b/apps/web/src/hooks/useVoiceTranscription.ts @@ -285,6 +285,7 @@ export function useVoiceTranscription({ } return; } + statusRef.current = "idle"; setStatus("idle"); setError( cause instanceof DOMException && cause.name === "NotAllowedError" From 692d3f0f7473620dbbd9c0268d1ea2026afae731 Mon Sep 17 00:00:00 2001 From: Yaroslav Kachur <268710642+KachurPro@users.noreply.github.com> Date: Fri, 14 Aug 2026 21:51:12 +0100 Subject: [PATCH 12/21] fix: preserve stop control during active turns --- apps/mobile/src/features/threads/ThreadComposer.tsx | 5 ++++- apps/web/src/components/chat/ChatComposer.tsx | 7 +++++-- 2 files changed, 9 insertions(+), 3 deletions(-) diff --git a/apps/mobile/src/features/threads/ThreadComposer.tsx b/apps/mobile/src/features/threads/ThreadComposer.tsx index b3451e27db8..718ef41884d 100644 --- a/apps/mobile/src/features/threads/ThreadComposer.tsx +++ b/apps/mobile/src/features/threads/ThreadComposer.tsx @@ -588,9 +588,10 @@ export const ThreadComposer = memo(function ThreadComposer(props: ThreadComposer const voiceTranscriptionReady = voiceTranscriptionSettings.loaded && voiceTranscriptionSettings.apiKey.trim().length > 0; const startVoiceTranscription = useCallback(async () => { + if (showStopAction) return; voiceTranscriptionOriginTargetKeyRef.current = voiceTranscriptionTargetKeyRef.current; await voiceTranscription.start(); - }, [voiceTranscription.start]); + }, [showStopAction, voiceTranscription.start]); const cancelVoiceTranscription = useCallback(async () => { voiceTranscriptionOriginTargetKeyRef.current = null; await voiceTranscription.cancel(); @@ -906,6 +907,7 @@ export const ThreadComposer = memo(function ThreadComposer(props: ThreadComposer void startVoiceTranscription()} /> @@ -939,6 +941,7 @@ export const ThreadComposer = memo(function ThreadComposer(props: ThreadComposer {voiceTranscriptionReady ? ( void startVoiceTranscription()} showChevron={false} diff --git a/apps/web/src/components/chat/ChatComposer.tsx b/apps/web/src/components/chat/ChatComposer.tsx index eecf6f176da..2fb792a0cbb 100644 --- a/apps/web/src/components/chat/ChatComposer.tsx +++ b/apps/web/src/components/chat/ChatComposer.tsx @@ -1311,9 +1311,10 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) const voiceTranscriptionReady = settings.voiceTranscriptionEnabled && settings.voiceTranscriptionModel.trim().length > 0; const startVoiceTranscription = useCallback(async () => { + if (phase === "running") return; voiceTranscriptionOriginTargetKeyRef.current = voiceTranscriptionTargetKeyRef.current; await voiceTranscription.start(); - }, [voiceTranscription.start]); + }, [phase, voiceTranscription.start]); const cancelVoiceTranscription = useCallback(() => { voiceTranscriptionOriginTargetKeyRef.current = null; voiceTranscription.cancel(); @@ -3319,7 +3320,9 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) type="button" size="icon-sm" variant="ghost" - disabled={isConnecting || projectSelectionRequired} + disabled={ + phase === "running" || isConnecting || projectSelectionRequired + } className="rounded-full text-muted-foreground" onClick={() => void startVoiceTranscription()} aria-label="Start dictation" From 78fc154c7580804bfb788080204b4bc43b722773 Mon Sep 17 00:00:00 2001 From: Yaroslav Kachur <268710642+KachurPro@users.noreply.github.com> Date: Fri, 14 Aug 2026 22:36:20 +0100 Subject: [PATCH 13/21] feat(mobile): configure voice transcription provider --- .../features/settings/SettingsRouteScreen.tsx | 8 +- .../SettingsVoiceDictationRouteScreen.tsx | 181 +++++++++++++++--- .../src/features/threads/ThreadComposer.tsx | 10 +- .../mobileVoiceTranscription.test.ts | 46 ++++- .../mobileVoiceTranscription.ts | 64 ++++++- .../useMobileVoiceTranscription.ts | 25 ++- .../voiceTranscriptionSettings.test.ts | 103 ++++++++++ .../voiceTranscriptionSettings.ts | 157 +++++++++++++-- docs/user/voice-dictation.md | 10 +- 9 files changed, 527 insertions(+), 77 deletions(-) create mode 100644 apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.test.ts diff --git a/apps/mobile/src/features/settings/SettingsRouteScreen.tsx b/apps/mobile/src/features/settings/SettingsRouteScreen.tsx index d87de1555a1..4b0a86ea865 100644 --- a/apps/mobile/src/features/settings/SettingsRouteScreen.tsx +++ b/apps/mobile/src/features/settings/SettingsRouteScreen.tsx @@ -46,7 +46,10 @@ import { useSavedRemoteConnections } from "../../state/use-remote-environment-re import { SettingsRow } from "./components/SettingsRow"; import { SettingsSection } from "./components/SettingsSection"; import { SettingsSwitchRow } from "./components/SettingsSwitchRow"; -import { useMobileVoiceTranscriptionSettings } from "../voice-dictation/voiceTranscriptionSettings"; +import { + activeMobileVoiceTranscriptionConfig, + useMobileVoiceTranscriptionSettings, +} from "../voice-dictation/voiceTranscriptionSettings"; type NotificationStatus = "checking" | "enabled" | "disabled" | "unsupported"; type LiveActivityStatus = "checking" | "enabled" | "disabled" | "signed-out" | "linking"; @@ -529,6 +532,7 @@ function GeneralSettingsSection() { !AsyncResult.isSuccess(preferencesResult) || preferencesResult.value.autoSettleOnMerge !== false; const voiceTranscription = useMobileVoiceTranscriptionSettings(); + const voiceTranscriptionConfig = activeMobileVoiceTranscriptionConfig(voiceTranscription); return ( @@ -537,7 +541,7 @@ function GeneralSettingsSection() { icon="mic" label="Voice Dictation" value={ - voiceTranscription.loaded && voiceTranscription.apiKey.trim().length > 0 + voiceTranscription.loaded && voiceTranscriptionConfig.apiKey.trim().length > 0 ? "Enabled" : "Set up" } diff --git a/apps/mobile/src/features/settings/SettingsVoiceDictationRouteScreen.tsx b/apps/mobile/src/features/settings/SettingsVoiceDictationRouteScreen.tsx index 0638993c863..b15a39c519f 100644 --- a/apps/mobile/src/features/settings/SettingsVoiceDictationRouteScreen.tsx +++ b/apps/mobile/src/features/settings/SettingsVoiceDictationRouteScreen.tsx @@ -1,40 +1,82 @@ +import type { VoiceTranscriptionProvider } from "@t3tools/contracts"; import { useEffect, useState } from "react"; import { ActivityIndicator, Alert, Pressable, ScrollView, TextInput, View } from "react-native"; import { useSafeAreaInsets } from "react-native-safe-area-context"; import { AppText as Text } from "../../components/AppText"; +import { SymbolView } from "../../components/AppSymbol"; import { useThemeColor } from "../../lib/useThemeColor"; import { - saveMobileVoiceTranscriptionApiKey, + MOBILE_VOICE_TRANSCRIPTION_PROVIDERS, + mobileVoiceTranscriptionProviderConfig, +} from "../voice-dictation/mobileVoiceTranscription"; +import { + type MobileVoiceTranscriptionProviderSettingsMap, + saveMobileVoiceTranscriptionSettings, useMobileVoiceTranscriptionSettings, } from "../voice-dictation/voiceTranscriptionSettings"; import { SettingsSection } from "./components/SettingsSection"; +const EMPTY_PROVIDER_SETTINGS: MobileVoiceTranscriptionProviderSettingsMap = { + openai: { + apiKey: "", + model: mobileVoiceTranscriptionProviderConfig("openai").defaultModel, + }, + groq: { + apiKey: "", + model: mobileVoiceTranscriptionProviderConfig("groq").defaultModel, + }, +}; + export function SettingsVoiceDictationRouteScreen() { const insets = useSafeAreaInsets(); const settings = useMobileVoiceTranscriptionSettings(); - const [draftApiKey, setDraftApiKey] = useState(""); + const [draftProvider, setDraftProvider] = useState("openai"); + const [draftProviders, setDraftProviders] = + useState(EMPTY_PROVIDER_SETTINGS); const [draftInitialized, setDraftInitialized] = useState(false); const foreground = useThemeColor("--color-foreground"); const placeholder = useThemeColor("--color-foreground-muted"); + const checkmarkColor = useThemeColor("--color-icon"); + const providerConfig = mobileVoiceTranscriptionProviderConfig(draftProvider); + const providerSettings = draftProviders[draftProvider]; useEffect(() => { if (!settings.loaded || draftInitialized) return; - setDraftApiKey(settings.apiKey); + setDraftProvider(settings.provider); + setDraftProviders({ + openai: { ...settings.providers.openai }, + groq: { ...settings.providers.groq }, + }); setDraftInitialized(true); - }, [draftInitialized, settings.apiKey, settings.loaded]); + }, [draftInitialized, settings.loaded, settings.provider, settings.providers]); + + const updateSelectedProvider = ( + patch: Partial, + ) => { + setDraftProviders((current) => ({ + ...current, + [draftProvider]: { ...current[draftProvider], ...patch }, + })); + }; const save = async () => { try { - await saveMobileVoiceTranscriptionApiKey(draftApiKey); + await saveMobileVoiceTranscriptionSettings({ + provider: draftProvider, + providers: draftProviders, + }); Alert.alert( - draftApiKey.trim() ? "Voice dictation enabled" : "Voice dictation disabled", - draftApiKey.trim() - ? "The microphone is now available in the message composer." - : "The saved API key was removed.", + providerSettings.apiKey.trim() ? "Voice dictation enabled" : "Voice dictation disabled", + providerSettings.apiKey.trim() + ? `The microphone now uses ${providerConfig.label} with ${providerSettings.model.trim()}.` + : `No ${providerConfig.label} API key is selected, so the microphone is hidden.`, ); } catch (cause) { - Alert.alert("Could not save API key", cause instanceof Error ? cause.message : "Try again."); + Alert.alert( + "Could not save voice dictation", + cause instanceof Error ? cause.message : "Try again.", + ); } }; @@ -48,38 +90,125 @@ export function SettingsVoiceDictationRouteScreen() { contentContainerClassName="gap-4 px-5 pt-4" contentContainerStyle={{ paddingBottom: Math.max(insets.bottom, 18) + 18 }} > - + + {MOBILE_VOICE_TRANSCRIPTION_PROVIDERS.map((provider, index) => ( + setDraftProvider(provider.id)} + className={ + index === 0 + ? "flex-row items-center gap-4 p-4" + : "flex-row items-center gap-4 border-t border-border-subtle p-4" + } + > + + {provider.label} + + {provider.id === "openai" + ? "GPT transcription models" + : "Fast OpenAI-compatible Whisper models"} + + + {draftProvider === provider.id ? ( + + ) : null} + + ))} + + + updateSelectedProvider({ apiKey })} + className="rounded-xl bg-subtle px-4 py-3 text-base" + style={{ color: foreground }} + accessibilityLabel={`${providerConfig.label} API key for voice dictation`} + /> + + + + + + + Choose a known model or enter any compatible model ID. + + updateSelectedProvider({ model })} className="rounded-xl bg-subtle px-4 py-3 text-base" style={{ color: foreground }} - accessibilityLabel="OpenAI API key for voice dictation" + accessibilityLabel="Voice transcription model ID" /> + + {providerConfig.modelOptions.map((model) => ( void save()} - className="h-11 items-center justify-center rounded-full bg-primary disabled:opacity-50" + onPress={() => updateSelectedProvider({ model: model.id })} + className="flex-row items-center gap-4 border-t border-border-subtle p-4" > - {settings.saving ? ( - - ) : ( - Save - )} + + {model.label} + {model.id} + + {providerSettings.model.trim() === model.id ? ( + + ) : null} - + ))} + + void save()} + className="h-11 items-center justify-center rounded-full bg-primary disabled:opacity-50" + > + {settings.saving ? ( + + ) : ( + Save + )} + + - The key stays in the iPhone secure store. Once it is saved, a microphone appears in the - composer. Audio is sent directly to OpenAI using gpt-4o-mini-transcribe. + Each provider keeps its own key and model in the iPhone secure store. Audio is sent + directly to the selected provider. Clearing the selected key hides the microphone. {settings.error ? ( {settings.error} diff --git a/apps/mobile/src/features/threads/ThreadComposer.tsx b/apps/mobile/src/features/threads/ThreadComposer.tsx index 718ef41884d..7bdfd6f9b09 100644 --- a/apps/mobile/src/features/threads/ThreadComposer.tsx +++ b/apps/mobile/src/features/threads/ThreadComposer.tsx @@ -77,7 +77,10 @@ import { } from "./use-thread-settings-sheet-presentation"; import { MobileVoiceTranscriptionPanel } from "../voice-dictation/MobileVoiceTranscriptionPanel"; import { useMobileVoiceTranscription } from "../voice-dictation/useMobileVoiceTranscription"; -import { useMobileVoiceTranscriptionSettings } from "../voice-dictation/voiceTranscriptionSettings"; +import { + activeMobileVoiceTranscriptionConfig, + useMobileVoiceTranscriptionSettings, +} from "../voice-dictation/voiceTranscriptionSettings"; /** * Height of the collapsed composer (pill + vertical padding, excluding safe-area inset). @@ -561,6 +564,7 @@ export const ThreadComposer = memo(function ThreadComposer(props: ThreadComposer props.selectedThread.title, ]); const voiceTranscriptionSettings = useMobileVoiceTranscriptionSettings(); + const voiceTranscriptionConfig = activeMobileVoiceTranscriptionConfig(voiceTranscriptionSettings); const appendVoiceTranscriptToDraft = useCallback( (transcript: string) => { if ( @@ -579,14 +583,14 @@ export const ThreadComposer = memo(function ThreadComposer(props: ThreadComposer [draftMessage, onChangeDraftMessage], ); const voiceTranscription = useMobileVoiceTranscription({ - apiKey: voiceTranscriptionSettings.apiKey, + ...voiceTranscriptionConfig, onTranscriptInsert: appendVoiceTranscriptToDraft, onTranscriptSend: (transcript) => { if (appendVoiceTranscriptToDraft(transcript)) void sendCurrentDraft(); }, }); const voiceTranscriptionReady = - voiceTranscriptionSettings.loaded && voiceTranscriptionSettings.apiKey.trim().length > 0; + voiceTranscriptionSettings.loaded && voiceTranscriptionConfig.apiKey.trim().length > 0; const startVoiceTranscription = useCallback(async () => { if (showStopAction) return; voiceTranscriptionOriginTargetKeyRef.current = voiceTranscriptionTargetKeyRef.current; diff --git a/apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.test.ts b/apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.test.ts index 3ddb5647f8c..a3c791d0172 100644 --- a/apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.test.ts +++ b/apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.test.ts @@ -1,16 +1,26 @@ import { afterEach, describe, expect, it, vi } from "vite-plus/test"; -import { - MOBILE_TRANSCRIPTION_MODEL, - transcribeMobileVoiceRecording, -} from "./mobileVoiceTranscription"; +import { transcribeMobileVoiceRecording } from "./mobileVoiceTranscription"; afterEach(() => { vi.unstubAllGlobals(); }); describe("transcribeMobileVoiceRecording", () => { - it("sends an iPhone recording directly to OpenAI", async () => { + it.each([ + { + provider: "openai" as const, + apiKey: " openai-secret ", + model: "gpt-4o-transcribe", + endpoint: "https://api.openai.com/v1/audio/transcriptions", + }, + { + provider: "groq" as const, + apiKey: " groq-secret ", + model: "whisper-large-v3-turbo", + endpoint: "https://api.groq.com/openai/v1/audio/transcriptions", + }, + ])("sends an iPhone recording directly to $provider", async (config) => { const entries: Array<[string, unknown]> = []; vi.stubGlobal( "FormData", @@ -23,18 +33,18 @@ describe("transcribeMobileVoiceRecording", () => { const fetchMock = vi.fn().mockResolvedValue(Response.json({ text: " hello " })); await expect( - transcribeMobileVoiceRecording("file:///recording.m4a", " secret-key ", fetchMock), + transcribeMobileVoiceRecording("file:///recording.m4a", config, fetchMock), ).resolves.toBe("hello"); expect(fetchMock).toHaveBeenCalledWith( - "https://api.openai.com/v1/audio/transcriptions", + config.endpoint, expect.objectContaining({ method: "POST", - headers: { authorization: "Bearer secret-key" }, + headers: { authorization: `Bearer ${config.apiKey.trim()}` }, }), ); expect(entries).toEqual([ - ["model", MOBILE_TRANSCRIPTION_MODEL], + ["model", config.model], [ "file", { @@ -45,4 +55,22 @@ describe("transcribeMobileVoiceRecording", () => { ], ]); }); + + it("includes the provider error when the request fails", async () => { + const fetchMock = vi + .fn() + .mockResolvedValue(Response.json({ error: { message: "Unknown model" } }, { status: 400 })); + + await expect( + transcribeMobileVoiceRecording( + "file:///recording.m4a", + { + provider: "groq", + apiKey: "groq-secret", + model: "future-model", + }, + fetchMock, + ), + ).rejects.toThrow("Unknown model"); + }); }); diff --git a/apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.ts b/apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.ts index b2de78bca3b..525fcece13f 100644 --- a/apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.ts +++ b/apps/mobile/src/features/voice-dictation/mobileVoiceTranscription.ts @@ -1,13 +1,61 @@ -const OPENAI_TRANSCRIPTION_URL = "https://api.openai.com/v1/audio/transcriptions"; -export const MOBILE_TRANSCRIPTION_MODEL = "gpt-4o-mini-transcribe"; +import type { VoiceTranscriptionProvider } from "@t3tools/contracts"; + +export interface MobileVoiceTranscriptionConfig { + readonly provider: VoiceTranscriptionProvider; + readonly apiKey: string; + readonly model: string; +} + +export interface MobileVoiceTranscriptionProviderConfig { + readonly id: VoiceTranscriptionProvider; + readonly label: string; + readonly endpoint: string; + readonly defaultModel: string; + readonly modelOptions: ReadonlyArray<{ + readonly id: string; + readonly label: string; + }>; +} + +export const MOBILE_VOICE_TRANSCRIPTION_PROVIDERS: ReadonlyArray = + [ + { + id: "openai", + label: "OpenAI", + endpoint: "https://api.openai.com/v1/audio/transcriptions", + defaultModel: "gpt-4o-transcribe", + modelOptions: [ + { id: "gpt-4o-transcribe", label: "GPT-4o Transcribe" }, + { id: "gpt-4o-mini-transcribe", label: "GPT-4o mini Transcribe" }, + { id: "whisper-1", label: "Whisper" }, + ], + }, + { + id: "groq", + label: "Groq", + endpoint: "https://api.groq.com/openai/v1/audio/transcriptions", + defaultModel: "whisper-large-v3-turbo", + modelOptions: [ + { id: "whisper-large-v3-turbo", label: "Whisper Large V3 Turbo" }, + { id: "whisper-large-v3", label: "Whisper Large V3" }, + ], + }, + ]; + +export function mobileVoiceTranscriptionProviderConfig( + provider: VoiceTranscriptionProvider, +): MobileVoiceTranscriptionProviderConfig { + return MOBILE_VOICE_TRANSCRIPTION_PROVIDERS.find((candidate) => candidate.id === provider)!; +} export async function transcribeMobileVoiceRecording( uri: string, - apiKey: string, + config: MobileVoiceTranscriptionConfig, fetchFn: typeof globalThis.fetch = globalThis.fetch, ): Promise { + const provider = mobileVoiceTranscriptionProviderConfig(config.provider); const form = new FormData(); - form.append("model", MOBILE_TRANSCRIPTION_MODEL); + form.append("model", config.model.trim()); form.append("file", { uri, name: "recording.m4a", @@ -17,9 +65,9 @@ export async function transcribeMobileVoiceRecording( const controller = new AbortController(); const timeout = setTimeout(() => controller.abort(), 2 * 60 * 1_000); try { - const response = await fetchFn(OPENAI_TRANSCRIPTION_URL, { + const response = await fetchFn(provider.endpoint, { method: "POST", - headers: { authorization: `Bearer ${apiKey.trim()}` }, + headers: { authorization: `Bearer ${config.apiKey.trim()}` }, body: form, signal: controller.signal, }); @@ -31,11 +79,11 @@ export async function transcribeMobileVoiceRecording( throw new Error( typeof payload?.error?.message === "string" ? payload.error.message - : "OpenAI rejected the transcription request.", + : `${provider.label} rejected the transcription request.`, ); } if (typeof payload?.text !== "string") { - throw new Error("OpenAI returned an invalid transcription response."); + throw new Error(`${provider.label} returned an invalid transcription response.`); } return payload.text.trim(); } catch (cause) { diff --git a/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts b/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts index 96a7db915ff..1d47a4f034b 100644 --- a/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts +++ b/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts @@ -11,7 +11,10 @@ import { } from "expo-audio"; import { useCallback, useEffect, useRef, useState } from "react"; -import { transcribeMobileVoiceRecording } from "./mobileVoiceTranscription"; +import { + type MobileVoiceTranscriptionConfig, + transcribeMobileVoiceRecording, +} from "./mobileVoiceTranscription"; const LEVEL_COUNT = 36; const FLAT_LEVELS = Array(LEVEL_COUNT).fill(0); @@ -21,7 +24,9 @@ const MAX_RECORDING_MS = 5 * 60 * 1_000; export type MobileVoiceTranscriptionStatus = "idle" | "recording" | "transcribing"; export function useMobileVoiceTranscription(input: { + readonly provider: MobileVoiceTranscriptionConfig["provider"]; readonly apiKey: string; + readonly model: string; readonly onTranscriptInsert: (text: string) => void; readonly onTranscriptSend: (text: string) => void; }) { @@ -43,13 +48,21 @@ export function useMobileVoiceTranscription(input: { const transcriptionAttemptRef = useRef(0); const timeoutRef = useRef | null>(null); const mountedRef = useRef(true); - const apiKeyRef = useRef(input.apiKey); + const configRef = useRef({ + provider: input.provider, + apiKey: input.apiKey, + model: input.model, + }); const onTranscriptInsertRef = useRef(input.onTranscriptInsert); const onTranscriptSendRef = useRef(input.onTranscriptSend); const startRef = useRef<() => Promise>(async () => undefined); const stopRef = useRef<(action?: "insert" | "send") => Promise>(async () => undefined); statusRef.current = status; - apiKeyRef.current = input.apiKey; + configRef.current = { + provider: input.provider, + apiKey: input.apiKey, + model: input.model, + }; onTranscriptInsertRef.current = input.onTranscriptInsert; onTranscriptSendRef.current = input.onTranscriptSend; @@ -100,7 +113,7 @@ export function useMobileVoiceTranscription(input: { if (mountedRef.current) setStatus("transcribing"); statusRef.current = "transcribing"; transcriptionAttempt = ++transcriptionAttemptRef.current; - const text = await transcribeMobileVoiceRecording(uri, apiKeyRef.current); + const text = await transcribeMobileVoiceRecording(uri, configRef.current); if (!mountedRef.current || transcriptionAttempt !== transcriptionAttemptRef.current) { return; } @@ -183,8 +196,8 @@ export function useMobileVoiceTranscription(input: { return; } if (statusRef.current !== "idle") return; - if (!apiKeyRef.current.trim()) { - setError("Save an OpenAI API key in Settings first."); + if (!configRef.current.apiKey.trim()) { + setError("Save an API key in Settings first."); return; } diff --git a/apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.test.ts b/apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.test.ts new file mode 100644 index 00000000000..d53c512eea9 --- /dev/null +++ b/apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.test.ts @@ -0,0 +1,103 @@ +import * as SecureStore from "expo-secure-store"; +import { beforeEach, describe, expect, it, vi } from "vite-plus/test"; + +const secureStore = vi.hoisted(() => new Map()); + +vi.mock("expo-secure-store", () => ({ + getItemAsync: vi.fn((key: string) => Promise.resolve(secureStore.get(key) ?? null)), + setItemAsync: vi.fn((key: string, value: string) => { + secureStore.set(key, value); + return Promise.resolve(); + }), + deleteItemAsync: vi.fn((key: string) => { + secureStore.delete(key); + return Promise.resolve(); + }), +})); + +describe("mobile voice transcription settings", () => { + beforeEach(() => { + secureStore.clear(); + vi.clearAllMocks(); + vi.resetModules(); + }); + + it("migrates the existing OpenAI key and uses the current default model", async () => { + secureStore.set("t3code.voice-transcription.openai-api-key", " legacy-key "); + const settings = await import("./voiceTranscriptionSettings"); + + await settings.loadMobileVoiceTranscriptionSettings(); + + expect(settings.getMobileVoiceTranscriptionSettingsSnapshot()).toMatchObject({ + provider: "openai", + loaded: true, + providers: { + openai: { apiKey: "legacy-key", model: "gpt-4o-transcribe" }, + groq: { apiKey: "", model: "whisper-large-v3-turbo" }, + }, + }); + }); + + it("stores separate provider keys and a custom model", async () => { + const settings = await import("./voiceTranscriptionSettings"); + await settings.loadMobileVoiceTranscriptionSettings(); + + await settings.saveMobileVoiceTranscriptionSettings({ + provider: "groq", + providers: { + openai: { apiKey: "openai-key", model: "gpt-4o-transcribe" }, + groq: { apiKey: "groq-key", model: "future-groq-transcribe" }, + }, + }); + + expect( + settings.activeMobileVoiceTranscriptionConfig( + settings.getMobileVoiceTranscriptionSettingsSnapshot(), + ), + ).toEqual({ + provider: "groq", + apiKey: "groq-key", + model: "future-groq-transcribe", + }); + expect(SecureStore.setItemAsync).toHaveBeenCalledWith( + "t3code.voice-transcription.settings.v2", + JSON.stringify({ + provider: "groq", + providers: { + openai: { apiKey: "openai-key", model: "gpt-4o-transcribe" }, + groq: { apiKey: "groq-key", model: "future-groq-transcribe" }, + }, + }), + ); + }); + + it("keeps the setup editable when saved settings are invalid", async () => { + secureStore.set("t3code.voice-transcription.settings.v2", "not-json"); + const settings = await import("./voiceTranscriptionSettings"); + + await settings.loadMobileVoiceTranscriptionSettings(); + + expect(settings.getMobileVoiceTranscriptionSettingsSnapshot()).toMatchObject({ + provider: "openai", + loaded: true, + error: "Saved voice dictation settings were invalid. Save them again.", + }); + }); + + it("can retry after the secure store is temporarily unavailable", async () => { + vi.mocked(SecureStore.getItemAsync).mockRejectedValueOnce(new Error("temporarily unavailable")); + const settings = await import("./voiceTranscriptionSettings"); + + await settings.loadMobileVoiceTranscriptionSettings(); + expect(settings.getMobileVoiceTranscriptionSettingsSnapshot()).toMatchObject({ + loaded: false, + error: "Could not read the saved voice dictation settings.", + }); + + await settings.loadMobileVoiceTranscriptionSettings(); + expect(settings.getMobileVoiceTranscriptionSettingsSnapshot()).toMatchObject({ + loaded: true, + error: null, + }); + }); +}); diff --git a/apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.ts b/apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.ts index b720a5fabbe..15f63c78c0b 100644 --- a/apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.ts +++ b/apps/mobile/src/features/voice-dictation/voiceTranscriptionSettings.ts @@ -1,17 +1,48 @@ +import type { VoiceTranscriptionProvider } from "@t3tools/contracts"; import * as SecureStore from "expo-secure-store"; import { useEffect, useSyncExternalStore } from "react"; -const OPENAI_API_KEY_STORAGE_KEY = "t3code.voice-transcription.openai-api-key"; +import { + type MobileVoiceTranscriptionConfig, + mobileVoiceTranscriptionProviderConfig, +} from "./mobileVoiceTranscription"; -export interface MobileVoiceTranscriptionSettingsSnapshot { +const LEGACY_OPENAI_API_KEY_STORAGE_KEY = "t3code.voice-transcription.openai-api-key"; +const SETTINGS_STORAGE_KEY = "t3code.voice-transcription.settings.v2"; + +export interface MobileVoiceTranscriptionProviderSettings { readonly apiKey: string; + readonly model: string; +} + +export type MobileVoiceTranscriptionProviderSettingsMap = Readonly< + Record +>; + +export interface MobileVoiceTranscriptionSettingsSnapshot { + readonly provider: VoiceTranscriptionProvider; + readonly providers: MobileVoiceTranscriptionProviderSettingsMap; readonly error: string | null; readonly loaded: boolean; readonly saving: boolean; } +function defaultProviderSettings(): MobileVoiceTranscriptionProviderSettingsMap { + return { + openai: { + apiKey: "", + model: mobileVoiceTranscriptionProviderConfig("openai").defaultModel, + }, + groq: { + apiKey: "", + model: mobileVoiceTranscriptionProviderConfig("groq").defaultModel, + }, + }; +} + let snapshot: MobileVoiceTranscriptionSettingsSnapshot = { - apiKey: "", + provider: "openai", + providers: defaultProviderSettings(), error: null, loaded: false, saving: false, @@ -30,20 +61,85 @@ function subscribe(listener: () => void) { return () => listeners.delete(listener); } +function normalizeProvider(value: unknown): VoiceTranscriptionProvider { + return value === "groq" ? "groq" : "openai"; +} + +function normalizeProviderSettings( + value: unknown, + provider: VoiceTranscriptionProvider, +): MobileVoiceTranscriptionProviderSettings { + const candidate = + typeof value === "object" && value !== null + ? (value as { readonly apiKey?: unknown; readonly model?: unknown }) + : null; + const model = typeof candidate?.model === "string" ? candidate.model.trim() : ""; + return { + apiKey: typeof candidate?.apiKey === "string" ? candidate.apiKey.trim() : "", + model: model || mobileVoiceTranscriptionProviderConfig(provider).defaultModel, + }; +} + +function parseStoredSettings( + value: string, +): Pick { + const parsed = JSON.parse(value) as { + readonly provider?: unknown; + readonly providers?: { readonly openai?: unknown; readonly groq?: unknown }; + }; + return { + provider: normalizeProvider(parsed.provider), + providers: { + openai: normalizeProviderSettings(parsed.providers?.openai, "openai"), + groq: normalizeProviderSettings(parsed.providers?.groq, "groq"), + }, + }; +} + export function loadMobileVoiceTranscriptionSettings(): Promise { if (snapshot.loaded) return Promise.resolve(); if (loadPromise) return loadPromise; const loadRevision = revision; - loadPromise = SecureStore.getItemAsync(OPENAI_API_KEY_STORAGE_KEY) - .then((apiKey) => { + loadPromise = Promise.all([ + SecureStore.getItemAsync(SETTINGS_STORAGE_KEY), + SecureStore.getItemAsync(LEGACY_OPENAI_API_KEY_STORAGE_KEY), + ]) + .then(([storedSettings, legacyOpenAiApiKey]) => { if (revision !== loadRevision) return; - publish({ apiKey: apiKey ?? "", error: null, loaded: true, saving: false }); + if (storedSettings) { + try { + const parsed = parseStoredSettings(storedSettings); + publish({ ...parsed, error: null, loaded: true, saving: false }); + return; + } catch { + publish({ + provider: "openai", + providers: defaultProviderSettings(), + error: "Saved voice dictation settings were invalid. Save them again.", + loaded: true, + saving: false, + }); + return; + } + } + const providers = defaultProviderSettings(); + publish({ + provider: "openai", + providers: { + ...providers, + openai: { ...providers.openai, apiKey: legacyOpenAiApiKey?.trim() ?? "" }, + }, + error: null, + loaded: true, + saving: false, + }); }) .catch(() => { if (revision !== loadRevision) return; publish({ - apiKey: "", - error: "Could not read the saved API key.", + provider: "openai", + providers: defaultProviderSettings(), + error: "Could not read the saved voice dictation settings.", loaded: false, saving: false, }); @@ -54,28 +150,49 @@ export function loadMobileVoiceTranscriptionSettings(): Promise { return loadPromise; } -export async function saveMobileVoiceTranscriptionApiKey(apiKey: string): Promise { - const normalized = apiKey.trim(); +export async function saveMobileVoiceTranscriptionSettings(input: { + readonly provider: VoiceTranscriptionProvider; + readonly providers: MobileVoiceTranscriptionProviderSettingsMap; +}): Promise { + const providers: MobileVoiceTranscriptionProviderSettingsMap = { + openai: normalizeProviderSettings(input.providers.openai, "openai"), + groq: normalizeProviderSettings(input.providers.groq, "groq"), + }; revision += 1; + const previous = snapshot; publish({ ...snapshot, error: null, saving: true }); try { - if (normalized) { - await SecureStore.setItemAsync(OPENAI_API_KEY_STORAGE_KEY, normalized); - } else { - await SecureStore.deleteItemAsync(OPENAI_API_KEY_STORAGE_KEY); - } - publish({ apiKey: normalized, error: null, loaded: true, saving: false }); - } catch { + await SecureStore.setItemAsync( + SETTINGS_STORAGE_KEY, + JSON.stringify({ provider: input.provider, providers }), + ); + await SecureStore.deleteItemAsync(LEGACY_OPENAI_API_KEY_STORAGE_KEY).catch(() => undefined); publish({ - ...snapshot, - error: "Could not save the API key.", + provider: input.provider, + providers, + error: null, loaded: true, saving: false, }); - throw new Error("Could not save the API key."); + } catch { + publish({ ...previous, error: "Could not save voice dictation settings.", saving: false }); + throw new Error("Could not save voice dictation settings."); } } +export function activeMobileVoiceTranscriptionConfig( + settings: Pick, +): MobileVoiceTranscriptionConfig { + return { + provider: settings.provider, + ...settings.providers[settings.provider], + }; +} + +export function getMobileVoiceTranscriptionSettingsSnapshot(): MobileVoiceTranscriptionSettingsSnapshot { + return snapshot; +} + export function useMobileVoiceTranscriptionSettings() { const current = useSyncExternalStore( subscribe, diff --git a/docs/user/voice-dictation.md b/docs/user/voice-dictation.md index 6cf69ffb7d3..a30121acd1b 100644 --- a/docs/user/voice-dictation.md +++ b/docs/user/voice-dictation.md @@ -31,6 +31,10 @@ larger than 25 MB are rejected. ## iPhone and Android -Open **Settings** → **Voice Dictation** and save an OpenAI API key. The key is kept in the device's -secure store. Native recordings use `gpt-4o-mini-transcribe` and are sent directly from the mobile -app to OpenAI. Clearing the saved key removes the microphone from the composer. +Open **Settings** → **Voice Dictation**, choose **OpenAI** or **Groq**, select a model, and save the +provider's API key. Each provider keeps its own key and model in the device's secure store. OpenAI +defaults to `gpt-4o-transcribe`; Groq defaults to `whisper-large-v3-turbo`. You can also enter a +custom compatible model ID without waiting for an app update. + +Native recordings are sent directly from the mobile app to the selected provider. Clearing the +selected provider's saved key removes the microphone from the composer. From 63b51f50e2ed17d50ac0f5d99e1130046eb146e2 Mon Sep 17 00:00:00 2001 From: Yaroslav Kachur <268710642+KachurPro@users.noreply.github.com> Date: Fri, 14 Aug 2026 22:43:06 +0100 Subject: [PATCH 14/21] fix: harden voice dictation controls --- .../voice-dictation/useMobileVoiceTranscription.ts | 1 + apps/web/src/components/chat/ChatComposer.tsx | 7 ++++--- .../src/components/chat/VoiceTranscriptionPanel.tsx | 10 +++++----- 3 files changed, 10 insertions(+), 8 deletions(-) diff --git a/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts b/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts index 1d47a4f034b..674a9449cc6 100644 --- a/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts +++ b/apps/mobile/src/features/voice-dictation/useMobileVoiceTranscription.ts @@ -97,6 +97,7 @@ export function useMobileVoiceTranscription(input: { await recorder.stop(); const uri = recorder.uri; await resetAudioMode(); + if (!mountedRef.current) return; if ( terminalActionRef.current === "abort" || beforeStop.durationMillis < MIN_RECORDING_MS || diff --git a/apps/web/src/components/chat/ChatComposer.tsx b/apps/web/src/components/chat/ChatComposer.tsx index 2fb792a0cbb..378bf01c99b 100644 --- a/apps/web/src/components/chat/ChatComposer.tsx +++ b/apps/web/src/components/chat/ChatComposer.tsx @@ -3197,7 +3197,8 @@ export const ChatComposer = memo(function ChatComposer(props: ChatComposerProps) ) : null} {/* Bottom toolbar */} - {isComposerCollapsedMobile ? null : activePendingApproval ? ( + {isComposerCollapsedMobile && + voiceTranscription.status === "idle" ? null : activePendingApproval ? (
void startVoiceTranscription()} aria-label="Start dictation" > diff --git a/apps/web/src/components/chat/VoiceTranscriptionPanel.tsx b/apps/web/src/components/chat/VoiceTranscriptionPanel.tsx index 073ff1cfcd4..11a80e5a95f 100644 --- a/apps/web/src/components/chat/VoiceTranscriptionPanel.tsx +++ b/apps/web/src/components/chat/VoiceTranscriptionPanel.tsx @@ -43,8 +43,8 @@ export function VoiceTranscriptionPanel({