From 471ef5f4c302af55b2019e538df81e8a5d371f9f Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Sat, 19 Sep 2026 01:41:19 -0700 Subject: [PATCH] fix(desktop): direct dictation requests time out per stt.openai.timeout instead of hanging The client-direct STT path in the Desktop (voice-client-direct.ts) issued a bare fetch to the provider with no AbortSignal, so a slow or wedged transcription endpoint left the microphone stuck on "transcribing" forever; the gateway's own transcription client already has a deadline. - tools/voice_client_config.py::_resolve_stt_client_config adds `timeout_s` (from `stt.openai.timeout`, default 60; groq/deepinfra riders inherit) to the direct STT config the gateway hands the Desktop. - voice-client-direct.ts routes all three direct STT fetches (openai-multipart, xai-stt, elevenlabs-stt) through sttFetch, which aborts at that deadline and surfaces "Transcription timed out after Ns". - Docs: desktop.md dictation paragraph notes the shared budget. Part of #112939 --- .../src/lib/voice-client-direct.test.ts | 58 +++++++++++++++++++ apps/desktop/src/lib/voice-client-direct.ts | 40 ++++++++++++- tests/tools/test_voice_client_config.py | 10 ++++ tools/voice_client_config.py | 6 +- website/docs/user-guide/desktop.md | 2 +- 5 files changed, 111 insertions(+), 5 deletions(-) diff --git a/apps/desktop/src/lib/voice-client-direct.test.ts b/apps/desktop/src/lib/voice-client-direct.test.ts index 435fed9029..a3c69b80f7 100644 --- a/apps/desktop/src/lib/voice-client-direct.test.ts +++ b/apps/desktop/src/lib/voice-client-direct.test.ts @@ -207,6 +207,64 @@ describe('transcribeAudioClientDirect', () => { expect((init.headers as Record)['xi-api-key']).toBe('gsk_test') expect((init.body as FormData).get('model_id')).toBe('scribe_v2') }) + + /** A fetch that only settles when its AbortSignal fires — a wedged STT endpoint. */ + function hangingFetch() { + return vi.fn( + (_url: string, init?: RequestInit) => + new Promise((_resolve, reject) => { + init?.signal?.addEventListener('abort', () => reject(new DOMException('aborted', 'AbortError'))) + }) + ) + } + + it('aborts a hanging transcription at the default 60 s instead of transcribing forever', async () => { + vi.useFakeTimers() + + try { + mockDesktopApi({ ok: true, stt: directStt, tts: relay }) + const fetchMock = hangingFetch() + vi.stubGlobal('fetch', fetchMock) + + const pending = transcribeAudioClientDirect(new Blob(['x'], { type: 'audio/webm' })) + const settled = vi.fn() + + pending.then(settled, settled) + await vi.advanceTimersByTimeAsync(0) + + const [, init] = fetchMock.mock.calls[0] as unknown as [string, RequestInit] + expect(init.signal).toBeInstanceOf(AbortSignal) + + await vi.advanceTimersByTimeAsync(59_000) + expect(settled).not.toHaveBeenCalled() + + await vi.advanceTimersByTimeAsync(1_000) + await expect(pending).rejects.toThrow(/Transcription timed out after 60s/) + } finally { + vi.useRealTimers() + } + }) + + it('honours the gateway-resolved stt.openai.timeout for the direct request', async () => { + vi.useFakeTimers() + + try { + mockDesktopApi({ ok: true, stt: { ...directStt, timeout_s: 5 }, tts: relay }) + vi.stubGlobal('fetch', hangingFetch()) + + const pending = transcribeAudioClientDirect(new Blob(['x'], { type: 'audio/webm' })) + const settled = vi.fn() + + pending.then(settled, settled) + await vi.advanceTimersByTimeAsync(4_900) + expect(settled).not.toHaveBeenCalled() + + await vi.advanceTimersByTimeAsync(200) + await expect(pending).rejects.toThrow(/Transcription timed out after 5s/) + } finally { + vi.useRealTimers() + } + }) }) describe('synthesizeSpeechClientDirect', () => { diff --git a/apps/desktop/src/lib/voice-client-direct.ts b/apps/desktop/src/lib/voice-client-direct.ts index 828c9fe690..dc456714cf 100644 --- a/apps/desktop/src/lib/voice-client-direct.ts +++ b/apps/desktop/src/lib/voice-client-direct.ts @@ -27,6 +27,8 @@ export interface DirectSttConfig { api_key: string model: null | string language: null | string + /** Seconds the gateway allows one transcription request (`stt.openai.timeout`); absent on older backends. */ + timeout_s?: null | number } export interface DirectTtsConfig { @@ -175,6 +177,38 @@ export function transcriptFromOpenAiMultipartBody(body: string): string { return trimmed } +const DEFAULT_STT_TIMEOUT_S = 60 + +/** Same budget the gateway's own transcription client uses (`stt.openai.timeout`, default 60 s). */ +export function sttTimeoutSeconds(stt: Pick): number { + const value = Number(stt.timeout_s) + + return Number.isFinite(value) && value > 0 ? value : DEFAULT_STT_TIMEOUT_S +} + +/** + * `fetch` with the STT deadline. A slow or wedged endpoint otherwise keeps the + * dictation UI in "transcribing" forever — the browser applies no timeout of + * its own to a POST that never answers. + */ +async function sttFetch(stt: DirectSttConfig, url: string, init: RequestInit): Promise { + const seconds = sttTimeoutSeconds(stt) + const controller = new AbortController() + const timer = setTimeout(() => controller.abort(), seconds * 1000) + + try { + return await fetch(url, { ...init, signal: controller.signal }) + } catch (error) { + if (controller.signal.aborted) { + throw new Error(`Transcription timed out after ${seconds}s (${stt.provider} did not answer)`) + } + + throw error + } finally { + clearTimeout(timer) + } +} + /** * Transcribe provider-direct. Returns the transcript ('' = silence), or null * when the profile's provider isn't client-callable — the caller relays. @@ -204,7 +238,7 @@ export async function transcribeAudioClientDirect(audio: Blob): Promise Dict[str, Any]: language = tt._resolve_stt_language( provider, stt_config, extra_keys=("language_code",) if provider == "elevenlabs" else ()) section = _section(stt_config, provider) + # Same deadline the gateway's own transcription client applies + # (``stt.openai.timeout``; riders such as groq/deepinfra inherit it), so a + # slow endpoint fails the Desktop's direct request instead of hanging it. + timeout_s = tc._config_number(_section(stt_config, "openai"), "timeout", 60.0) def direct(wire: str, base_url: Any, api_key: str, model: Any) -> Dict[str, Any]: - return _direct(wire, provider, base_url, api_key, model, language=language) + return _direct(wire, provider, base_url, api_key, model, language=language, timeout_s=timeout_s) def env_base_url(env_var: str, default: str) -> str: from hermes_cli.config import get_env_value diff --git a/website/docs/user-guide/desktop.md b/website/docs/user-guide/desktop.md index 52cd692549..b031ce0a11 100644 --- a/website/docs/user-guide/desktop.md +++ b/website/docs/user-guide/desktop.md @@ -101,7 +101,7 @@ With **Group by → Projects**, each project row previews its three most recent The model picker lives in the **composer**, just left of the microphone. Click it to switch the model; hover a model row for its options (thinking, effort, fast). Next to it, a **reasoning pill** shows the active model's effort level (`Med`, `High`, …) and opens the same options directly, so you can change effort without finding the model's row. The pill is hidden for models whose catalog reports no reasoning control. When the gateway flags a switch as risky (a large cached context, an expensive model, a data-training tier), the app asks first in a dialog: **Switch anyway** applies it, **Keep current model** (or Esc) leaves everything as it was. -The **microphone** is dictation; hover it and the other voice toggles fan out above it — **Read replies aloud** and the **wake word** ear. A toggle that is on shows as a solid disc. Starting a full voice conversation stays on the primary button to the right. In the HUD and in narrow tiles the same controls fold into one menu behind the mic instead. +The **microphone** is dictation; hover it and the other voice toggles fan out above it — **Read replies aloud** and the **wake word** ear. A toggle that is on shows as a solid disc. Starting a full voice conversation stays on the primary button to the right. In the HUD and in narrow tiles the same controls fold into one menu behind the mic instead. When dictation talks to the speech-to-text provider directly (client-direct voice), the request honours the same `stt.openai.timeout` budget (default 60 s) as the gateway's own transcription client, so a slow endpoint fails with "Transcription timed out" instead of leaving the mic stuck on transcribing. - **The composer picker is sticky UI state and never touches your default.** It's remembered locally (per device) and **follows** across new chats and restarts instead of snapping back to the default — pick a model once and the next `Cmd/Ctrl+N` opens on it. With a live chat, switching models scopes the change to that **current chat**; either way the selection rides along when the session is created/switched and is **never** written to the profile default — with one exception: on a fresh profile that has no `model.default`/`model.provider` configured yet, the first pick is persisted so the app has a real default instead of falling through to a stray API-key env var on restart. Persistence follows the same rule as `/model` (`model.persist_switch_by_default`); use **Settings → Model** to change the default deliberately. (Switching [profiles](#sessions--profiles) reseeds to that profile's own default.) - **Set the default in Settings → Model.** That "main" model is your **per-profile global default** — it's what new chats, crons, subagents, and auxiliary tasks start from, and it's the only place that writes it. Each [profile](#sessions--profiles) keeps its own default.