fix(desktop): direct dictation requests time out per stt.openai.timeout instead of hanging
The client-direct STT path in the Desktop (voice-client-direct.ts) issued a bare fetch to the provider with no AbortSignal, so a slow or wedged transcription endpoint left the microphone stuck on "transcribing" forever; the gateway's own transcription client already has a deadline. - tools/voice_client_config.py::_resolve_stt_client_config adds `timeout_s` (from `stt.openai.timeout`, default 60; groq/deepinfra riders inherit) to the direct STT config the gateway hands the Desktop. - voice-client-direct.ts routes all three direct STT fetches (openai-multipart, xai-stt, elevenlabs-stt) through sttFetch, which aborts at that deadline and surfaces "Transcription timed out after Ns". - Docs: desktop.md dictation paragraph notes the shared budget. Part of #112939
This commit is contained in:
@@ -207,6 +207,64 @@ describe('transcribeAudioClientDirect', () => {
|
||||
expect((init.headers as Record<string, string>)['xi-api-key']).toBe('gsk_test')
|
||||
expect((init.body as FormData).get('model_id')).toBe('scribe_v2')
|
||||
})
|
||||
|
||||
/** A fetch that only settles when its AbortSignal fires — a wedged STT endpoint. */
|
||||
function hangingFetch() {
|
||||
return vi.fn(
|
||||
(_url: string, init?: RequestInit) =>
|
||||
new Promise<Response>((_resolve, reject) => {
|
||||
init?.signal?.addEventListener('abort', () => reject(new DOMException('aborted', 'AbortError')))
|
||||
})
|
||||
)
|
||||
}
|
||||
|
||||
it('aborts a hanging transcription at the default 60 s instead of transcribing forever', async () => {
|
||||
vi.useFakeTimers()
|
||||
|
||||
try {
|
||||
mockDesktopApi({ ok: true, stt: directStt, tts: relay })
|
||||
const fetchMock = hangingFetch()
|
||||
vi.stubGlobal('fetch', fetchMock)
|
||||
|
||||
const pending = transcribeAudioClientDirect(new Blob(['x'], { type: 'audio/webm' }))
|
||||
const settled = vi.fn()
|
||||
|
||||
pending.then(settled, settled)
|
||||
await vi.advanceTimersByTimeAsync(0)
|
||||
|
||||
const [, init] = fetchMock.mock.calls[0] as unknown as [string, RequestInit]
|
||||
expect(init.signal).toBeInstanceOf(AbortSignal)
|
||||
|
||||
await vi.advanceTimersByTimeAsync(59_000)
|
||||
expect(settled).not.toHaveBeenCalled()
|
||||
|
||||
await vi.advanceTimersByTimeAsync(1_000)
|
||||
await expect(pending).rejects.toThrow(/Transcription timed out after 60s/)
|
||||
} finally {
|
||||
vi.useRealTimers()
|
||||
}
|
||||
})
|
||||
|
||||
it('honours the gateway-resolved stt.openai.timeout for the direct request', async () => {
|
||||
vi.useFakeTimers()
|
||||
|
||||
try {
|
||||
mockDesktopApi({ ok: true, stt: { ...directStt, timeout_s: 5 }, tts: relay })
|
||||
vi.stubGlobal('fetch', hangingFetch())
|
||||
|
||||
const pending = transcribeAudioClientDirect(new Blob(['x'], { type: 'audio/webm' }))
|
||||
const settled = vi.fn()
|
||||
|
||||
pending.then(settled, settled)
|
||||
await vi.advanceTimersByTimeAsync(4_900)
|
||||
expect(settled).not.toHaveBeenCalled()
|
||||
|
||||
await vi.advanceTimersByTimeAsync(200)
|
||||
await expect(pending).rejects.toThrow(/Transcription timed out after 5s/)
|
||||
} finally {
|
||||
vi.useRealTimers()
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
describe('synthesizeSpeechClientDirect', () => {
|
||||
|
||||
@@ -27,6 +27,8 @@ export interface DirectSttConfig {
|
||||
api_key: string
|
||||
model: null | string
|
||||
language: null | string
|
||||
/** Seconds the gateway allows one transcription request (`stt.openai.timeout`); absent on older backends. */
|
||||
timeout_s?: null | number
|
||||
}
|
||||
|
||||
export interface DirectTtsConfig {
|
||||
@@ -175,6 +177,38 @@ export function transcriptFromOpenAiMultipartBody(body: string): string {
|
||||
return trimmed
|
||||
}
|
||||
|
||||
const DEFAULT_STT_TIMEOUT_S = 60
|
||||
|
||||
/** Same budget the gateway's own transcription client uses (`stt.openai.timeout`, default 60 s). */
|
||||
export function sttTimeoutSeconds(stt: Pick<DirectSttConfig, 'timeout_s'>): number {
|
||||
const value = Number(stt.timeout_s)
|
||||
|
||||
return Number.isFinite(value) && value > 0 ? value : DEFAULT_STT_TIMEOUT_S
|
||||
}
|
||||
|
||||
/**
|
||||
* `fetch` with the STT deadline. A slow or wedged endpoint otherwise keeps the
|
||||
* dictation UI in "transcribing" forever — the browser applies no timeout of
|
||||
* its own to a POST that never answers.
|
||||
*/
|
||||
async function sttFetch(stt: DirectSttConfig, url: string, init: RequestInit): Promise<Response> {
|
||||
const seconds = sttTimeoutSeconds(stt)
|
||||
const controller = new AbortController()
|
||||
const timer = setTimeout(() => controller.abort(), seconds * 1000)
|
||||
|
||||
try {
|
||||
return await fetch(url, { ...init, signal: controller.signal })
|
||||
} catch (error) {
|
||||
if (controller.signal.aborted) {
|
||||
throw new Error(`Transcription timed out after ${seconds}s (${stt.provider} did not answer)`)
|
||||
}
|
||||
|
||||
throw error
|
||||
} finally {
|
||||
clearTimeout(timer)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Transcribe provider-direct. Returns the transcript ('' = silence), or null
|
||||
* when the profile's provider isn't client-callable — the caller relays.
|
||||
@@ -204,7 +238,7 @@ export async function transcribeAudioClientDirect(audio: Blob): Promise<null | s
|
||||
form.set('language', stt.language)
|
||||
}
|
||||
|
||||
const response = await fetch(`${stt.base_url.replace(/\/+$/, '')}/audio/transcriptions`, {
|
||||
const response = await sttFetch(stt, `${stt.base_url.replace(/\/+$/, '')}/audio/transcriptions`, {
|
||||
method: 'POST',
|
||||
headers: { Authorization: `Bearer ${stt.api_key}` },
|
||||
body: form,
|
||||
@@ -227,7 +261,7 @@ export async function transcribeAudioClientDirect(audio: Blob): Promise<null | s
|
||||
form.set('language', stt.language)
|
||||
}
|
||||
|
||||
const response = await fetch(`${stt.base_url.replace(/\/+$/, '')}/stt`, {
|
||||
const response = await sttFetch(stt, `${stt.base_url.replace(/\/+$/, '')}/stt`, {
|
||||
method: 'POST',
|
||||
headers: { Authorization: `Bearer ${stt.api_key}` },
|
||||
body: form,
|
||||
@@ -255,7 +289,7 @@ export async function transcribeAudioClientDirect(audio: Blob): Promise<null | s
|
||||
form.set('language_code', stt.language)
|
||||
}
|
||||
|
||||
const response = await fetch(`${stt.base_url.replace(/\/+$/, '')}/speech-to-text`, {
|
||||
const response = await sttFetch(stt, `${stt.base_url.replace(/\/+$/, '')}/speech-to-text`, {
|
||||
method: 'POST',
|
||||
headers: { 'xi-api-key': stt.api_key },
|
||||
body: form,
|
||||
|
||||
@@ -189,6 +189,16 @@ def test_xai_env_key_goes_direct(voice_home, monkeypatch):
|
||||
assert stt["api_key"] == "xai_key1"
|
||||
|
||||
|
||||
def test_direct_stt_carries_the_gateway_transcription_timeout(voice_home, monkeypatch):
|
||||
"""The Desktop's direct request must honour ``stt.openai.timeout`` (default 60 s) like the gateway."""
|
||||
voice_home({"stt": {"provider": "xai"}})
|
||||
monkeypatch.setenv("XAI_API_KEY", "xai_key1")
|
||||
assert _resolve()["stt"]["timeout_s"] == 60
|
||||
|
||||
voice_home({"stt": {"provider": "xai", "openai": {"timeout": "5"}}})
|
||||
assert _resolve()["stt"]["timeout_s"] == 5
|
||||
|
||||
|
||||
def test_resolution_never_raises(voice_home, monkeypatch):
|
||||
"""A broken config section degrades to relay, never a 500."""
|
||||
voice_home({"stt": "not-a-dict", "tts": ["also", "wrong"]})
|
||||
|
||||
@@ -95,9 +95,13 @@ def _resolve_stt_client_config() -> Dict[str, Any]:
|
||||
language = tt._resolve_stt_language(
|
||||
provider, stt_config, extra_keys=("language_code",) if provider == "elevenlabs" else ())
|
||||
section = _section(stt_config, provider)
|
||||
# Same deadline the gateway's own transcription client applies
|
||||
# (``stt.openai.timeout``; riders such as groq/deepinfra inherit it), so a
|
||||
# slow endpoint fails the Desktop's direct request instead of hanging it.
|
||||
timeout_s = tc._config_number(_section(stt_config, "openai"), "timeout", 60.0)
|
||||
|
||||
def direct(wire: str, base_url: Any, api_key: str, model: Any) -> Dict[str, Any]:
|
||||
return _direct(wire, provider, base_url, api_key, model, language=language)
|
||||
return _direct(wire, provider, base_url, api_key, model, language=language, timeout_s=timeout_s)
|
||||
|
||||
def env_base_url(env_var: str, default: str) -> str:
|
||||
from hermes_cli.config import get_env_value
|
||||
|
||||
@@ -101,7 +101,7 @@ With **Group by → Projects**, each project row previews its three most recent
|
||||
|
||||
The model picker lives in the **composer**, just left of the microphone. Click it to switch the model; hover a model row for its options (thinking, effort, fast). Next to it, a **reasoning pill** shows the active model's effort level (`Med`, `High`, …) and opens the same options directly, so you can change effort without finding the model's row. The pill is hidden for models whose catalog reports no reasoning control. When the gateway flags a switch as risky (a large cached context, an expensive model, a data-training tier), the app asks first in a dialog: **Switch anyway** applies it, **Keep current model** (or Esc) leaves everything as it was.
|
||||
|
||||
The **microphone** is dictation; hover it and the other voice toggles fan out above it — **Read replies aloud** and the **wake word** ear. A toggle that is on shows as a solid disc. Starting a full voice conversation stays on the primary button to the right. In the HUD and in narrow tiles the same controls fold into one menu behind the mic instead.
|
||||
The **microphone** is dictation; hover it and the other voice toggles fan out above it — **Read replies aloud** and the **wake word** ear. A toggle that is on shows as a solid disc. Starting a full voice conversation stays on the primary button to the right. In the HUD and in narrow tiles the same controls fold into one menu behind the mic instead. When dictation talks to the speech-to-text provider directly (client-direct voice), the request honours the same `stt.openai.timeout` budget (default 60 s) as the gateway's own transcription client, so a slow endpoint fails with "Transcription timed out" instead of leaving the mic stuck on transcribing.
|
||||
|
||||
- **The composer picker is sticky UI state and never touches your default.** It's remembered locally (per device) and **follows** across new chats and restarts instead of snapping back to the default — pick a model once and the next `Cmd/Ctrl+N` opens on it. With a live chat, switching models scopes the change to that **current chat**; either way the selection rides along when the session is created/switched and is **never** written to the profile default — with one exception: on a fresh profile that has no `model.default`/`model.provider` configured yet, the first pick is persisted so the app has a real default instead of falling through to a stray API-key env var on restart. Persistence follows the same rule as `/model` (`model.persist_switch_by_default`); use **Settings → Model** to change the default deliberately. (Switching [profiles](#sessions--profiles) reseeds to that profile's own default.)
|
||||
- **Set the default in Settings → Model.** That "main" model is your **per-profile global default** — it's what new chats, crons, subagents, and auxiliary tasks start from, and it's the only place that writes it. Each [profile](#sessions--profiles) keeps its own default.
|
||||
|
||||
Reference in New Issue
Block a user