fix(desktop): unify boot-class getConnection() budgets on one shared 45s constant

Follow-up to the #95039 salvage: the cherry-picked bound used the 20s
RECONNECT_ATTEMPT_TIMEOUT_MS on boot()/softSwitch() getConnection(), but a
reviewer note (and the Phase A registry-restore work) established that
boot-class awaits must ride out a full backend cold spawn — main's spawn
budget is 45s (DEFAULT_BACKEND_READY_TIMEOUT_MS). A 20s renderer bound would
latch boot errors on healthy-but-slow cold boots.

Introduce BACKEND_BOOT_WAIT_TIMEOUT_MS (45s) in lib/with-timeout.ts as the
single shared boot-class budget, point boot()/softSwitch() getConnection()
and connections.ts BOOT_DESCRIPTOR_WAIT_TIMEOUT_MS at it, and keep the 20s
reconnect budget only for reconnect-class awaits against an already-spawned
backend. No magic-number drift: 45_000 now appears once in renderer code.
This commit is contained in:
Teknium
2026-08-26 03:49:19 -07:00
parent 31f3de1f06
commit 574bd7175c
4 changed files with 27 additions and 12 deletions

View File

@@ -1153,11 +1153,11 @@ describe('useGatewayBoot remote reconnect loop (real hook, fake socket)', () =>
expect($desktopBoot.get().error).toBeNull()
// Advance past the internal reconnect-attempt timeout (20s) — the
// Advance past the shared backend-boot budget (45s) — the
// stalled await must reject on its own so boot()'s catch runs instead of
// waiting indefinitely on main.
await act(async () => {
await vi.advanceTimersByTimeAsync(20_000)
await vi.advanceTimersByTimeAsync(45_000)
})
expect($desktopBoot.get().error).toBeTruthy()
@@ -1191,11 +1191,11 @@ describe('useGatewayBoot remote reconnect loop (real hook, fake socket)', () =>
expect($gatewaySwitching.get()).toBe(true)
// Advance past the internal reconnect-attempt timeout (20s) — the
// Advance past the shared backend-boot budget (45s) — the
// stalled await must reject so the `finally` clears $gatewaySwitching
// instead of latching the switch UI frozen forever.
await act(async () => {
await vi.advanceTimersByTimeAsync(20_000)
await vi.advanceTimersByTimeAsync(45_000)
})
expect($gatewaySwitching.get()).toBe(false)

View File

@@ -7,7 +7,7 @@ import { HermesGateway } from '@/hermes'
import { translateNow } from '@/i18n'
import { desktopDefaultCwd } from '@/lib/desktop-fs'
import { reconnectBackoffDelayMs } from '@/lib/reconnect-backoff'
import { withTimeout } from '@/lib/with-timeout'
import { withTimeout, BACKEND_BOOT_WAIT_TIMEOUT_MS } from '@/lib/with-timeout'
import {
$desktopBoot,
applyDesktopBootProgress,
@@ -542,10 +542,12 @@ export function useGatewayBoot({
// on its pinned profile's backend across a soft switch.
// Bounded for the same reason as attemptReconnect() (#93454): a wedged
// main-process round-trip must not latch $gatewaySwitching stuck —
// the `finally` below only runs once this promise settles.
// the `finally` below only runs once this promise settles. Uses the
// shared backend-boot budget rather than the reconnect budget because
// ensureBackend may cold-spawn a pooled helper backend here.
const conn = await withTimeout(
desktop.getConnection(windowProfileOverride() ?? undefined),
RECONNECT_ATTEMPT_TIMEOUT_MS,
BACKEND_BOOT_WAIT_TIMEOUT_MS,
'Timed out reconnecting to Hermes backend'
)
@@ -895,10 +897,12 @@ export function useGatewayBoot({
// backend directly — ensureBackend spawns/reuses it from the pool.
// Everything else keeps dialing the primary.
// Bounded like the reconnect path (#93454): a wedged main-process
// round-trip must not hang "Starting Hermes…" forever.
// round-trip must not hang "Starting Hermes…" forever. Initial boot
// rides out a full backend cold spawn, so it gets the shared 45s
// backend-boot budget, not the 20s reconnect budget.
const conn = await withTimeout(
desktop.getConnection(windowProfileOverride() ?? undefined),
RECONNECT_ATTEMPT_TIMEOUT_MS,
BACKEND_BOOT_WAIT_TIMEOUT_MS,
'Timed out connecting to Hermes backend'
)

View File

@@ -1,3 +1,13 @@
/** Shared budget for any renderer await that rides out a primary backend
* cold boot (initial getConnection(), the registry restore's descriptor
* wait). Matches the main-process spawn budget
* (DEFAULT_BACKEND_READY_TIMEOUT_MS in electron/backend-health.ts): a
* healthy cold boot publishes well within this; anything longer means the
* backend is not coming and the caller should fail instead of hanging.
* Reconnect-class awaits against an already-spawned backend use the shorter
* RECONNECT_ATTEMPT_TIMEOUT_MS (use-gateway-boot.ts) instead. */
export const BACKEND_BOOT_WAIT_TIMEOUT_MS = 45_000
/** Rejection raised by withTimeout. The bounded work is NOT cancelled — the
* caller decides what a straggler that settles later means. */
export class TimeoutError extends Error {

View File

@@ -2,7 +2,7 @@ import { atom, computed } from 'nanostores'
import type { DesktopConnectionsRegistry } from '@/global'
import { persistStringRecord, storedStringRecord } from '@/lib/storage'
import { isTimeoutError, withTimeout } from '@/lib/with-timeout'
import { isTimeoutError, withTimeout, BACKEND_BOOT_WAIT_TIMEOUT_MS } from '@/lib/with-timeout'
import { $connectionsRegistry } from '@/store/connection-registry-state'
import {
beginGatewaySwitch,
@@ -34,8 +34,9 @@ const SWITCH_COMMIT_TIMEOUT_MS = 20_000
const SWITCH_REMEMBER_TIMEOUT_MS = 5_000
// Matches the primary spawn budget: a healthy cold boot publishes well within
// this; anything longer means the primary is not coming and the registry
// restore should stop waiting for it.
const BOOT_DESCRIPTOR_WAIT_TIMEOUT_MS = 45_000
// restore should stop waiting for it. Shared constant so the boot-class
// budgets can't drift apart (see with-timeout.ts).
const BOOT_DESCRIPTOR_WAIT_TIMEOUT_MS = BACKEND_BOOT_WAIT_TIMEOUT_MS
export { $connectionsRegistry } from '@/store/connection-registry-state'