fix(agent): keep local Responses endpoints out of hosted codex watchdog clamps

#86444 widened the large-context stale floor, the 1500s hard ceiling and the
TTFB scale-up/cap gate from OpenAI-Codex to every codex_responses route. That
made the #92302 local-endpoint TTFB branch unreachable (local TTFB fell back to
120s instead of agent.local_stream_stale_timeout) and silently clamped a local
server's configured stale timeout to 1500s. Evaluate is_local_endpoint once and
exclude local endpoints from the hosted clamps; xAI keeps the #86444 behaviour.

Note: xAI large requests now get the raised stale floor but keep first-event
idle semantics (progress gating stays OpenAI-Codex only).
This commit is contained in:
kshitijk4poor
2026-09-24 17:32:37 +05:30
committed by kshitij
parent d6d77986f2
commit 6c94f52ef1
2 changed files with 25 additions and 3 deletions

View File

@@ -1165,7 +1165,11 @@ def _resolve_nonstream_watchdogs(agent, api_kwargs: dict) -> _NonStreamWatchdogs
est_tokens = estimate_request_context_tokens(api_kwargs)
effort_floor = _high_effort_silence_floor(agent) if codex else 0.0
codex_floor = 0.0
if codex:
# Local Responses servers keep their configured local stale/TTFB grace: the hosted
# large-context floor, hard ceiling and TTFB scale-up/cap below must not tighten it.
base_url = getattr(agent, "base_url", None)
local = bool(base_url) and is_local_endpoint(base_url)
if codex and not local:
# Raise the stale floor for large payloads so healthy gateway-scale
# requests aren't aborted mid-prefill.
codex_floor = openai_codex_stale_timeout_floor(est_tokens)
@@ -1188,7 +1192,7 @@ def _resolve_nonstream_watchdogs(agent, api_kwargs: dict) -> _NonStreamWatchdogs
ttfb_timeout = env_float("HERMES_CODEX_TTFB_TIMEOUT_SECONDS", 120.0)
if ttfb_timeout <= 0:
ttfb_enabled = False
elif codex:
elif codex and not local:
# Large requests legitimately spend tens of seconds in admission/prefill before the
# first SSE event: scale the cutoff up to the idle default unless TTFB_STRICT is set.
disable_above = env_float("HERMES_CODEX_TTFB_DISABLE_ABOVE_TOKENS", 10_000.0)
@@ -1206,7 +1210,7 @@ def _resolve_nonstream_watchdogs(agent, api_kwargs: dict) -> _NonStreamWatchdogs
"(context=~%s tokens) per HERMES_CODEX_TTFB_MAX_SECONDS.", ttfb_timeout, ttfb_cap,
f"{est_tokens:,}")
ttfb_timeout = ttfb_cap
elif not ttfb_explicit and (base_url := getattr(agent, "base_url", None)) and is_local_endpoint(base_url):
elif not ttfb_explicit and local:
# A local server prefills for minutes before its first event; the chat-completions
# siblings already grant local endpoints the local stale ceiling, so the Responses
# transport gets the same grace instead of the 120s hosted cutoff (#92302).

View File

@@ -5,6 +5,8 @@ Ensures:
the large-context stale timeout floor (e.g. 1200s at >100K tokens, 900s at >50K tokens)
in interruptible_api_call. A baseline short timeout of 0.2s is elevated so a 0.5s call succeeds without stale_call_kill.
2. Large requests scale TTFB timeout for xai-oauth (codex_responses) instead of killing at a short TTFB cutoff.
3. The 1500s codex hard ceiling clamps hosted codex_responses (xAI) but does not tighten a local
Responses endpoint's (e.g. 127.0.0.1) configured stale timeout.
"""
from __future__ import annotations
@@ -134,3 +136,19 @@ def test_xai_responses_ttfb_scaled_for_large_requests(monkeypatch):
result = h.interruptible_api_call(agent, {"messages": [{"role": "user", "content": large_text}], "stream": True})
assert result.ok is True
assert "codex_ttfb_kill" not in closes
def test_hard_ceiling_clamps_hosted_but_not_local_responses_endpoints(monkeypatch):
"""The 1500s hard ceiling bounds hosted codex_responses (xAI) but must not tighten a
local Responses server's configured stale timeout."""
from agent import chat_completion_helpers as h
monkeypatch.delenv("HERMES_CODEX_HARD_TIMEOUT_SECONDS", raising=False)
kwargs = {"input": [{"role": "user", "content": "hi"}]}
hosted = _make_mock_agent()
local = _make_mock_agent(provider="custom", base_url="http://127.0.0.1:1234/v1")
for agent in (hosted, local):
agent._compute_non_stream_stale_timeout = lambda api_kwargs: 3000.0
assert h._resolve_nonstream_watchdogs(hosted, kwargs).stale_timeout == 1500.0
assert h._resolve_nonstream_watchdogs(local, kwargs).stale_timeout == 3000.0