fix(agent): keep local Responses endpoints out of hosted codex watchdog clamps
#86444 widened the large-context stale floor, the 1500s hard ceiling and the TTFB scale-up/cap gate from OpenAI-Codex to every codex_responses route. That made the #92302 local-endpoint TTFB branch unreachable (local TTFB fell back to 120s instead of agent.local_stream_stale_timeout) and silently clamped a local server's configured stale timeout to 1500s. Evaluate is_local_endpoint once and exclude local endpoints from the hosted clamps; xAI keeps the #86444 behaviour. Note: xAI large requests now get the raised stale floor but keep first-event idle semantics (progress gating stays OpenAI-Codex only).
This commit is contained in:
@@ -1165,7 +1165,11 @@ def _resolve_nonstream_watchdogs(agent, api_kwargs: dict) -> _NonStreamWatchdogs
|
||||
est_tokens = estimate_request_context_tokens(api_kwargs)
|
||||
effort_floor = _high_effort_silence_floor(agent) if codex else 0.0
|
||||
codex_floor = 0.0
|
||||
if codex:
|
||||
# Local Responses servers keep their configured local stale/TTFB grace: the hosted
|
||||
# large-context floor, hard ceiling and TTFB scale-up/cap below must not tighten it.
|
||||
base_url = getattr(agent, "base_url", None)
|
||||
local = bool(base_url) and is_local_endpoint(base_url)
|
||||
if codex and not local:
|
||||
# Raise the stale floor for large payloads so healthy gateway-scale
|
||||
# requests aren't aborted mid-prefill.
|
||||
codex_floor = openai_codex_stale_timeout_floor(est_tokens)
|
||||
@@ -1188,7 +1192,7 @@ def _resolve_nonstream_watchdogs(agent, api_kwargs: dict) -> _NonStreamWatchdogs
|
||||
ttfb_timeout = env_float("HERMES_CODEX_TTFB_TIMEOUT_SECONDS", 120.0)
|
||||
if ttfb_timeout <= 0:
|
||||
ttfb_enabled = False
|
||||
elif codex:
|
||||
elif codex and not local:
|
||||
# Large requests legitimately spend tens of seconds in admission/prefill before the
|
||||
# first SSE event: scale the cutoff up to the idle default unless TTFB_STRICT is set.
|
||||
disable_above = env_float("HERMES_CODEX_TTFB_DISABLE_ABOVE_TOKENS", 10_000.0)
|
||||
@@ -1206,7 +1210,7 @@ def _resolve_nonstream_watchdogs(agent, api_kwargs: dict) -> _NonStreamWatchdogs
|
||||
"(context=~%s tokens) per HERMES_CODEX_TTFB_MAX_SECONDS.", ttfb_timeout, ttfb_cap,
|
||||
f"{est_tokens:,}")
|
||||
ttfb_timeout = ttfb_cap
|
||||
elif not ttfb_explicit and (base_url := getattr(agent, "base_url", None)) and is_local_endpoint(base_url):
|
||||
elif not ttfb_explicit and local:
|
||||
# A local server prefills for minutes before its first event; the chat-completions
|
||||
# siblings already grant local endpoints the local stale ceiling, so the Responses
|
||||
# transport gets the same grace instead of the 120s hosted cutoff (#92302).
|
||||
|
||||
@@ -5,6 +5,8 @@ Ensures:
|
||||
the large-context stale timeout floor (e.g. 1200s at >100K tokens, 900s at >50K tokens)
|
||||
in interruptible_api_call. A baseline short timeout of 0.2s is elevated so a 0.5s call succeeds without stale_call_kill.
|
||||
2. Large requests scale TTFB timeout for xai-oauth (codex_responses) instead of killing at a short TTFB cutoff.
|
||||
3. The 1500s codex hard ceiling clamps hosted codex_responses (xAI) but does not tighten a local
|
||||
Responses endpoint's (e.g. 127.0.0.1) configured stale timeout.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -134,3 +136,19 @@ def test_xai_responses_ttfb_scaled_for_large_requests(monkeypatch):
|
||||
result = h.interruptible_api_call(agent, {"messages": [{"role": "user", "content": large_text}], "stream": True})
|
||||
assert result.ok is True
|
||||
assert "codex_ttfb_kill" not in closes
|
||||
|
||||
|
||||
def test_hard_ceiling_clamps_hosted_but_not_local_responses_endpoints(monkeypatch):
|
||||
"""The 1500s hard ceiling bounds hosted codex_responses (xAI) but must not tighten a
|
||||
local Responses server's configured stale timeout."""
|
||||
from agent import chat_completion_helpers as h
|
||||
|
||||
monkeypatch.delenv("HERMES_CODEX_HARD_TIMEOUT_SECONDS", raising=False)
|
||||
kwargs = {"input": [{"role": "user", "content": "hi"}]}
|
||||
hosted = _make_mock_agent()
|
||||
local = _make_mock_agent(provider="custom", base_url="http://127.0.0.1:1234/v1")
|
||||
for agent in (hosted, local):
|
||||
agent._compute_non_stream_stale_timeout = lambda api_kwargs: 3000.0
|
||||
|
||||
assert h._resolve_nonstream_watchdogs(hosted, kwargs).stale_timeout == 1500.0
|
||||
assert h._resolve_nonstream_watchdogs(local, kwargs).stale_timeout == 3000.0
|
||||
|
||||
Reference in New Issue
Block a user