fix(agent): apply large-context watchdog accommodations to all codex_responses transports
(cherry picked from commit 51a60661a4c777e20df7edfab5af9f3c9956aac5)
This commit is contained in:
@@ -1165,7 +1165,7 @@ def _resolve_nonstream_watchdogs(agent, api_kwargs: dict) -> _NonStreamWatchdogs
|
||||
est_tokens = estimate_request_context_tokens(api_kwargs)
|
||||
effort_floor = _high_effort_silence_floor(agent) if codex else 0.0
|
||||
codex_floor = 0.0
|
||||
if codex and openai_codex_backend:
|
||||
if codex:
|
||||
# Raise the stale floor for large payloads so healthy gateway-scale
|
||||
# requests aren't aborted mid-prefill.
|
||||
codex_floor = openai_codex_stale_timeout_floor(est_tokens)
|
||||
@@ -1188,13 +1188,13 @@ def _resolve_nonstream_watchdogs(agent, api_kwargs: dict) -> _NonStreamWatchdogs
|
||||
ttfb_timeout = env_float("HERMES_CODEX_TTFB_TIMEOUT_SECONDS", 120.0)
|
||||
if ttfb_timeout <= 0:
|
||||
ttfb_enabled = False
|
||||
elif openai_codex_backend:
|
||||
elif codex:
|
||||
# Large requests legitimately spend tens of seconds in admission/prefill before the
|
||||
# first SSE event: scale the cutoff up to the idle default unless TTFB_STRICT is set.
|
||||
disable_above = env_float("HERMES_CODEX_TTFB_DISABLE_ABOVE_TOKENS", 10_000.0)
|
||||
strict = os.environ.get("HERMES_CODEX_TTFB_STRICT", "").strip().lower() in {"1", "true", "yes", "on"}
|
||||
if not strict and disable_above > 0 and est_tokens >= disable_above and ttfb_timeout < idle_default:
|
||||
logger.info("Scaling openai-codex no-event TTFB watchdog from %.0fs to %.0fs "
|
||||
logger.info("Scaling codex-responses no-event TTFB watchdog from %.0fs to %.0fs "
|
||||
"for large request (context=~%s tokens >= %.0f). "
|
||||
"Set HERMES_CODEX_TTFB_STRICT=1 to keep the smaller cutoff.", ttfb_timeout, idle_default,
|
||||
f"{est_tokens:,}", disable_above)
|
||||
@@ -1202,7 +1202,7 @@ def _resolve_nonstream_watchdogs(agent, api_kwargs: dict) -> _NonStreamWatchdogs
|
||||
# Opt-in ceiling (0 = off): a 120s default here silently undid the scale-up above (#91621).
|
||||
ttfb_cap = env_float("HERMES_CODEX_TTFB_MAX_SECONDS", 0.0)
|
||||
if ttfb_cap > 0 and ttfb_timeout > ttfb_cap:
|
||||
logger.info("Capping openai-codex no-event TTFB timeout from %.0fs to %.0fs "
|
||||
logger.info("Capping codex-responses no-event TTFB timeout from %.0fs to %.0fs "
|
||||
"(context=~%s tokens) per HERMES_CODEX_TTFB_MAX_SECONDS.", ttfb_timeout, ttfb_cap,
|
||||
f"{est_tokens:,}")
|
||||
ttfb_timeout = ttfb_cap
|
||||
|
||||
136
tests/agent/test_responses_watchdog_accommodations.py
Normal file
136
tests/agent/test_responses_watchdog_accommodations.py
Normal file
@@ -0,0 +1,136 @@
|
||||
"""Regression tests for #86444 - large-context watchdog accommodations for codex_responses transports.
|
||||
|
||||
Ensures:
|
||||
1. xAI Responses transport (api_mode="codex_responses", provider="xai-oauth") receives
|
||||
the large-context stale timeout floor (e.g. 1200s at >100K tokens, 900s at >50K tokens)
|
||||
in interruptible_api_call. A baseline short timeout of 0.2s is elevated so a 0.5s call succeeds without stale_call_kill.
|
||||
2. Large requests scale TTFB timeout for xai-oauth (codex_responses) instead of killing at a short TTFB cutoff.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
import time
|
||||
import types
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
sys.modules.setdefault("fire", types.SimpleNamespace(Fire=lambda *a, **k: None))
|
||||
sys.modules.setdefault("firecrawl", types.SimpleNamespace(Firecrawl=object))
|
||||
sys.modules.setdefault("fal_client", types.SimpleNamespace())
|
||||
|
||||
|
||||
def _make_mock_agent(provider="xai-oauth", api_mode="codex_responses", base_url="https://api.x.ai/v1"):
|
||||
agent = MagicMock()
|
||||
agent.provider = provider
|
||||
agent.api_mode = api_mode
|
||||
agent.base_url = base_url
|
||||
agent._base_url_lower = base_url.lower()
|
||||
agent._base_url_hostname = "api.x.ai"
|
||||
agent._emit_status = lambda *a, **k: None
|
||||
agent._buffer_status = lambda *a, **k: None
|
||||
agent._touch_activity = lambda *a, **k: None
|
||||
agent._codex_stream_last_event_ts = None
|
||||
agent._interrupt_requested = False
|
||||
agent.timeout = 180.0
|
||||
agent._compute_non_stream_stale_timeout = lambda api_kwargs: 0.2
|
||||
return agent
|
||||
|
||||
|
||||
def test_xai_responses_receives_stale_timeout_floor_in_api_call(monkeypatch):
|
||||
"""interruptible_api_call elevates stale timeout to context floor (>=900s) for xai-oauth codex_responses.
|
||||
|
||||
When baseline _compute_non_stream_stale_timeout is 0.2s and TTFB watchdog is disabled (0),
|
||||
a 0.5s execution would fail with stale_call_kill on base (stale timeout at 0.2s), but passes when the floor elevates it to >=900s.
|
||||
"""
|
||||
from agent import chat_completion_helpers as h
|
||||
|
||||
agent = _make_mock_agent(provider="xai-oauth", api_mode="codex_responses", base_url="https://api.x.ai/v1")
|
||||
assert h._is_openai_codex_backend(agent) is False
|
||||
|
||||
monkeypatch.setenv("HERMES_CODEX_TTFB_TIMEOUT_SECONDS", "0")
|
||||
closes = []
|
||||
monkeypatch.setattr(agent, "_create_request_openai_client", lambda **k: SimpleNamespace())
|
||||
monkeypatch.setattr(agent, "_close_request_openai_client", lambda *a, **k: None)
|
||||
monkeypatch.setattr(
|
||||
agent, "_abort_request_openai_client",
|
||||
lambda c, reason=None: closes.append(reason),
|
||||
)
|
||||
|
||||
# 65k tokens estimate -> floor is >=900.0s
|
||||
large_text = "word " * 65_000
|
||||
api_kwargs = {"messages": [{"role": "user", "content": large_text}], "stream": False}
|
||||
|
||||
sentinel = SimpleNamespace(ok=True)
|
||||
stop = {"flag": False}
|
||||
|
||||
def fake_hang_or_delay(api_kwargs, client=None, on_first_delta=None):
|
||||
deadline = time.time() + 0.5
|
||||
while time.time() < deadline and not stop["flag"] and not agent._interrupt_requested:
|
||||
time.sleep(0.02)
|
||||
if "stale_call_kill" in closes:
|
||||
raise RuntimeError("aborted by stale kill")
|
||||
return sentinel
|
||||
|
||||
monkeypatch.setattr(agent, "_run_codex_stream", fake_hang_or_delay)
|
||||
|
||||
try:
|
||||
res = h.interruptible_api_call(agent, api_kwargs)
|
||||
assert res is sentinel
|
||||
assert "stale_call_kill" not in closes
|
||||
finally:
|
||||
stop["flag"] = True
|
||||
|
||||
|
||||
def test_xai_responses_ttfb_scaled_for_large_requests(monkeypatch):
|
||||
"""Large requests scale TTFB timeout for xai-oauth (codex_responses) instead of killing at small TTFB cutoff."""
|
||||
from agent import chat_completion_helpers as h
|
||||
|
||||
agent = _make_mock_agent(provider="xai-oauth", api_mode="codex_responses", base_url="https://api.x.ai/v1")
|
||||
assert h._is_openai_codex_backend(agent) is False
|
||||
|
||||
monkeypatch.setenv("HERMES_CODEX_TTFB_TIMEOUT_SECONDS", "0.2")
|
||||
|
||||
closes = []
|
||||
monkeypatch.setattr(agent, "_create_request_openai_client", lambda **k: SimpleNamespace())
|
||||
monkeypatch.setattr(agent, "_close_request_openai_client", lambda *a, **k: None)
|
||||
monkeypatch.setattr(
|
||||
agent, "_abort_request_openai_client",
|
||||
lambda c, reason=None: closes.append(reason),
|
||||
)
|
||||
|
||||
# 1. Small request (<10k tokens) should trip TTFB watchdog at 0.2s
|
||||
stop_small = {"flag": False}
|
||||
|
||||
def fake_hang_small(api_kwargs, client=None, on_first_delta=None):
|
||||
deadline = time.time() + 5
|
||||
while time.time() < deadline and not stop_small["flag"] and not agent._interrupt_requested:
|
||||
time.sleep(0.02)
|
||||
raise RuntimeError("hang")
|
||||
|
||||
monkeypatch.setattr(agent, "_run_codex_stream", fake_hang_small)
|
||||
|
||||
try:
|
||||
with pytest.raises(TimeoutError):
|
||||
h.interruptible_api_call(agent, {"messages": [{"role": "user", "content": "hello"}], "stream": True})
|
||||
assert "codex_ttfb_kill" in closes
|
||||
finally:
|
||||
stop_small["flag"] = True
|
||||
|
||||
# 2. Large request (>50k tokens) scales TTFB timeout to 120s, so it does NOT trip at 0.2s
|
||||
closes.clear()
|
||||
large_text = "word " * 65_000
|
||||
stop_large = {"flag": False}
|
||||
|
||||
def fake_stream_delayed(api_kwargs, client=None, on_first_delta=None):
|
||||
# Sleeps for 0.5s (longer than the initial 0.2s TTFB cutoff)
|
||||
time.sleep(0.5)
|
||||
return SimpleNamespace(ok=True)
|
||||
|
||||
monkeypatch.setattr(agent, "_run_codex_stream", fake_stream_delayed)
|
||||
|
||||
result = h.interruptible_api_call(agent, {"messages": [{"role": "user", "content": large_text}], "stream": True})
|
||||
assert result.ok is True
|
||||
assert "codex_ttfb_kill" not in closes
|
||||
Reference in New Issue
Block a user