diff --git a/gateway/run.py b/gateway/run.py index c4cc9d5ddb..87465dd524 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -918,8 +918,9 @@ def _warm_turn_machinery_sync() -> int: """Synchronously initialize first-turn prerequisites (executor thread); returns the schema count. Covers the lazy init seen in skeleton turns: ``run_agent`` import graph, tool schemas (+ ``check_fn`` - TTL cache), and the local Python toolchain probe (#106064). Context files remain lazy because they - need the active turn's agent and model context.""" + TTL cache), the local Python toolchain probe (#106064), and the default route's context-window + metadata (#105986) — a catalog HTTP probe that must not sit between the first inbound turn and its + inference request. Context files remain lazy because they need the active turn's agent.""" import run_agent # noqa: F401 # heavy import graph, cached in sys.modules import model_tools @@ -933,6 +934,13 @@ def _warm_turn_machinery_sync() -> int: from tools.env_probe import get_environment_probe_line get_environment_probe_line() + try: + # Same route/credential/profile rules as the turn itself; primes the process-local catalog + # caches (codex OAuth, OpenRouter) so AIAgent construction on the first turn is a cache hit. + ctx = _resolve_gateway_model_context() + logger.info("Model context warmed: %s -> %d tokens (%s)", ctx.model, ctx.context_length, ctx.context_source) + except Exception: + logger.debug("model-context warm-up failed (non-fatal)", exc_info=True) return len(tool_defs) diff --git a/tests/gateway/test_startup_model_context_warmup.py b/tests/gateway/test_startup_model_context_warmup.py new file mode 100644 index 0000000000..9ba2102ba4 --- /dev/null +++ b/tests/gateway/test_startup_model_context_warmup.py @@ -0,0 +1,48 @@ +"""Model-context warm-up inside the gateway boot warm-up (#105986). + +The startup warm-up primed the import graph and tool schemas, but the default +route's context-window metadata was only resolved on the first inbound turn — +a blocking catalog HTTP probe (codex OAuth, OpenRouter metadata) inside AIAgent +construction, between the submit ACK and the inference request. The warm-up now +resolves the default route's model context up front with the same route / +credential rules as the turn itself, so the probe's caches are primed before +the inbound gate opens. +""" + +import gateway.run as gateway_run + + +def _quiet_tool_side(monkeypatch, tool_count): + import model_tools + + monkeypatch.setattr(model_tools, "get_tool_definitions", lambda quiet_mode=False: ["t"] * tool_count) + monkeypatch.setattr( + "hermes_cli.config.load_config_readonly", lambda: {"agent": {"environment_probe": False}}) + + +def test_model_context_warmup_primes_default_route(monkeypatch): + """Warm-up resolves the default gateway route's model context exactly once.""" + resolved: list = [] + + def fake_resolve(model=None, route=None): + resolved.append((model, route)) + return gateway_run._GatewayModelContext( + model="m", provider="p", base_url="", context_length=128000, context_source="detected") + + monkeypatch.setattr(gateway_run, "_resolve_gateway_model_context", fake_resolve) + _quiet_tool_side(monkeypatch, 3) + + assert gateway_run._warm_turn_machinery_sync() == 3 + assert resolved == [(None, None)] + + +def test_model_context_warmup_failure_is_non_fatal(monkeypatch): + """A resolver failure degrades to lazy init — warm-up still returns the tool count.""" + + def boom(model=None, route=None): + raise RuntimeError("catalog unreachable") + + monkeypatch.setattr(gateway_run, "_resolve_gateway_model_context", boom) + _quiet_tool_side(monkeypatch, 7) + + assert gateway_run._warm_turn_machinery_sync() == 7