From 614f11d3f6a0eddc381f05f5c4dd759d0f752be4 Mon Sep 17 00:00:00 2001 From: Konstantin Khlopkov Date: Sat, 19 Sep 2026 00:01:52 -0700 Subject: [PATCH] perf(gateway): prime provider context metadata during startup warm-up The gateway boot warm-up primed the run_agent import graph and tool schemas, but the configured route's context-window metadata was only resolved on the first inbound turn, so AIAgent construction paid the catalog HTTP probe (codex OAuth /backend-api/codex/models, OpenRouter metadata) between the submit ACK and the inference request. Resolve it inside the existing executor-thread warm-up via _resolve_gateway_model_context(), which applies the same route/credential/profile rules as the turn (including custom_providers pins), and log the resolved window. Failures stay non-fatal: lazy init and the bounded startup gate are unchanged. Fixes #105986 Salvaged from #105994 by @kokhlo; trimmed to two invariant tests and rebased onto main where the context-file warm-up was already deferred (87661da3145). --- gateway/run.py | 12 ++++- .../test_startup_model_context_warmup.py | 48 +++++++++++++++++++ 2 files changed, 58 insertions(+), 2 deletions(-) create mode 100644 tests/gateway/test_startup_model_context_warmup.py diff --git a/gateway/run.py b/gateway/run.py index c4cc9d5ddb..87465dd524 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -918,8 +918,9 @@ def _warm_turn_machinery_sync() -> int: """Synchronously initialize first-turn prerequisites (executor thread); returns the schema count. Covers the lazy init seen in skeleton turns: ``run_agent`` import graph, tool schemas (+ ``check_fn`` - TTL cache), and the local Python toolchain probe (#106064). Context files remain lazy because they - need the active turn's agent and model context.""" + TTL cache), the local Python toolchain probe (#106064), and the default route's context-window + metadata (#105986) — a catalog HTTP probe that must not sit between the first inbound turn and its + inference request. Context files remain lazy because they need the active turn's agent.""" import run_agent # noqa: F401 # heavy import graph, cached in sys.modules import model_tools @@ -933,6 +934,13 @@ def _warm_turn_machinery_sync() -> int: from tools.env_probe import get_environment_probe_line get_environment_probe_line() + try: + # Same route/credential/profile rules as the turn itself; primes the process-local catalog + # caches (codex OAuth, OpenRouter) so AIAgent construction on the first turn is a cache hit. + ctx = _resolve_gateway_model_context() + logger.info("Model context warmed: %s -> %d tokens (%s)", ctx.model, ctx.context_length, ctx.context_source) + except Exception: + logger.debug("model-context warm-up failed (non-fatal)", exc_info=True) return len(tool_defs) diff --git a/tests/gateway/test_startup_model_context_warmup.py b/tests/gateway/test_startup_model_context_warmup.py new file mode 100644 index 0000000000..9ba2102ba4 --- /dev/null +++ b/tests/gateway/test_startup_model_context_warmup.py @@ -0,0 +1,48 @@ +"""Model-context warm-up inside the gateway boot warm-up (#105986). + +The startup warm-up primed the import graph and tool schemas, but the default +route's context-window metadata was only resolved on the first inbound turn — +a blocking catalog HTTP probe (codex OAuth, OpenRouter metadata) inside AIAgent +construction, between the submit ACK and the inference request. The warm-up now +resolves the default route's model context up front with the same route / +credential rules as the turn itself, so the probe's caches are primed before +the inbound gate opens. +""" + +import gateway.run as gateway_run + + +def _quiet_tool_side(monkeypatch, tool_count): + import model_tools + + monkeypatch.setattr(model_tools, "get_tool_definitions", lambda quiet_mode=False: ["t"] * tool_count) + monkeypatch.setattr( + "hermes_cli.config.load_config_readonly", lambda: {"agent": {"environment_probe": False}}) + + +def test_model_context_warmup_primes_default_route(monkeypatch): + """Warm-up resolves the default gateway route's model context exactly once.""" + resolved: list = [] + + def fake_resolve(model=None, route=None): + resolved.append((model, route)) + return gateway_run._GatewayModelContext( + model="m", provider="p", base_url="", context_length=128000, context_source="detected") + + monkeypatch.setattr(gateway_run, "_resolve_gateway_model_context", fake_resolve) + _quiet_tool_side(monkeypatch, 3) + + assert gateway_run._warm_turn_machinery_sync() == 3 + assert resolved == [(None, None)] + + +def test_model_context_warmup_failure_is_non_fatal(monkeypatch): + """A resolver failure degrades to lazy init — warm-up still returns the tool count.""" + + def boom(model=None, route=None): + raise RuntimeError("catalog unreachable") + + monkeypatch.setattr(gateway_run, "_resolve_gateway_model_context", boom) + _quiet_tool_side(monkeypatch, 7) + + assert gateway_run._warm_turn_machinery_sync() == 7