perf(gateway): prime provider context metadata during startup warm-up
The gateway boot warm-up primed the run_agent import graph and tool schemas, but
the configured route's context-window metadata was only resolved on the first
inbound turn, so AIAgent construction paid the catalog HTTP probe (codex OAuth
/backend-api/codex/models, OpenRouter metadata) between the submit ACK and the
inference request. Resolve it inside the existing executor-thread warm-up via
_resolve_gateway_model_context(), which applies the same route/credential/profile
rules as the turn (including custom_providers pins), and log the resolved window.
Failures stay non-fatal: lazy init and the bounded startup gate are unchanged.
Fixes #105986
Salvaged from #105994 by @kokhlo; trimmed to two invariant tests and rebased onto
main where the context-file warm-up was already deferred (87661da314).
This commit is contained in:
committed by
Teknium
parent
6a2452393f
commit
614f11d3f6
@@ -918,8 +918,9 @@ def _warm_turn_machinery_sync() -> int:
|
||||
"""Synchronously initialize first-turn prerequisites (executor thread); returns the schema count.
|
||||
|
||||
Covers the lazy init seen in skeleton turns: ``run_agent`` import graph, tool schemas (+ ``check_fn``
|
||||
TTL cache), and the local Python toolchain probe (#106064). Context files remain lazy because they
|
||||
need the active turn's agent and model context."""
|
||||
TTL cache), the local Python toolchain probe (#106064), and the default route's context-window
|
||||
metadata (#105986) — a catalog HTTP probe that must not sit between the first inbound turn and its
|
||||
inference request. Context files remain lazy because they need the active turn's agent."""
|
||||
import run_agent # noqa: F401 # heavy import graph, cached in sys.modules
|
||||
import model_tools
|
||||
|
||||
@@ -933,6 +934,13 @@ def _warm_turn_machinery_sync() -> int:
|
||||
from tools.env_probe import get_environment_probe_line
|
||||
|
||||
get_environment_probe_line()
|
||||
try:
|
||||
# Same route/credential/profile rules as the turn itself; primes the process-local catalog
|
||||
# caches (codex OAuth, OpenRouter) so AIAgent construction on the first turn is a cache hit.
|
||||
ctx = _resolve_gateway_model_context()
|
||||
logger.info("Model context warmed: %s -> %d tokens (%s)", ctx.model, ctx.context_length, ctx.context_source)
|
||||
except Exception:
|
||||
logger.debug("model-context warm-up failed (non-fatal)", exc_info=True)
|
||||
return len(tool_defs)
|
||||
|
||||
|
||||
|
||||
48
tests/gateway/test_startup_model_context_warmup.py
Normal file
48
tests/gateway/test_startup_model_context_warmup.py
Normal file
@@ -0,0 +1,48 @@
|
||||
"""Model-context warm-up inside the gateway boot warm-up (#105986).
|
||||
|
||||
The startup warm-up primed the import graph and tool schemas, but the default
|
||||
route's context-window metadata was only resolved on the first inbound turn —
|
||||
a blocking catalog HTTP probe (codex OAuth, OpenRouter metadata) inside AIAgent
|
||||
construction, between the submit ACK and the inference request. The warm-up now
|
||||
resolves the default route's model context up front with the same route /
|
||||
credential rules as the turn itself, so the probe's caches are primed before
|
||||
the inbound gate opens.
|
||||
"""
|
||||
|
||||
import gateway.run as gateway_run
|
||||
|
||||
|
||||
def _quiet_tool_side(monkeypatch, tool_count):
|
||||
import model_tools
|
||||
|
||||
monkeypatch.setattr(model_tools, "get_tool_definitions", lambda quiet_mode=False: ["t"] * tool_count)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.config.load_config_readonly", lambda: {"agent": {"environment_probe": False}})
|
||||
|
||||
|
||||
def test_model_context_warmup_primes_default_route(monkeypatch):
|
||||
"""Warm-up resolves the default gateway route's model context exactly once."""
|
||||
resolved: list = []
|
||||
|
||||
def fake_resolve(model=None, route=None):
|
||||
resolved.append((model, route))
|
||||
return gateway_run._GatewayModelContext(
|
||||
model="m", provider="p", base_url="", context_length=128000, context_source="detected")
|
||||
|
||||
monkeypatch.setattr(gateway_run, "_resolve_gateway_model_context", fake_resolve)
|
||||
_quiet_tool_side(monkeypatch, 3)
|
||||
|
||||
assert gateway_run._warm_turn_machinery_sync() == 3
|
||||
assert resolved == [(None, None)]
|
||||
|
||||
|
||||
def test_model_context_warmup_failure_is_non_fatal(monkeypatch):
|
||||
"""A resolver failure degrades to lazy init — warm-up still returns the tool count."""
|
||||
|
||||
def boom(model=None, route=None):
|
||||
raise RuntimeError("catalog unreachable")
|
||||
|
||||
monkeypatch.setattr(gateway_run, "_resolve_gateway_model_context", boom)
|
||||
_quiet_tool_side(monkeypatch, 7)
|
||||
|
||||
assert gateway_run._warm_turn_machinery_sync() == 7
|
||||
Reference in New Issue
Block a user