fix(agent_init): judge the served window only for local endpoints; clamp after the floor as before

Review findings on the first cut:

* A hosted provider with a stale model.ollama_num_ctx passed the floor for
  a 40K model although nothing ever raises a hosted window. The served
  window now counts only when the endpoint is local (is_local_endpoint),
  the same gate the num_ctx probe uses.
* Moving the whole num_ctx phase ahead of the floor also moved tek's
  compressor clamp ahead of it, which flipped two observables: a
  model.context_length above a sub-64K num_ctx was rejected instead of
  constructed-and-clamped, and the floor's message reported the clamped
  value with advice (set model.context_length) that could not help. The
  phase is split: resolution runs before the floor, the clamp
  (_clamp_compressor_to_ollama_num_ctx) stays at its original position,
  so every case main constructed still constructs with identical
  compressor numbers and the floor's message is unchanged.

Tests: the harness is a module-level helper so the new class no longer
re-collects the parent's tests (13 -> 11 collected); the negative pins
the local-endpoint gate (red when the gate is dropped) instead of a case
main already rejected.
This commit is contained in:
kshitijk4poor
2026-09-19 11:38:54 +05:30
committed by kshitij
parent 3de7140bcd
commit afaa53e5fe
2 changed files with 45 additions and 39 deletions

View File

@@ -1912,11 +1912,11 @@ def _enforce_minimum_context(agent):
# Reject windows below the 64K floor needed for reliable tool-calling; an explicit
# positive model.context_length on LM Studio is allowed below the floor.
_ctx = getattr(agent.context_compressor, "context_length", 0)
# An Ollama server serves num_ctx, not the GGUF's advertised window: a Modelfile or
# A local Ollama server serves num_ctx, not the GGUF's advertised window: a Modelfile or
# model.ollama_num_ctx at 64K+ is a usable window even when the metadata says 40K (#100437).
_served = getattr(agent, "_ollama_num_ctx", None)
if isinstance(_served, int) and not isinstance(_served, bool) and _served > 0:
_ctx = max(_ctx or 0, _served)
# Only a local endpoint can honour num_ctx, so a stale override never admits a hosted model.
if agent._ollama_num_ctx and agent.base_url and is_local_endpoint(agent.base_url):
_ctx = max(_ctx or 0, agent._ollama_num_ctx)
_allow_lmstudio_explicit_below_floor = (
str(agent.provider or "").strip().lower() == "lmstudio"
and isinstance(agent._config_context_length, int)
@@ -2043,6 +2043,9 @@ def _configure_ollama_num_ctx(agent, _model_cfg, _config_context_length):
"Ollama num_ctx: will request %d tokens (model max from /api/show)",
agent._ollama_num_ctx,
)
def _clamp_compressor_to_ollama_num_ctx(agent):
# Recalibrate the compressor to the served window: every request runs at num_ctx, so a
# trigger derived from the probed model window could sit above it and never fire.
# A config that sets only model.ollama_num_ctx (without model.context_length) previously left the
@@ -2331,12 +2334,12 @@ def init_agent(
agent, _agent_cfg, base_url
)
_build_context_engine(agent, _agent_cfg, cs, _custom_providers, _effective_context_length, session_db)
# num_ctx before the floor: the served Ollama window is part of what the floor judges.
_configure_ollama_num_ctx(agent, _model_cfg, _config_context_length)
_enforce_minimum_context(agent)
_warn_nonagentic_hermes_model(agent)
_inject_context_engine_tools(agent)
_init_usage_state(agent)
_clamp_compressor_to_ollama_num_ctx(agent)
_emit_compression_summary(agent, cs)
_snapshot_primary_runtime(agent)

View File

@@ -140,39 +140,40 @@ class TestQueryOllamaSupportsVision:
# ═══════════════════════════════════════════════════════════════════════
def _build_agent(cfg, probed_ctx, base_url="http://localhost:11434/v1"):
import agent.context_compressor as cc_mod
with (
patch("model_tools.get_tool_definitions", return_value=[]),
patch("model_tools.check_toolset_requirements", return_value={}),
patch("agent.process_bootstrap.OpenAI"),
patch("hermes_cli.config.load_config", return_value=cfg),
patch("hermes_cli.config.load_config_readonly", return_value=cfg),
patch(
"agent.model_metadata.get_model_context_length",
return_value=probed_ctx,
),
patch.object(
cc_mod, "get_model_context_length", return_value=probed_ctx,
),
):
from run_agent import AIAgent
return AIAgent(
model="gemma3:27b",
api_key="ollama",
base_url=base_url,
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
)
class TestCompressorClampsToNumCtx:
"""A config setting ONLY model.ollama_num_ctx (no model.context_length)
must not leave the compressor targeting the probed model window while
requests run at the smaller served num_ctx."""
def _build_agent(self, cfg, probed_ctx):
import agent.context_compressor as cc_mod
with (
patch("model_tools.get_tool_definitions", return_value=[]),
patch("model_tools.check_toolset_requirements", return_value={}),
patch("agent.process_bootstrap.OpenAI"),
patch("hermes_cli.config.load_config", return_value=cfg),
patch("hermes_cli.config.load_config_readonly", return_value=cfg),
patch(
"agent.model_metadata.get_model_context_length",
return_value=probed_ctx,
),
patch.object(
cc_mod, "get_model_context_length", return_value=probed_ctx,
),
):
from run_agent import AIAgent
return AIAgent(
model="gemma3:27b",
api_key="ollama",
base_url="http://localhost:11434/v1",
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
)
def test_num_ctx_only_config_clamps_compressor_window(self):
agent = self._build_agent(
agent = _build_agent(
{"agent": {}, "model": {"ollama_num_ctx": 65536}}, probed_ctx=262144
)
assert agent._ollama_num_ctx == 65536
@@ -183,7 +184,7 @@ class TestCompressorClampsToNumCtx:
assert agent.context_compressor.threshold_tokens < 65536
def test_larger_num_ctx_does_not_inflate_compressor_window(self):
agent = self._build_agent(
agent = _build_agent(
{"agent": {}, "model": {"ollama_num_ctx": 131072}}, probed_ctx=65536
)
# num_ctx above the resolved window must not RAISE the compressor
@@ -191,16 +192,18 @@ class TestCompressorClampsToNumCtx:
assert agent.context_compressor.context_length == 65536
class TestServedNumCtxSatisfiesTheFloor(TestCompressorClampsToNumCtx):
"""#100437: the 64K floor judges the window Ollama actually serves. A Modelfile or
model.ollama_num_ctx at 64K+ is usable even when the GGUF metadata advertises 40K, so
class TestServedNumCtxSatisfiesTheFloor:
"""#100437: the 64K floor judges the window a local Ollama server actually serves. A Modelfile
or model.ollama_num_ctx at 64K+ is usable even when the GGUF metadata advertises 40K, so
construction must succeed; the compressor still targets the smaller probed window."""
def test_explicit_num_ctx_above_the_floor_admits_a_small_metadata_window(self):
agent = self._build_agent({"agent": {}, "model": {"ollama_num_ctx": 65536}}, probed_ctx=40960)
agent = _build_agent({"agent": {}, "model": {"ollama_num_ctx": 65536}}, probed_ctx=40960)
assert agent._ollama_num_ctx == 65536
assert agent.context_compressor.context_length == 40960 # one-directional clamp unchanged
def test_served_window_below_the_floor_is_still_rejected(self):
def test_served_window_counts_only_for_a_local_endpoint(self):
"""Only a local server honours num_ctx; a stale override must not admit a hosted 40K model."""
with pytest.raises(ValueError, match="below the minimum"):
self._build_agent({"agent": {}, "model": {"ollama_num_ctx": 32768}}, probed_ctx=40960)
_build_agent({"agent": {}, "model": {"ollama_num_ctx": 65536}}, probed_ctx=40960,
base_url="https://openrouter.ai/api/v1")