From afaa53e5fe8c99d52bb0df512638a3223e30da99 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 19 Sep 2026 11:38:54 +0530 Subject: [PATCH] fix(agent_init): judge the served window only for local endpoints; clamp after the floor as before Review findings on the first cut: * A hosted provider with a stale model.ollama_num_ctx passed the floor for a 40K model although nothing ever raises a hosted window. The served window now counts only when the endpoint is local (is_local_endpoint), the same gate the num_ctx probe uses. * Moving the whole num_ctx phase ahead of the floor also moved tek's compressor clamp ahead of it, which flipped two observables: a model.context_length above a sub-64K num_ctx was rejected instead of constructed-and-clamped, and the floor's message reported the clamped value with advice (set model.context_length) that could not help. The phase is split: resolution runs before the floor, the clamp (_clamp_compressor_to_ollama_num_ctx) stays at its original position, so every case main constructed still constructs with identical compressor numbers and the floor's message is unchanged. Tests: the harness is a module-level helper so the new class no longer re-collects the parent's tests (13 -> 11 collected); the negative pins the local-endpoint gate (red when the gate is dropped) instead of a case main already rejected. --- agent/agent_init.py | 13 +++--- tests/agent/test_ollama_num_ctx.py | 71 ++++++++++++++++-------------- 2 files changed, 45 insertions(+), 39 deletions(-) diff --git a/agent/agent_init.py b/agent/agent_init.py index a76cb4b007..b48037180b 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -1912,11 +1912,11 @@ def _enforce_minimum_context(agent): # Reject windows below the 64K floor needed for reliable tool-calling; an explicit # positive model.context_length on LM Studio is allowed below the floor. _ctx = getattr(agent.context_compressor, "context_length", 0) - # An Ollama server serves num_ctx, not the GGUF's advertised window: a Modelfile or + # A local Ollama server serves num_ctx, not the GGUF's advertised window: a Modelfile or # model.ollama_num_ctx at 64K+ is a usable window even when the metadata says 40K (#100437). - _served = getattr(agent, "_ollama_num_ctx", None) - if isinstance(_served, int) and not isinstance(_served, bool) and _served > 0: - _ctx = max(_ctx or 0, _served) + # Only a local endpoint can honour num_ctx, so a stale override never admits a hosted model. + if agent._ollama_num_ctx and agent.base_url and is_local_endpoint(agent.base_url): + _ctx = max(_ctx or 0, agent._ollama_num_ctx) _allow_lmstudio_explicit_below_floor = ( str(agent.provider or "").strip().lower() == "lmstudio" and isinstance(agent._config_context_length, int) @@ -2043,6 +2043,9 @@ def _configure_ollama_num_ctx(agent, _model_cfg, _config_context_length): "Ollama num_ctx: will request %d tokens (model max from /api/show)", agent._ollama_num_ctx, ) + + +def _clamp_compressor_to_ollama_num_ctx(agent): # Recalibrate the compressor to the served window: every request runs at num_ctx, so a # trigger derived from the probed model window could sit above it and never fire. # A config that sets only model.ollama_num_ctx (without model.context_length) previously left the @@ -2331,12 +2334,12 @@ def init_agent( agent, _agent_cfg, base_url ) _build_context_engine(agent, _agent_cfg, cs, _custom_providers, _effective_context_length, session_db) - # num_ctx before the floor: the served Ollama window is part of what the floor judges. _configure_ollama_num_ctx(agent, _model_cfg, _config_context_length) _enforce_minimum_context(agent) _warn_nonagentic_hermes_model(agent) _inject_context_engine_tools(agent) _init_usage_state(agent) + _clamp_compressor_to_ollama_num_ctx(agent) _emit_compression_summary(agent, cs) _snapshot_primary_runtime(agent) diff --git a/tests/agent/test_ollama_num_ctx.py b/tests/agent/test_ollama_num_ctx.py index 9bbb882f31..bffb9ef2c4 100644 --- a/tests/agent/test_ollama_num_ctx.py +++ b/tests/agent/test_ollama_num_ctx.py @@ -140,39 +140,40 @@ class TestQueryOllamaSupportsVision: # ═══════════════════════════════════════════════════════════════════════ +def _build_agent(cfg, probed_ctx, base_url="http://localhost:11434/v1"): + import agent.context_compressor as cc_mod + with ( + patch("model_tools.get_tool_definitions", return_value=[]), + patch("model_tools.check_toolset_requirements", return_value={}), + patch("agent.process_bootstrap.OpenAI"), + patch("hermes_cli.config.load_config", return_value=cfg), + patch("hermes_cli.config.load_config_readonly", return_value=cfg), + patch( + "agent.model_metadata.get_model_context_length", + return_value=probed_ctx, + ), + patch.object( + cc_mod, "get_model_context_length", return_value=probed_ctx, + ), + ): + from run_agent import AIAgent + return AIAgent( + model="gemma3:27b", + api_key="ollama", + base_url=base_url, + quiet_mode=True, + skip_context_files=True, + skip_memory=True, + ) + + class TestCompressorClampsToNumCtx: """A config setting ONLY model.ollama_num_ctx (no model.context_length) must not leave the compressor targeting the probed model window while requests run at the smaller served num_ctx.""" - def _build_agent(self, cfg, probed_ctx): - import agent.context_compressor as cc_mod - with ( - patch("model_tools.get_tool_definitions", return_value=[]), - patch("model_tools.check_toolset_requirements", return_value={}), - patch("agent.process_bootstrap.OpenAI"), - patch("hermes_cli.config.load_config", return_value=cfg), - patch("hermes_cli.config.load_config_readonly", return_value=cfg), - patch( - "agent.model_metadata.get_model_context_length", - return_value=probed_ctx, - ), - patch.object( - cc_mod, "get_model_context_length", return_value=probed_ctx, - ), - ): - from run_agent import AIAgent - return AIAgent( - model="gemma3:27b", - api_key="ollama", - base_url="http://localhost:11434/v1", - quiet_mode=True, - skip_context_files=True, - skip_memory=True, - ) - def test_num_ctx_only_config_clamps_compressor_window(self): - agent = self._build_agent( + agent = _build_agent( {"agent": {}, "model": {"ollama_num_ctx": 65536}}, probed_ctx=262144 ) assert agent._ollama_num_ctx == 65536 @@ -183,7 +184,7 @@ class TestCompressorClampsToNumCtx: assert agent.context_compressor.threshold_tokens < 65536 def test_larger_num_ctx_does_not_inflate_compressor_window(self): - agent = self._build_agent( + agent = _build_agent( {"agent": {}, "model": {"ollama_num_ctx": 131072}}, probed_ctx=65536 ) # num_ctx above the resolved window must not RAISE the compressor @@ -191,16 +192,18 @@ class TestCompressorClampsToNumCtx: assert agent.context_compressor.context_length == 65536 -class TestServedNumCtxSatisfiesTheFloor(TestCompressorClampsToNumCtx): - """#100437: the 64K floor judges the window Ollama actually serves. A Modelfile or - model.ollama_num_ctx at 64K+ is usable even when the GGUF metadata advertises 40K, so +class TestServedNumCtxSatisfiesTheFloor: + """#100437: the 64K floor judges the window a local Ollama server actually serves. A Modelfile + or model.ollama_num_ctx at 64K+ is usable even when the GGUF metadata advertises 40K, so construction must succeed; the compressor still targets the smaller probed window.""" def test_explicit_num_ctx_above_the_floor_admits_a_small_metadata_window(self): - agent = self._build_agent({"agent": {}, "model": {"ollama_num_ctx": 65536}}, probed_ctx=40960) + agent = _build_agent({"agent": {}, "model": {"ollama_num_ctx": 65536}}, probed_ctx=40960) assert agent._ollama_num_ctx == 65536 assert agent.context_compressor.context_length == 40960 # one-directional clamp unchanged - def test_served_window_below_the_floor_is_still_rejected(self): + def test_served_window_counts_only_for_a_local_endpoint(self): + """Only a local server honours num_ctx; a stale override must not admit a hosted 40K model.""" with pytest.raises(ValueError, match="below the minimum"): - self._build_agent({"agent": {}, "model": {"ollama_num_ctx": 32768}}, probed_ctx=40960) + _build_agent({"agent": {}, "model": {"ollama_num_ctx": 65536}}, probed_ctx=40960, + base_url="https://openrouter.ai/api/v1")