fix(agent_init): judge the served window only for local endpoints; clamp after the floor as before
Review findings on the first cut: * A hosted provider with a stale model.ollama_num_ctx passed the floor for a 40K model although nothing ever raises a hosted window. The served window now counts only when the endpoint is local (is_local_endpoint), the same gate the num_ctx probe uses. * Moving the whole num_ctx phase ahead of the floor also moved tek's compressor clamp ahead of it, which flipped two observables: a model.context_length above a sub-64K num_ctx was rejected instead of constructed-and-clamped, and the floor's message reported the clamped value with advice (set model.context_length) that could not help. The phase is split: resolution runs before the floor, the clamp (_clamp_compressor_to_ollama_num_ctx) stays at its original position, so every case main constructed still constructs with identical compressor numbers and the floor's message is unchanged. Tests: the harness is a module-level helper so the new class no longer re-collects the parent's tests (13 -> 11 collected); the negative pins the local-endpoint gate (red when the gate is dropped) instead of a case main already rejected.
This commit is contained in:
@@ -1912,11 +1912,11 @@ def _enforce_minimum_context(agent):
|
||||
# Reject windows below the 64K floor needed for reliable tool-calling; an explicit
|
||||
# positive model.context_length on LM Studio is allowed below the floor.
|
||||
_ctx = getattr(agent.context_compressor, "context_length", 0)
|
||||
# An Ollama server serves num_ctx, not the GGUF's advertised window: a Modelfile or
|
||||
# A local Ollama server serves num_ctx, not the GGUF's advertised window: a Modelfile or
|
||||
# model.ollama_num_ctx at 64K+ is a usable window even when the metadata says 40K (#100437).
|
||||
_served = getattr(agent, "_ollama_num_ctx", None)
|
||||
if isinstance(_served, int) and not isinstance(_served, bool) and _served > 0:
|
||||
_ctx = max(_ctx or 0, _served)
|
||||
# Only a local endpoint can honour num_ctx, so a stale override never admits a hosted model.
|
||||
if agent._ollama_num_ctx and agent.base_url and is_local_endpoint(agent.base_url):
|
||||
_ctx = max(_ctx or 0, agent._ollama_num_ctx)
|
||||
_allow_lmstudio_explicit_below_floor = (
|
||||
str(agent.provider or "").strip().lower() == "lmstudio"
|
||||
and isinstance(agent._config_context_length, int)
|
||||
@@ -2043,6 +2043,9 @@ def _configure_ollama_num_ctx(agent, _model_cfg, _config_context_length):
|
||||
"Ollama num_ctx: will request %d tokens (model max from /api/show)",
|
||||
agent._ollama_num_ctx,
|
||||
)
|
||||
|
||||
|
||||
def _clamp_compressor_to_ollama_num_ctx(agent):
|
||||
# Recalibrate the compressor to the served window: every request runs at num_ctx, so a
|
||||
# trigger derived from the probed model window could sit above it and never fire.
|
||||
# A config that sets only model.ollama_num_ctx (without model.context_length) previously left the
|
||||
@@ -2331,12 +2334,12 @@ def init_agent(
|
||||
agent, _agent_cfg, base_url
|
||||
)
|
||||
_build_context_engine(agent, _agent_cfg, cs, _custom_providers, _effective_context_length, session_db)
|
||||
# num_ctx before the floor: the served Ollama window is part of what the floor judges.
|
||||
_configure_ollama_num_ctx(agent, _model_cfg, _config_context_length)
|
||||
_enforce_minimum_context(agent)
|
||||
_warn_nonagentic_hermes_model(agent)
|
||||
_inject_context_engine_tools(agent)
|
||||
_init_usage_state(agent)
|
||||
_clamp_compressor_to_ollama_num_ctx(agent)
|
||||
_emit_compression_summary(agent, cs)
|
||||
_snapshot_primary_runtime(agent)
|
||||
|
||||
|
||||
@@ -140,39 +140,40 @@ class TestQueryOllamaSupportsVision:
|
||||
# ═══════════════════════════════════════════════════════════════════════
|
||||
|
||||
|
||||
def _build_agent(cfg, probed_ctx, base_url="http://localhost:11434/v1"):
|
||||
import agent.context_compressor as cc_mod
|
||||
with (
|
||||
patch("model_tools.get_tool_definitions", return_value=[]),
|
||||
patch("model_tools.check_toolset_requirements", return_value={}),
|
||||
patch("agent.process_bootstrap.OpenAI"),
|
||||
patch("hermes_cli.config.load_config", return_value=cfg),
|
||||
patch("hermes_cli.config.load_config_readonly", return_value=cfg),
|
||||
patch(
|
||||
"agent.model_metadata.get_model_context_length",
|
||||
return_value=probed_ctx,
|
||||
),
|
||||
patch.object(
|
||||
cc_mod, "get_model_context_length", return_value=probed_ctx,
|
||||
),
|
||||
):
|
||||
from run_agent import AIAgent
|
||||
return AIAgent(
|
||||
model="gemma3:27b",
|
||||
api_key="ollama",
|
||||
base_url=base_url,
|
||||
quiet_mode=True,
|
||||
skip_context_files=True,
|
||||
skip_memory=True,
|
||||
)
|
||||
|
||||
|
||||
class TestCompressorClampsToNumCtx:
|
||||
"""A config setting ONLY model.ollama_num_ctx (no model.context_length)
|
||||
must not leave the compressor targeting the probed model window while
|
||||
requests run at the smaller served num_ctx."""
|
||||
|
||||
def _build_agent(self, cfg, probed_ctx):
|
||||
import agent.context_compressor as cc_mod
|
||||
with (
|
||||
patch("model_tools.get_tool_definitions", return_value=[]),
|
||||
patch("model_tools.check_toolset_requirements", return_value={}),
|
||||
patch("agent.process_bootstrap.OpenAI"),
|
||||
patch("hermes_cli.config.load_config", return_value=cfg),
|
||||
patch("hermes_cli.config.load_config_readonly", return_value=cfg),
|
||||
patch(
|
||||
"agent.model_metadata.get_model_context_length",
|
||||
return_value=probed_ctx,
|
||||
),
|
||||
patch.object(
|
||||
cc_mod, "get_model_context_length", return_value=probed_ctx,
|
||||
),
|
||||
):
|
||||
from run_agent import AIAgent
|
||||
return AIAgent(
|
||||
model="gemma3:27b",
|
||||
api_key="ollama",
|
||||
base_url="http://localhost:11434/v1",
|
||||
quiet_mode=True,
|
||||
skip_context_files=True,
|
||||
skip_memory=True,
|
||||
)
|
||||
|
||||
def test_num_ctx_only_config_clamps_compressor_window(self):
|
||||
agent = self._build_agent(
|
||||
agent = _build_agent(
|
||||
{"agent": {}, "model": {"ollama_num_ctx": 65536}}, probed_ctx=262144
|
||||
)
|
||||
assert agent._ollama_num_ctx == 65536
|
||||
@@ -183,7 +184,7 @@ class TestCompressorClampsToNumCtx:
|
||||
assert agent.context_compressor.threshold_tokens < 65536
|
||||
|
||||
def test_larger_num_ctx_does_not_inflate_compressor_window(self):
|
||||
agent = self._build_agent(
|
||||
agent = _build_agent(
|
||||
{"agent": {}, "model": {"ollama_num_ctx": 131072}}, probed_ctx=65536
|
||||
)
|
||||
# num_ctx above the resolved window must not RAISE the compressor
|
||||
@@ -191,16 +192,18 @@ class TestCompressorClampsToNumCtx:
|
||||
assert agent.context_compressor.context_length == 65536
|
||||
|
||||
|
||||
class TestServedNumCtxSatisfiesTheFloor(TestCompressorClampsToNumCtx):
|
||||
"""#100437: the 64K floor judges the window Ollama actually serves. A Modelfile or
|
||||
model.ollama_num_ctx at 64K+ is usable even when the GGUF metadata advertises 40K, so
|
||||
class TestServedNumCtxSatisfiesTheFloor:
|
||||
"""#100437: the 64K floor judges the window a local Ollama server actually serves. A Modelfile
|
||||
or model.ollama_num_ctx at 64K+ is usable even when the GGUF metadata advertises 40K, so
|
||||
construction must succeed; the compressor still targets the smaller probed window."""
|
||||
|
||||
def test_explicit_num_ctx_above_the_floor_admits_a_small_metadata_window(self):
|
||||
agent = self._build_agent({"agent": {}, "model": {"ollama_num_ctx": 65536}}, probed_ctx=40960)
|
||||
agent = _build_agent({"agent": {}, "model": {"ollama_num_ctx": 65536}}, probed_ctx=40960)
|
||||
assert agent._ollama_num_ctx == 65536
|
||||
assert agent.context_compressor.context_length == 40960 # one-directional clamp unchanged
|
||||
|
||||
def test_served_window_below_the_floor_is_still_rejected(self):
|
||||
def test_served_window_counts_only_for_a_local_endpoint(self):
|
||||
"""Only a local server honours num_ctx; a stale override must not admit a hosted 40K model."""
|
||||
with pytest.raises(ValueError, match="below the minimum"):
|
||||
self._build_agent({"agent": {}, "model": {"ollama_num_ctx": 32768}}, probed_ctx=40960)
|
||||
_build_agent({"agent": {}, "model": {"ollama_num_ctx": 65536}}, probed_ctx=40960,
|
||||
base_url="https://openrouter.ai/api/v1")
|
||||
|
||||
Reference in New Issue
Block a user