fix(agent): clamp displayed context usage to the model window
This commit is contained in:
@@ -108,6 +108,7 @@ def context_usage_fields(compressor: Any) -> Dict[str, Any]:
|
||||
maximum = getattr(compressor, "context_length", 0) or 0
|
||||
if not used or not maximum:
|
||||
return {}
|
||||
used = min(used, maximum)
|
||||
source = context_display_source(compressor)
|
||||
return {"context_used": used, "context_max": maximum,
|
||||
"context_percent": max(0, min(100, round(used / maximum * 100))),
|
||||
@@ -163,6 +164,9 @@ def compute_session_context_breakdown(agent: Any, messages: Optional[List[dict]]
|
||||
if delta and delta[0].get("role") == "assistant":
|
||||
delta = delta[1:]
|
||||
source = "provider_usage_plus_estimate" if delta else "provider_usage"
|
||||
# A single prompt can never exceed the model window; any excess is estimate drift.
|
||||
if context_max:
|
||||
context_used = min(context_used, context_max)
|
||||
|
||||
return {
|
||||
"categories": [
|
||||
|
||||
@@ -329,6 +329,8 @@ class CLIStatusBarMixin:
|
||||
except Exception:
|
||||
pass
|
||||
context_length = max(0, getattr(compressor, "context_length", 0) or 0)
|
||||
if context_length:
|
||||
context_tokens = min(context_tokens, context_length)
|
||||
snapshot["context_tokens"] = context_tokens
|
||||
snapshot["context_length"] = context_length or None
|
||||
from agent.context_pin import is_context_pinned
|
||||
|
||||
@@ -47,6 +47,28 @@ def test_breakdown_includes_major_categories():
|
||||
assert data["context_max"] == 200_000
|
||||
assert data["estimated_total"] > 0
|
||||
|
||||
def test_context_used_never_exceeds_model_window():
|
||||
"""Regression for #109760: anchor + appended-delta estimate can overshoot the window
|
||||
(355.8k / 262.1k); one prompt can never be larger than the model's context."""
|
||||
from agent.context_breakdown import context_usage_fields
|
||||
from agent.usage_anchor import capture_usage_anchor
|
||||
|
||||
history = [{"role": "user", "content": "start"}, {"role": "assistant", "content": "ok"}]
|
||||
agent, parts = _make_agent(context_length=262_144)
|
||||
agent._turn_base_usage_anchor = capture_usage_anchor(250_000, 1_000, history)
|
||||
history = history + [{"role": "user", "content": "x" * 800_000}]
|
||||
|
||||
with patch("agent.system_prompt.build_system_prompt_parts", return_value=parts):
|
||||
data = compute_session_context_breakdown(agent, history)
|
||||
|
||||
assert data["context_source"] == "provider_usage_plus_estimate"
|
||||
assert data["context_used"] <= data["context_max"]
|
||||
assert data["context_percent"] == 100
|
||||
|
||||
seeded = context_usage_fields(MagicMock(context_length=262_144, last_prompt_tokens=400_000,
|
||||
last_real_prompt_tokens=150_000))
|
||||
assert seeded["context_used"] <= seeded["context_max"]
|
||||
|
||||
# ── /context renderers (pure functions over the payload) ────────────────────
|
||||
|
||||
from agent.context_breakdown import ( # noqa: E402
|
||||
|
||||
Reference in New Issue
Block a user