Files
hermes-agent/tests/agent/test_context_breakdown.py
jackulau 9f5440e23d fix(agent): report the conversation category even when it is empty
The category filter dropped every zero-token category from the breakdown
payload. For mcp/memory/skills that is right — zero means "not
configured" and the row's absence says so. For conversation it hid the
row on any session whose transcript was empty or pre-turn, so the
Desktop Context usage panel showed System prompt/Tools/Memory but no
Conversation until the first turn completed (#87903): "the transcript
is empty" rendered identically to "the breakdown never measured it".

Zero for the conversation is a MEASUREMENT of something every session
has, so it is exempted from the drop via _ALWAYS_REPORTED; the
membership rule is documented at the constant so later additions argue
from the same principle. Structurally absent categories stay dropped.

Test: an empty-transcript breakdown reports conversation at 0 while
unconfigured optional categories remain omitted.

The desktop half (retained pre-turn snapshot) is already fixed on main:
useContextBreakdown nulls the snapshot mid-turn and the statusbar gauge
falls back to the streamed usage.

Python half salvaged from PR #87925 (author credited).

Fixes #87903
2026-09-27 06:54:56 -05:00

136 lines
5.7 KiB
Python

"""Tests for live session context breakdown."""
from unittest.mock import MagicMock, patch
from agent.context_breakdown import compute_session_context_breakdown
def _make_agent(
*,
stable: str = "identity and guidance",
context: str = "",
volatile: str = "timestamp line",
tools: list | None = None,
context_length: int = 200_000,
last_prompt_tokens: int = 0,
):
agent = MagicMock()
agent.model = "openai/gpt-5.4"
agent.tools = tools or [
{"type": "function", "function": {"name": "terminal", "description": "run"}},
{"type": "function", "function": {"name": "mcp_demo_tool", "description": "mcp"}},
{"type": "function", "function": {"name": "delegate_task", "description": "spawn"}},
]
agent._memory_store = None
agent._memory_enabled = True
agent._user_profile_enabled = True
agent.context_compressor = MagicMock(
context_length=context_length,
last_prompt_tokens=last_prompt_tokens,
)
return agent, {"stable": stable, "context": context, "volatile": volatile}
def test_breakdown_includes_major_categories():
stable = (
"base guidance\n"
"<available_skills>\n demo:\n - hello: hi\n</available_skills>"
)
context = "# Project Context\nFollow AGENTS.md"
volatile = "Current time: now"
history = [{"role": "user", "content": "hello there"}]
agent, parts = _make_agent(stable=stable, context=context, volatile=volatile)
with patch("agent.system_prompt.build_system_prompt_parts", return_value=parts):
data = compute_session_context_breakdown(agent, history)
ids = {item["id"] for item in data["categories"]}
assert {"system_prompt", "tool_definitions", "rules", "skills", "mcp", "subagent_definitions", "conversation"} <= ids
assert data["context_max"] == 200_000
assert data["estimated_total"] > 0
def test_context_used_never_exceeds_model_window():
"""Regression for #109760: anchor + appended-delta estimate can overshoot the window
(355.8k / 262.1k); one prompt can never be larger than the model's context."""
from agent.context_breakdown import context_usage_fields
from agent.usage_anchor import capture_usage_anchor
history = [{"role": "user", "content": "start"}, {"role": "assistant", "content": "ok"}]
agent, parts = _make_agent(context_length=262_144)
agent._turn_base_usage_anchor = capture_usage_anchor(250_000, 1_000, history)
history = history + [{"role": "user", "content": "x" * 800_000}]
with patch("agent.system_prompt.build_system_prompt_parts", return_value=parts):
data = compute_session_context_breakdown(agent, history)
assert data["context_source"] == "provider_usage_plus_estimate"
assert data["context_used"] <= data["context_max"]
assert data["context_percent"] == 100
seeded = context_usage_fields(MagicMock(context_length=262_144, last_prompt_tokens=400_000,
last_real_prompt_tokens=150_000))
assert seeded["context_used"] <= seeded["context_max"]
def test_empty_transcript_still_reports_the_conversation_category():
"""Zero conversation tokens is a measurement, not an absence (#87903).
A fresh session (or one whose first turn never completed) estimates the
conversation at zero. Dropping the row there made "the transcript is empty"
render identically to "the breakdown never measured it" — the Desktop
Context usage panel hid Conversation until a turn completed. Structurally
absent categories (mcp/memory/skills with nothing configured) stay dropped.
"""
agent, parts = _make_agent(
tools=[{"type": "function", "function": {"name": "terminal", "description": "run"}}]
)
history = [] # nothing said yet
with patch("agent.system_prompt.build_system_prompt_parts", return_value=parts):
data = compute_session_context_breakdown(agent, history)
categories = {item["id"]: item["tokens"] for item in data["categories"]}
assert categories["conversation"] == 0
# Optional surfaces with nothing configured are still omitted at zero —
# "not configured" is exactly what their absence communicates.
assert "mcp" not in categories
assert "skills" not in categories
# ── /context renderers (pure functions over the payload) ────────────────────
from agent.context_breakdown import ( # noqa: E402
render_context_breakdown_lines,
render_context_grid,
)
def _payload(**overrides):
base = {
"categories": [
{"id": "system_prompt", "label": "System prompt", "tokens": 10_000},
{"id": "tool_definitions", "label": "Tool definitions", "tokens": 20_000},
{"id": "skills", "label": "Skills", "tokens": 5_000},
{"id": "conversation", "label": "Conversation", "tokens": 15_000},
],
"context_max": 200_000,
"context_percent": 25,
"context_used": 50_000,
"estimated_total": 50_000,
"model": "openai/gpt-test",
}
base.update(overrides)
return base
def test_grid_is_5x20_and_mostly_free():
rows = render_context_grid(_payload())
assert len(rows) == 5
cells = " ".join(rows).split(" ")
assert len(cells) == 100
# 50k / 200k → 25 used cells, 75 free
assert cells.count("·") == 75
# Category glyphs proportional: 10k→5, 20k→10, 5k→2-3, 15k→7-8 cells
assert cells.count("■") == 5
assert cells.count("▣") == 10
def test_breakdown_lines_grid_toggle():
with_grid = render_context_breakdown_lines(_payload(), grid=True)
without = render_context_breakdown_lines(_payload(), grid=False)
assert any("·" in line for line in with_grid[:5])
assert not any("·" in line for line in without[:2])