test(agent): replace source-scanning marker test with a renderer behaviour test
AGENTS.md bans tests that read .py source. Replace the tokenize scan with a test that drives the real renderers (clarify summary, fallback turn, verbatim user section, oversized record, active-task line with both quote kinds) and asserts the guard regex matches and no bare idiom appears, plus the small-budget skip. The LSP test asserts via _COMPRESSION_MARKER_RE rather than a literal that a count-less copy would also satisfy. Test count unchanged (2 new vs base). Co-authored-by: ahisblessed <ahisblessed@users.noreply.github.com>
This commit is contained in:
@@ -1,6 +1,7 @@
|
||||
"""Tests for the diagnostic reporter (formatting layer)."""
|
||||
from __future__ import annotations
|
||||
|
||||
from agent.compression_marker import _COMPRESSION_MARKER_RE
|
||||
from agent.lsp.reporter import (
|
||||
MAX_PER_FILE,
|
||||
format_diagnostic,
|
||||
@@ -44,8 +45,7 @@ def test_truncate_above_limit_appends_marker():
|
||||
s = "x" * 10000
|
||||
out = truncate(s, limit=200)
|
||||
# Non-imitable elision marker (#121548), not the old bare truncation idiom.
|
||||
assert "HERMES-CONTEXT-COMPRESSION" in out
|
||||
assert out.endswith("⟫")
|
||||
assert _COMPRESSION_MARKER_RE.search(out)
|
||||
assert len(out) <= 200
|
||||
|
||||
|
||||
|
||||
@@ -2,32 +2,28 @@
|
||||
|
||||
The bare bracketed truncation idiom those renderers used to compose was imitated
|
||||
from replayed context into new durable writes (see #83435/#83714). All elision now
|
||||
routes through ``agent.compression_marker.elide`` / ``elide_middle``, and this file
|
||||
pins guard parity and the source-level invariant that the imitable idiom never
|
||||
returns to agent-facing strings.
|
||||
routes through ``agent.compression_marker.elide`` / ``elide_middle``; this file pins
|
||||
guard parity and that the real renderers emit a guard-visible marker, never the idiom.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pathlib
|
||||
import json
|
||||
import re
|
||||
import tokenize
|
||||
|
||||
from agent.compression_marker import (
|
||||
_COMPRESSION_MARKER_RE,
|
||||
elide,
|
||||
elide_middle,
|
||||
)
|
||||
|
||||
REPO_ROOT = pathlib.Path(__file__).resolve().parents[2]
|
||||
AGENT_ROOT = REPO_ROOT / "agent"
|
||||
# Agent-facing renderers outside agent/: Slack payload dumps and trajectory training data.
|
||||
SCANNED_PATHS = (
|
||||
*sorted(AGENT_ROOT.rglob("*.py")),
|
||||
*sorted((REPO_ROOT / "plugins" / "platforms" / "slack").rglob("*.py")),
|
||||
REPO_ROOT / "trajectory_compressor.py",
|
||||
from agent.context_compressor import (
|
||||
ContextCompressor,
|
||||
_build_verbatim_user_section,
|
||||
_compact_fallback_turn,
|
||||
_summarize_tool_result,
|
||||
)
|
||||
|
||||
# Any "...[<words> truncated]" variant, not just the bare one (e.g. "...[fallback summary truncated]").
|
||||
IMITABLE_MARKER_RE = re.compile(r"(?:\.\.\.|…)\s?\[[^\]\n]*truncated\]")
|
||||
IMITABLE_MARKER_RE = re.compile(r"(?:\.\.\.|…)\s?\[[^\]\n]*truncat")
|
||||
|
||||
|
||||
def test_minted_marker_is_caught_by_the_dispatch_boundary_guard():
|
||||
@@ -36,19 +32,24 @@ def test_minted_marker_is_caught_by_the_dispatch_boundary_guard():
|
||||
assert _COMPRESSION_MARKER_RE.search(out)
|
||||
|
||||
|
||||
def test_no_imitable_truncation_marker_in_agent_strings():
|
||||
"""The imitable idiom must appear nowhere in agent/ strings — comments only.
|
||||
def test_real_renderers_emit_a_guard_visible_marker_never_the_imitable_idiom():
|
||||
"""Oversized input through the real renderers yields the guarded marker, not the idiom.
|
||||
|
||||
Regression for the open-coded renderers: the marker is minted by the shared
|
||||
helpers now, so any such literal in a string token is a new imitation surface.
|
||||
The active-task line quotes both apostrophes and double quotes so repr() cannot escape
|
||||
the marker's own apostrophe out of the guard's reach.
|
||||
"""
|
||||
offenders = []
|
||||
for path in SCANNED_PATHS:
|
||||
with tokenize.open(str(path)) as fh:
|
||||
for tok in tokenize.generate_tokens(fh.readline):
|
||||
if tok.type != tokenize.STRING:
|
||||
continue
|
||||
for match in IMITABLE_MARKER_RE.finditer(tok.string):
|
||||
rel = path.relative_to(REPO_ROOT)
|
||||
offenders.append(f"{rel}:{tok.start[0]}: {match.group()}")
|
||||
assert not offenders, "imitable truncation markers in agent/ strings:\n" + "\n".join(offenders)
|
||||
quoted = "a'b\"c " * 400
|
||||
clarify = json.dumps({"user_response": "A" * 5000})
|
||||
outputs = {
|
||||
"clarify": _summarize_tool_result("clarify", "{}", clarify),
|
||||
"fallback_turn": _compact_fallback_turn("z " * 5000),
|
||||
"verbatim_user": _build_verbatim_user_section([{"role": "user", "content": "q" * 30000}]),
|
||||
"record": ContextCompressor._bound_oversized_record("r" * 50000, 4000),
|
||||
"active_task": ContextCompressor._latest_user_task_snapshot([{"role": "user", "content": quoted}]),
|
||||
}
|
||||
for name, out in outputs.items():
|
||||
assert out and _COMPRESSION_MARKER_RE.search(out), (name, out[-300:] if out else out)
|
||||
assert not IMITABLE_MARKER_RE.search(out), (name, out[-300:])
|
||||
# A budget too small for marker + content skips the straddler instead of a marker-only quote.
|
||||
fills_budget = [{"role": "user", "content": "u" * 3_998}] * 6 # 23,988 of the 24,000 budget
|
||||
assert "chars omitted" not in _build_verbatim_user_section([{"role": "user", "content": "v" * 500}, *fills_budget])
|
||||
|
||||
Reference in New Issue
Block a user