Adds a Python port of tamaratran/fast-jev-compaction as an eval policy (`engine: jev`): TypeSafe's Jev decision model scores every tool call and result over the whole history and stale ones are dropped or truncated, with no summary and no rewriting of user/assistant text. Transport is OpenRouter's Decisions API (~typesafe/jev-latest). A state that cannot fit the plugin's 25K-token ceiling is recorded as a fallback (the plugin's own behaviour) rather than scored. Every arm now reports what its compaction step cost (calls, tokens, USD — Jev reports cost directly; summary calls are metered through the compressor's call_llm binding and priced at OpenRouter list), and the run writes the harness's own answer/judge token bill so the eval's spend is visible. The question cache key includes --cap-tokens so different caps on one transcript no longer share a bank. Why: measure remaining tokens, compaction cost and recall accuracy of the "decide, don't summarize" approach against current/lean on real Hermes lineages before deciding whether any of it belongs in the compressor.
46 lines
2.2 KiB
Python
46 lines
2.2 KiB
Python
"""Invariants of the fast-jev-compaction eval arm (evals/compaction/jev_arm.py).
|
|
|
|
Offline: the fake asker stands in for Jev. The contract under test is the
|
|
plugin's own: nothing is rewritten, only tool calls/results go, and no tool
|
|
result is ever left without its call (or vice versa).
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from evals.compaction.fixtures import synthetic_transcript, total_tokens
|
|
from evals.compaction.jev_arm import JevCompactor, JevOptions, fake_asker, message_text
|
|
|
|
|
|
def _pairs(messages):
|
|
call_ids = {tc["id"] for m in messages for tc in m.get("tool_calls") or []}
|
|
result_ids = {m["tool_call_id"] for m in messages if m.get("role") == "tool"}
|
|
return call_ids, result_ids
|
|
|
|
|
|
def test_drop_decisions_never_orphan_and_never_rewrite_text():
|
|
msgs = synthetic_transcript(40)
|
|
texts_before = [message_text(m) for m in msgs if m.get("role") != "tool"]
|
|
comp = JevCompactor(asker=fake_asker(keep_call=0.2, keep_result=0.1))
|
|
out = comp.compress(msgs)
|
|
|
|
call_ids, result_ids = _pairs(out)
|
|
assert call_ids == result_ids, "a dropped call must take its result with it, and only its result"
|
|
assert total_tokens(out) < total_tokens(msgs)
|
|
# user/assistant text is preserved verbatim and in order (only tool_calls / tool rows change)
|
|
texts_after = [message_text(m) for m in out if m.get("role") != "tool"]
|
|
assert texts_after == [t for t in texts_before if t.strip()]
|
|
# pinned calls (first row / newest rows) are never candidates
|
|
assert comp.stats["pinned"] >= 1 and comp.stats["calls_dropped"] == comp.stats["calls"] - comp.stats["pinned"]
|
|
|
|
|
|
def test_drop_result_keeps_call_and_bounded_head():
|
|
msgs = synthetic_transcript(30)
|
|
comp = JevCompactor(asker=fake_asker(keep_call=0.9, keep_result=0.1), options=JevOptions(truncate_head_chars=50))
|
|
out = comp.compress(msgs)
|
|
|
|
call_ids, result_ids = _pairs(out)
|
|
assert call_ids == result_ids and len(out) == len(msgs)
|
|
truncated = [m for m in out if m.get("role") == "tool" and "fast-jev-compaction truncated" in m["content"]]
|
|
assert truncated and all(m["content"].startswith("step output") for m in truncated)
|
|
assert all(len(m["content"]) < 300 for m in truncated)
|
|
assert comp.stats["results_dropped"] == len(truncated)
|