From e2990428e7e65053d5aeb4e56ac3fc13680879e5 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 15 Aug 2026 17:40:43 -0700 Subject: [PATCH] docs(evals): ship transcript-building scripts + full eval detail in-repo - scripts/reconstruct_lineage.py: rebuild full uncompacted lineage transcripts from a state.db COPY (descendant-tree walk, content-hash dedupe, synthetic-artifact strip, system_prompts hash resolution) - scripts/replay_lineage.py + scripts/build_html_report.py: replay a 500K prefix through any checkout's compressor and render before/after side-by-side with compaction artifacts color-coded - README: transcript-building workflow, scoping-tripwire section - results/SCORECARD-2026-08-15.md: full per-transcript scorecards, all 4 exam question banks, survival analysis, methodology + caveats --- evals/compaction/README.md | 33 +++ .../results/SCORECARD-2026-08-15.md | 264 ++++++++++++++++++ evals/compaction/scripts/build_html_report.py | 184 ++++++++++++ .../compaction/scripts/reconstruct_lineage.py | 88 ++++++ evals/compaction/scripts/replay_lineage.py | 81 ++++++ 5 files changed, 650 insertions(+) create mode 100644 evals/compaction/scripts/build_html_report.py create mode 100644 evals/compaction/scripts/reconstruct_lineage.py create mode 100644 evals/compaction/scripts/replay_lineage.py diff --git a/evals/compaction/README.md b/evals/compaction/README.md index c4c77ef73f..476fd8296f 100644 --- a/evals/compaction/README.md +++ b/evals/compaction/README.md @@ -29,6 +29,39 @@ Transcripts are NOT committed (they contain real session data). Point `--transcript` at a local file. See `fixtures.py` for the expected shape and a synthetic-transcript generator used by CI smoke tests. +## Building transcripts from real sessions (`scripts/`) + +Compaction rotations mean a single active session rarely exceeds ~300K +tokens, but the *lineage* (parent→children chain) carries the full +uncompacted history. The scripts reconstruct those into eval transcripts: + +```bash +# 1. ALWAYS copy the DB first — never point at the live state.db +cp ~/.hermes/state.db /tmp/state_copy.db + +# 2. Find big lineages (sessions with parent_session_id form chains), then: +python evals/compaction/scripts/reconstruct_lineage.py \ + /tmp/state_copy.db /tmp/lineage.json + +# 3. (optional) Replay a 500K prefix through one checkout's compressor and +# dump before/after for the HTML viewer: +python evals/compaction/scripts/replay_lineage.py /tmp/lineage.json out.json 500000 +python evals/compaction/scripts/build_html_report.py report.html +``` + +`reconstruct_lineage.py` walks the whole descendant tree chronologically, +dedupes rotation-copied rows by content hash, strips synthetic compaction +artifacts (summaries, todo snapshots), and resolves the system prompt through +the `system_prompts` dedup table (sessions only carry a hash). The HTML +report renders before/after transcripts side by side with compaction +artifacts color-coded. + +## Region-scoping tripwire + +`test_region_scoping.py` plants sentinels in head/middle/tail and asserts the +summarizer's serialized-turns input carries ONLY the middle (compacted) +region in both legacy and lean modes. Run it directly or via pytest. + ## Policies Defined in `policies.py`. Each policy maps to `ContextCompressor` constructor diff --git a/evals/compaction/results/SCORECARD-2026-08-15.md b/evals/compaction/results/SCORECARD-2026-08-15.md index 22a8d524b9..98b681c4b8 100644 --- a/evals/compaction/results/SCORECARD-2026-08-15.md +++ b/evals/compaction/results/SCORECARD-2026-08-15.md @@ -51,3 +51,267 @@ lean+recovery 70.0 @ 62K 80.0 @ 41K 43.3 @ 45K 80.0 @ 50K 68. Ship lean as opt-in (compression.tail_mode: lean, legacy default), harness as the permanent gate. Iterate prmerge-class recall behind the flag (query mining, per-epoch anchor windows) before default flip. + + +## Appendix: full per-transcript detail + +### Transcript: sweep + +| policy | recall | tokens before | tokens after | compress s | +|---|---|---|---|---| +| lean | 40.0% | 499,625 | 61,567 | 114.9 | +| lean+recovery | 70.0% | 499,625 | 61,792 | 114.8 | + +
15 exam questions (questions-30b95351c7.json) + +1. **What is the reason given for never using 'git checkout pr-branch -- ' on stale branches?** + gold: `the stale file version silently deletes newer main code` +2. **According to the transcript, how much RSS memory does the gateway balloon to every ~2h in the regression reported in issue #81625?** + gold: `~60GB` +3. **Which specific Electron setting is suspected of causing the Windows occlusion freeze in issue #83420?** + gold: `backgroundThrottling` +4. **What exact error message is returned when 'gh pr merge --auto' is attempted on the NousResearch/hermes-agent repository?** + gold: `Auto merge is not allowed for this repository (enablePullRequestAutoMerge)` +5. **What is the specified 'Rule 0' that must be included in a subagent brief?** + gold: `load the skill first` +6. **In the July 2026 title-cluster sweep, what was the title of the missed first submitter PR #35416?** + gold: `add config gate for title generation` +7. **Which file path is noted as containing the #34034/#28149 manifest guard 'test_bundled_plugin_manifests_ship_in_both_wheel_and_sdist'?** + gold: `tests/test_packaging_metadata.py` +8. **What was the result of the 'npm ci' command run in /home/teknium/salv-desktop according to the background process notification?** + gold: `completed normally (exit code 0)` +9. **What was the 'Root Cause A' identified for why 'uv sync --extra all --locked' failed daily in issue #79434?** + gold: `relative exclude-newer makes the committed lock stale every day` +10. **How many tasks are reported as done in the 'fangliquanflq' desktop retry truncation PR #86605?** + gold: `13` +11. **In the 'salv-cron' worktree, what was the exit code when the agent tried to execute a 'BLOCKED (hardline)' command?** + gold: `-1` +12. **What is the full title block text for the technical schematic infographic generated for the Gateway Drain?** + gold: `GATEWAY DRAIN × CRON — SHUTDOWN CONTRACT` +13. **Which PR number's watcher reported '=== ALL GREEN (streak=1, checks=46) ===' at [03:56:19]?** + gold: `82980` +14. **What is the specific Gist ID created for the PR infographic host in the cron cluster?** + gold: `ee33edd5804689243f974536ef7aecb9` +15. **What was the final merge SHA for Cluster D's Trigger-now PR #70638?** + gold: `f9d64b9a9d8b306f64851c1a13869d96ad5d7869` + +
+ +
15 exam questions (questions-5be475cde0.json) + +1. **What exact command did the agent use to search for open issues related to a specific topic during Phase 1 of the cluster-sweep salvage?** + gold: `gh issue list --search "" --state open --limit 100 --json number,title` +2. **According to Teknium's design intent, what is the status of 'platform toolsets' in the codebase?** + gold: `platform toolsets are vestigial, never exposed` +3. **During the July sweep, which specific issue's config bridge was found to already exist at the exact line it was claimed to be missing?** + gold: `#32263` +4. **In the Aug 2026 cron-summarizer cluster sweep, which two PR numbers were discovered post-merge as the true first submitters?** + gold: `#60593, #61969` +5. **What is the recommended Git command to find when a specific symbol fix landed on the main branch?** + gold: `git log -S ""` +6. **Why did the #39719 salvage silently delete 236 lines of code from cli-config.yaml.example?** + gold: `the stale file version silently deletes newer main code` +7. **What is the rule for salvaging commits with placeholder identities like 'pwn@example.com'?** + gold: `do NOT cherry-pick. Surgical reapply as maintainer-authored commit, Co-authored-by the GitHub PR author` +8. **How should an agent handle a 'gh pr merge' 502 error?** + gold: `retry the same command once after the "Merge already in progress" settles (~45s); check PR state between attempts` +9. **Which two properties shape almost every design decision in Hermes according to the Development Guide?** + gold: `Per-conversation prompt caching is sacred and The core is a narrow waist; capability lives at the edges.` +10. **What error message does the live-checkout git guard display when blocking a history-rewriting command?** + gold: `Blocked: `git ` would rewrite Hermes's live source checkout (/home/teknium/.hermes/hermes-agent) and can mix module ` +11. **What happened to the Desktop cluster's 'npm ci' command that resulted in an error writing to /tmp/ccH06T4r.s?** + gold: `No space left on device` +12. **What was the GraphQL API rate limit remaining for the user when the 'API rate limit already exceeded' error first occurred?** + gold: `0` +13. **Which PR was identified as the salvage of HexLab98's #85283 to fix hung inline API calls?** + gold: `#86645` +14. **Why did PR #79268 fix invisible overlays in the TUI?** + gold: `renderNodeToOutput skips boxes Yoga squeezes to height 0` +15. **What was the specific ModuleNotFoundError message caused by the wheel subpackage discovery trap in #34701?** + gold: `ModuleNotFoundError: No module named 'hermes_cli.dashboard_auth'` + +
+ +### Transcript: gui + +| policy | recall | tokens before | tokens after | compress s | +|---|---|---|---|---| +| lean | 60.0% | 499,818 | 41,232 | 118.1 | +| lean+recovery | 80.0% | 499,818 | 41,306 | 115.2 | + +
15 exam questions (questions-36d3d87e0b.json) + +1. **What is the PR number for the authored fix addressing mid-turn message ordering bugs in Hermes Desktop?** + gold: `#86617` +2. **According to the contribution rubric in AGENTS.md, which type of config belongs in '.env' and which belongs in 'config.yaml'?** + gold: `.env is for secrets only (API keys, tokens, passwords). All behavioral settings... go in config.yaml.` +3. **What specific file and line number were identified as the cause of an AssertionError (assert 56 == 55) in the Python tests?** + gold: `tests/hermes_cli/test_session_recovery_lost_and_found.py:327` +4. **What was the root cause of issue #73793 regarding mid-turn message rendering?** + gold: `redirect/steer paths spliced the mid-turn user bubble BEFORE the active assistant stream row` +5. **Which PR was verified to already be on 'main', resulting in nothing needing to be salvaged for it?** + gold: `#84287` +6. **In the Desktop virtualized-scrolling cluster, what was the fix for issue #79157 (scrollbar unclickable)?** + gold: `pane sash grab band made asymmetric 1px/7px` +7. **Which contributor's email was mapped to 'baihemax' during the attribution audit of PR #86588?** + gold: `602028@ky-tech.com.cn` +8. **What error message does the Hermes terminal tool return when a git command is blocked to prevent rewriting the live source checkout?** + gold: `Blocked: `git ` would rewrite Hermes's live source checkout` +9. **What is the core design principle regarding 'Narrow Waist' in Hermes development?** + gold: `The core is a narrow waist; capability lives at the edges.` +10. **What was the result of the rebase-merge attempt for PR #86589?** + gold: `GraphQL: Pull Request has merge conflicts (mergePullRequest)` +11. **In the infographic style picker, what vibe is associated with the 'designers-republic' style?** + gold: `The Designers Republic: flat orange+violet vector schematic on pewter grey` +12. **Why was PR #76286 excluded from the compaction/compression transcript-visibility cluster?** + gold: `conflicts with main in 4 files and introduces a second competing display-dedupe scheme` +13. **What is the 'Provenance note' date for the pr-infographic-workflow.md reference file?** + gold: `May 23 2026` +14. **What specific TypeScript error caused PR #86772 to fail CI linting after a rebase?** + gold: `Property 'onToggleUnread' is missing in type` +15. **According to the Desktop Engineering Guide, who is the authority for process lifecycle and the native filesystem?** + gold: `Electron` + +
+ +
15 exam questions (questions-9c55c707b6.json) + +1. **What two PR numbers are associated with the 'sidebar-nav-rows-and-overlay-panels.md' and 'hud-mode-internals.md' references in the initial tool content?** + gold: `#85162 and #82285` +2. **According to AGENTS.md, what is the 'one exception' to the rule that nothing should rebuild the system prompt mid-conversation?** + gold: `context compression` +3. **In the Contribution Rubric, what are the three allowed reasons for an automated triage sweeper to close a PR?** + gold: `implemented_on_main, cannot_reproduce, incoherent` +4. **Which contributor is credited with adding the 'Brazilian Portuguese localization' in PR #86292?** + gold: `@gui8515` +5. **What specific error message is reported in issue #83562 regarding the Windows Desktop update?** + gold: `Hermes backend exited (0)` +6. **What is the 'core problem' identified in the parallel-subagent-salvage-orchestration.md reference?** + gold: `subagents share the parent's worktree + main checkout` +7. **Why was the 'nix (macos-latest)' build failing in the salvage batches according to the orchestration reference?** + gold: `Nix build failed due to stale npm lockfile hash` +8. **Which subagent ID was assigned the goal of salvaging the 'inflight-journal duplicate-answer cluster'?** + gold: `sa-2-7318d0ba` +9. **In PR #86595, why was PR #80707 by upperagent excluded from the salvage?** + gold: `violating this PR's UI-read-only invariant` +10. **What was the root cause of the failure in Python tests slice 4/12 for PR #86597?** + gold: `AssertionError: assert 't2' == 't1'` +11. **What did the fix for issue #79157 in PR #86589 involve?** + gold: `pane sash grab band made asymmetric 1px/7px` +12. **According to the root cause analysis for #73793, which two files spliced the mid-turn user message at streamIndex?** + gold: `use-prompt-actions/index.ts and session-tile-actions.ts` +13. **What was the head SHA for the 'salvage/desktop-busy-state' branch in PR #86604?** + gold: `bddadfe9e21e24b3d52e2b15f138c42474dede42` +14. **Why was the merge of PR #86589 aborted during the 'Merge all' command?** + gold: `GraphQL: Pull Request has merge conflicts (mergePullRequest)` +15. **What specific file was modified to fix the 'artifacts page timestamps render 1970' issue via PR #86749?** + gold: `apps/desktop/src/app/session/hooks/use-session-actions/utils.ts` + +
+ +### Transcript: prmerge + +| policy | recall | tokens before | tokens after | compress s | +|---|---|---|---|---| +| uncompacted_control | 96.7% | 499,663 | 499,663 | — | +| current | 33.3% | 499,663 | 155,399 | 14.9 | +| lean | 23.3% | 499,663 | 44,419 | 105.4 | +| lean+recovery | 43.3% | 499,663 | 44,977 | 95.8 | + +
15 exam questions (questions-703ae2774a.json) + +1. **Which PR number added the public subagent lifecycle API?** + gold: `#63359` +2. **What is the name of the typed service added to PluginContext for launching and monitoring child sessions?** + gold: `subagent_lifecycle` +3. **How many contract and security tests were included with the subagent lifecycle API PR?** + gold: `42` +4. **What specific gap was identified regarding the `ctx.inject_message()` function in gateway sessions?** + gold: `cannot currently trigger a turn in an existing gateway session` +5. **Which PR implements gateway-safe plugin injection by extending `ctx.inject_message()` with a keyword-only `session_key`?** + gold: `#64436` +6. **What are the two specific constraints placed on redaction patterns in the pattern registry to prevent exposing data?** + gold: `must compile, must start with ≥2 literal characters` +7. **Which contributor authorized sustained help for the Phase 0–1 expansion track?** + gold: `Daniel` +8. **What is the issue number for the disposition gap concerning `pre_command` middleware and MCP tool access?** + gold: `#64204` +9. **What configuration setting is required to opt-in to reasoning deltas in streaming output?** + gold: `plugins.stream_reasoning_deltas: true` +10. **How many additions and across how many files were made in PR #63359?** + gold: `650 additions across 4 files` +11. **What is the name of the reference plugin shipped with the redaction pattern registry?** + gold: `nvapi-redaction` +12. **List the four observer-only streaming output plugin hooks added in PR #64317.** + gold: `on_stream_start, on_stream_delta, on_stream_end, on_interim_message` +13. **What was addressed in the update to PR #58541 regarding lifecycle hooks?** + gold: `created-hook timing and added kanban_task_promoted` +14. **Which sub-issue number is associated with the 'developer tooling' (scaffold + Plugin Doctor + test harness)?** + gold: `#64230` +15. **What was the Round 3 review's outcome for PR #63359 and @asimons81?** + gold: `sub-issue #65447` + +
+ +### Transcript: acp + +| policy | recall | tokens before | tokens after | compress s | +|---|---|---|---|---| +| uncompacted_control | 100.0% | 498,906 | 498,906 | — | +| current | 30.0% | 498,906 | 160,223 | 15.8 | +| lean | 36.7% | 498,906 | 49,523 | 143.3 | +| lean+recovery | 80.0% | 498,906 | 49,721 | 135.6 | + +
15 exam questions (questions-f45358df19.json) + +1. **What was the specific reason Teknium gave for reverting PR #30179 in July 2026?** + gold: `WTF??? REVERT! DAMMIT` +2. **On which specific PR did Teknium say, 'tf are you saying to me. Stop giving me such random verbose details'?** + gold: `PR #6391` +3. **Which file path should be checked for the canonical list of provider models?** + gold: `hermes_cli/models.py` +4. **What was the identified bug in PR #2314 regarding provider names?** + gold: `checking for "alibaba-coding-plan"` +5. **What is the mandatory line limit for PR reviews requested by Teknium?** + gold: `<= 15 lines` +6. **What exact error message did the agent receive when attempting to checkout a worktree while in the live source directory?** + gold: `Blocked: `git checkout` would rewrite Hermes's live source checkout (/home/teknium/.hermes/hermes-agent) and can mix mod` +7. **Why was PR #74658 necessary to fix Slack 'broken on main'?** + gold: `SlackResponse isn't a dict subclass, so every gate is always False.` +8. **What was the final merge commit SHA for the Slack SDK response fix on main?** + gold: `24ba86627515ad5fda69a39ef338c365713448bc` +9. **In the 'Pop-laboratory' style infographic for the Auxiliary Client fix, what were the two specific outcomes shown in cell 2?** + gold: `Messages wrapper keeps /anthropic and OpenAI fallback keeps /v1` +10. **What specific SQL update was added to the migration path in hermes_cli/kanban_db.py to prevent losing active wake on upgrade?** + gold: `UPDATE kanban_notify_subs SET delivery_mode = 'notify+wake' WHERE platform != 'tui'` +11. **Which test failed in CI slice 5/12 for the kanban delivery modes PR?** + gold: `tests/gateway/test_kanban_notifier_apiserver_wake.py::test_apiserver_sub_wakes_real_session_via_self_post` +12. **According to the transcript, why is squash merging banned as of July 2026?** + gold: `DevOps policy` +13. **Which contributor authored the first fix for issue #73030 in July?** + gold: `@Tranquil-Flow` +14. **What was the 'Superman-style' shield error in the first generation of the Kanban infographic?** + gold: `red "S" inside the diamond shield` +15. **What specific file was modified to add the 'scope_id_for_chat' method for Slack?** + gold: `plugins/platforms/slack/adapter.py` + +
+ +## Methodology notes + +- Transcripts: 4 real session lineages reconstructed from a state.db copy + (sweep campaign 42 rotations / GUI desktop 34 / PR-merge 17 / ACP review + 17), chronological 500K-token prefix, tool-group aligned. +- Question generation: main model, from the region the CURRENT policy would + summarize (most conservative boundary), cached per transcript so every + policy answers the identical exam. +- Answering: fresh LLM sees ONLY the post-compaction context (closed-book) or + context + one FTS5+BM25 search round-trip over the archived region + (+recovery). Judge sees gold; answerer never does. Scoring 2/1/0. +- Known caveats: 15 questions/transcript => +-1 question ~ 3.3pts noise; + sweep/gui current-policy rows predate a question-bank regeneration + (prmerge/acp are same-bank across all arms); the recovery sim conservatively + approximates production session_search (same engine, no windowing). +- Cost shape: lean compaction = ~25 aux-model digest calls (~2min, one-time + per compaction) vs 1 call today; every post-compaction turn is ~110K input + tokens cheaper. Break-even ~1 turn. diff --git a/evals/compaction/scripts/build_html_report.py b/evals/compaction/scripts/build_html_report.py new file mode 100644 index 0000000000..6a39d936e8 --- /dev/null +++ b/evals/compaction/scripts/build_html_report.py @@ -0,0 +1,184 @@ +#!/usr/bin/env python3 +"""Build a self-contained HTML report comparing compaction runs. + +Usage: build_report.py +Expects runs/_.json pairs from run_compaction.py. +""" +import html +import json +import sys +from pathlib import Path + +RUNS = Path(sys.argv[1]) +OUT = sys.argv[2] + +pairs = {} +for f in sorted(RUNS.glob("*.json")): + co, sid = f.stem.split("_", 1) + if co == "main": + co, sid = "main-co", f.stem[len("main-co_"):] + elif co == "pr": + co, sid = "pr-co", f.stem[len("pr-co_"):] + data = json.loads(f.read_text(encoding="utf-8")) + pairs.setdefault(sid, {})[co] = data + +E = html.escape + +def msg_class(m): + role = m.get("role", "?") + c = m.get("content") or "" + if isinstance(c, str): + if "[CONTEXT COMPACTION" in c or "[CONTEXT SUMMARY" in c: + return "summary" + if "SKILL_PRUNED" in c: + return "skillpruned" + if "SKILL POLICY DIGEST" in c or "SKILL_POLICY_DIGEST" in c: + return "digest" + if "preserved across context compression" in c: + return "todosnap" + return role + +def render_msg(m, idx): + role = m.get("role", "?") + c = m.get("content") + if not isinstance(c, str): + c = json.dumps(c, default=str)[:2000] + tool = m.get("tool_name") or "" + tcs = m.get("tool_calls") or [] + tc_names = ", ".join( + (t.get("function", {}) or {}).get("name", "?") for t in tcs if isinstance(t, dict) + ) + cls = msg_class(m) + nchars = len(c) + label = role + if tool: + label += f" · {tool}" + if tc_names: + label += f" → {tc_names}" + preview = c[:180].replace("\n", " ") + full = c if nchars <= 20000 else c[:20000] + f"\n…[{nchars-20000:,} more chars]" + return ( + f'
#{idx}' + f'{E(label)}' + f'{nchars:,}ch' + f'{E(preview)}' + f"
{E(full)}
" + ) + +def render_column(title, data, key): + meta = data["meta"] + msgs = data[key] + body = "".join(render_msg(m, i) for i, m in enumerate(msgs)) + return ( + f'

{E(title)}

' + f'
{meta[key.replace("before","before_msgs").replace("after","after_msgs")] if False else len(msgs)} msgs · ' + f'~{(meta["before_tokens_est"] if key=="before" else meta["after_tokens_est"]):,} tok
' + f'
{body}
' + ) + +def survival_stats(before, after): + after_texts = set() + for m in after: + c = m.get("content") + if isinstance(c, str) and c: + after_texts.add(c[:400]) + kept = sum(1 for m in before if isinstance(m.get("content"), str) and (m.get("content") or "")[:400] in after_texts) + return kept + +sections = [] +toc = [] +for sid, versions in pairs.items(): + if "main-co" not in versions or "pr-co" not in versions: + continue + main_d, pr_d = versions["main-co"], versions["pr-co"] + title = main_d["meta"].get("title") or sid + mm, pm = main_d["meta"], pr_d["meta"] + + def count_markers(msgs, needle): + return sum((m.get("content") or "").count(needle) for m in msgs if isinstance(m.get("content"), str)) + + rows = [] + def stat(name, mv, pv): + cls = "diff" if mv != pv else "" + rows.append(f"{E(name)}{E(str(mv))}{E(str(pv))}") + + stat("Messages after", mm["after_msgs"], pm["after_msgs"]) + stat("Est. tokens after", f"{mm['after_tokens_est']:,}", f"{pm['after_tokens_est']:,}") + stat("Reduction", f"{100-100*mm['after_tokens_est']//max(1,mm['before_tokens_est'])}%", f"{100-100*pm['after_tokens_est']//max(1,pm['before_tokens_est'])}%") + stat("Compress time", f"{mm['elapsed_s']}s", f"{pm['elapsed_s']}s") + stat("SKILL_PRUNED markers", count_markers(main_d["after"], "SKILL_PRUNED"), count_markers(pr_d["after"], "SKILL_PRUNED")) + stat("Policy digest blocks", count_markers(main_d["after"], "SKILL POLICY DIGEST") + count_markers(main_d["after"], "SKILL_POLICY_DIGEST"), count_markers(pr_d["after"], "SKILL POLICY DIGEST") + count_markers(pr_d["after"], "SKILL_POLICY_DIGEST")) + stat("Todo snapshot present", "yes" if count_markers(main_d["after"], "preserved across context compression") else "no", "yes" if count_markers(pr_d["after"], "preserved across context compression") else "no") + stat("Kept-verbatim msgs", survival_stats(main_d["before"], main_d["after"]), survival_stats(pr_d["before"], pr_d["after"])) + stat("Summary error", mm.get("summary_error") or "—", pm.get("summary_error") or "—") + + todo_html = "" + for label, d in (("main", mm), ("PR #87090", pm)): + tb = d.get("todo_injection_block") + if tb: + todo_html += f"

Todo injection block — {E(label)}

{E(tb)}
" + + anchor = f"s-{sid}" + toc.append(f'{E(title)} ({sid})') + sections.append(f""" +
+

{E(title)} {sid}

+{"".join(rows)}
mainPR #87090
+{todo_html} +
+{render_column("BEFORE (original transcript)", main_d, "before")} +{render_column("AFTER — main", main_d, "after")} +{render_column("AFTER — PR #87090", pr_d, "after")} +
+
""") + +page = f""" +Compaction comparison — main vs PR #87090 + +

Compaction comparison — current main (7619564fb) vs PR #87090 (41fd511f6)

+

Real sessions from state.db (copy), replayed through each checkout's ContextCompressor with force=True. Real LLM summaries. Click any row to expand the full message.

+
+compaction summary +SKILL_PRUNED marker +policy digest +todo snapshot +user +
+ +{"".join(sections)} +""" + +Path(OUT).write_text(page, encoding="utf-8") +print(f"wrote {OUT} ({len(page):,} bytes, {len(sections)} sessions)") diff --git a/evals/compaction/scripts/reconstruct_lineage.py b/evals/compaction/scripts/reconstruct_lineage.py new file mode 100644 index 0000000000..5006c5e3c8 --- /dev/null +++ b/evals/compaction/scripts/reconstruct_lineage.py @@ -0,0 +1,88 @@ +#!/usr/bin/env python3 +"""Reconstruct the full uncompacted transcript of a session LINEAGE. + +Rotation children start with a copy of the compressed parent (head + summary + +tail). To rebuild the real history: walk the chain root->leaf, append messages +not seen before (hash of role+content+tool_calls), and skip synthetic +compaction summaries / todo snapshots so we get the organic transcript. + +Usage: reconstruct_lineage.py + +ALWAYS run against a COPY of state.db, never the live file. +""" +import hashlib +import json +import sqlite3 +import sys + +DB = sys.argv[1] +ROOT = sys.argv[2] +OUT = sys.argv[3] + +db = sqlite3.connect(DB) +db.row_factory = sqlite3.Row + +# collect the whole descendant tree, chronological by started_at +import collections +children = collections.defaultdict(list) +for r in db.execute( + "SELECT id, parent_session_id FROM sessions WHERE parent_session_id IS NOT NULL" +): + children[r["parent_session_id"]].append(r["id"]) +chain = [] +frontier = [ROOT] +while frontier: + sid = frontier.pop(0) + chain.append(sid) + frontier.extend(children.get(sid, [])) +starts = {r["id"]: r["started_at"] or "" for r in db.execute( + f"SELECT id, started_at FROM sessions WHERE id IN ({','.join('?'*len(chain))})", chain)} +chain.sort(key=lambda s: starts.get(s, "")) +print(f"chain: {len(chain)} sessions") + +SYNTH_MARKERS = ( + "[CONTEXT COMPACTION", "[CONTEXT SUMMARY", "[PRIOR CONTEXT", + "preserved across context compression", +) + +seen = set() +out = [] +sysprompt = None +for sid in chain: + if sysprompt is None: + row = db.execute( + "SELECT s.system_prompt, sp.prompt AS dedup_prompt FROM sessions s " + "LEFT JOIN system_prompts sp ON sp.hash = s.system_prompt_hash " + "WHERE s.id=?", (sid,)).fetchone() + if row: + sysprompt = row["system_prompt"] or row["dedup_prompt"] or None + for r in db.execute( + "SELECT * FROM messages WHERE session_id=? ORDER BY id", (sid,) + ): + c = r["content"] or "" + if any(m in c for m in SYNTH_MARKERS): + continue # synthetic compaction artifact, not organic history + h = hashlib.md5( + (r["role"] + "\x00" + c + "\x00" + (r["tool_calls"] or "")).encode( + "utf-8", "replace") + ).hexdigest() + if h in seen: + continue + seen.add(h) + m = {"role": r["role"], "content": c} + if r["tool_calls"]: + try: + m["tool_calls"] = json.loads(r["tool_calls"]) + except Exception: + pass + if r["tool_call_id"]: + m["tool_call_id"] = r["tool_call_id"] + if r["tool_name"]: + m["tool_name"] = r["tool_name"] + out.append(m) + +msgs = [{"role": "system", "content": sysprompt or ""}] + out +chars = sum(len(m.get("content") or "") + len(json.dumps(m.get("tool_calls", ""), default=str)) for m in msgs) +print(f"reconstructed: {len(msgs)} msgs, {chars:,} chars (~{chars//4:,} tok)") +json.dump({"root": ROOT, "chain": chain, "messages": msgs}, open(OUT, "w", encoding="utf-8"), default=str) +print(f"wrote {OUT}") diff --git a/evals/compaction/scripts/replay_lineage.py b/evals/compaction/scripts/replay_lineage.py new file mode 100644 index 0000000000..8b9cf0b929 --- /dev/null +++ b/evals/compaction/scripts/replay_lineage.py @@ -0,0 +1,81 @@ +#!/usr/bin/env python3 +"""Replay a ~500K-token prefix of a reconstructed lineage through compaction. + +Usage: run_lineage_compaction.py [cap_tokens] + +Takes the chronological prefix of the lineage at the token cap (default 500K = +the 50% trigger on a 1M-context model), aligned to a tool-group boundary, and +runs ContextCompressor.compress() exactly as the live trigger would. +""" +import copy +import json +import sys +import time +from pathlib import Path + +CHECKOUT = sys.argv[1] +LINEAGE = sys.argv[2] +OUT = sys.argv[3] +CAP = int(sys.argv[4]) if len(sys.argv) > 4 else 500_000 + +sys.path.insert(0, CHECKOUT) + +data = json.load(open(LINEAGE, encoding="utf-8")) +msgs = data["messages"] + +def tok(m): + t = len(m.get("content") or "") // 4 + tc = m.get("tool_calls") + if tc: + t += len(json.dumps(tc, default=str)) // 4 + return t + +# chronological prefix up to CAP tokens +prefix = [] +total = 0 +for m in msgs: + t = tok(m) + if total + t > CAP and len(prefix) > 10: + break + prefix.append(m) + total += t + +# align the end: never end on an assistant msg with tool_calls whose results +# were cut off; drop trailing orphans +while prefix and prefix[-1].get("tool_calls"): + prefix.pop() +# also drop trailing tool results with no preceding assistant tool_calls kept +# (compress()'s _sanitize_tool_pairs would handle it, but keep input clean) + +before_tokens = sum(tok(m) for m in prefix) +print(f"[{Path(CHECKOUT).name}] {Path(LINEAGE).stem}: prefix {len(prefix)} msgs ~{before_tokens:,} tok (cap {CAP:,})") + +from agent.context_compressor import ContextCompressor # noqa: E402 + +model = "anthropic/claude-fable-5" +comp = ContextCompressor(model=model, quiet_mode=True) +before = copy.deepcopy(prefix) +t0 = time.time() +compressed = comp.compress(prefix, current_tokens=before_tokens, force=True) +dt = time.time() - t0 +after_tokens = sum(tok(m) for m in compressed) +print(f" -> {len(compressed)} msgs ~{after_tokens:,} tok in {dt:.1f}s (err={getattr(comp,'_last_summary_error',None)})") + +json.dump({ + "meta": { + "checkout": Path(CHECKOUT).name, + "session_id": data["root"], + "title": f"lineage {data['root']} ({len(data['chain'])} rotations)", + "model": model, + "elapsed_s": round(dt, 1), + "before_msgs": len(before), + "after_msgs": len(compressed), + "before_tokens_est": before_tokens, + "after_tokens_est": after_tokens, + "summary_error": getattr(comp, "_last_summary_error", None), + "todo_injection_block": None, + }, + "before": before, + "after": compressed, +}, open(OUT, "w", encoding="utf-8"), default=str) +print(f" wrote {OUT}")