From dd6229fe11700c39eeaba0c85398cb4b28526f38 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Tue, 22 Sep 2026 15:05:20 +0530 Subject: [PATCH] fix(curator): ledger trim splits rows on the physical newline only MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `_trim_oldest` used `str.splitlines()`, which also breaks on U+2028/U+2029/ U+0085. `append_entry` writes rows with `ensure_ascii=False`, so a row whose evidence contains one of those code points was counted as two lines and rewritten as two malformed physical lines — contradicting "lines in the retained tail are never rewritten". Split on b"\n" (dropping the trailing empty element), re-join with b"\n", and account sizes in bytes. Sibling readers (`compact_ledger`, `gc_blobs`, `list_entries`) use the same idiom pre-existing on main and are left for a follow-up. PROOF: tests/tools/test_skill_ledger.py::test_trim_oldest_when_still_over_cap gains one assertion (a U+2028 row survives a trim byte-for-byte); it fails with the old `splitlines()` body ("At index 85 diff: b'\n' != b'\xe2'") and passes with this change. 23 passed. --- tests/tools/test_skill_ledger.py | 8 ++++++++ tools/skill_ledger.py | 12 ++++++++---- 2 files changed, 16 insertions(+), 4 deletions(-) diff --git a/tests/tools/test_skill_ledger.py b/tests/tools/test_skill_ledger.py index de0218a8be..ae96e79f55 100644 --- a/tests/tools/test_skill_ledger.py +++ b/tests/tools/test_skill_ledger.py @@ -697,3 +697,11 @@ def test_trim_oldest_when_still_over_cap(ledger_env, monkeypatch): assert compactions == [], "an append under the cap must not re-run the sweep" assert skill_ledger.ledger_path().stat().st_size <= 8192 assert "{not json at all" in skill_ledger.ledger_path().read_text(encoding="utf-8") + # U+2028 inside a row (ensure_ascii=False leaves it unescaped) is not a row boundary for the + # trim: a cap that fits only the newest row keeps that row byte-for-byte, not its second half. + u_row = (json.dumps({"id": "u2028", "skill": "my-skill", "action": "edit", + "evidence": {"note": "line one\u2028line two"}}, ensure_ascii=False) + "\n").encode("utf-8") + with open(skill_ledger.ledger_path(), "ab") as fh: + fh.write(u_row) + assert skill_ledger._trim_oldest(len(u_row) + 8) >= 1 + assert skill_ledger.ledger_path().read_bytes() == u_row, "a retained row containing U+2028 survives intact" diff --git a/tools/skill_ledger.py b/tools/skill_ledger.py index 5b5ae39567..5868ade955 100644 --- a/tools/skill_ledger.py +++ b/tools/skill_ledger.py @@ -366,17 +366,21 @@ def _trim_oldest(max_bytes: int) -> int: raw = _read_ledger("trim skipped") if raw is None: return 0 - lines = raw.decode("utf-8").splitlines() - kept: List[str] = [] + # Split on the physical row terminator only: ``str.splitlines`` would also split on + # U+2028/U+2029/U+0085, which ``ensure_ascii=False`` rows may legitimately contain. + lines = raw.split(b"\n") + if lines and lines[-1] == b"": + lines.pop() + kept: List[bytes] = [] size = 0 for line in reversed(lines): # the first iteration always keeps lines[-1]: the newest entry - addition = len(line.encode("utf-8")) + 1 + addition = len(line) + 1 if kept and size + addition > max_bytes: break kept.append(line) size += addition kept.reverse() - data = ("\n".join(kept) + "\n").encode("utf-8") if kept else b"" + data = b"\n".join(kept) + b"\n" if kept else b"" dropped = len(lines) - len(kept) if dropped <= 0: return 0