fix(curator): ledger trim splits rows on the physical newline only
`_trim_oldest` used `str.splitlines()`, which also breaks on U+2028/U+2029/
U+0085. `append_entry` writes rows with `ensure_ascii=False`, so a row whose
evidence contains one of those code points was counted as two lines and
rewritten as two malformed physical lines — contradicting "lines in the
retained tail are never rewritten". Split on b"\n" (dropping the trailing
empty element), re-join with b"\n", and account sizes in bytes.
Sibling readers (`compact_ledger`, `gc_blobs`, `list_entries`) use the same
idiom pre-existing on main and are left for a follow-up.
PROOF: tests/tools/test_skill_ledger.py::test_trim_oldest_when_still_over_cap
gains one assertion (a U+2028 row survives a trim byte-for-byte); it fails
with the old `splitlines()` body ("At index 85 diff: b'\n' != b'\xe2'") and
passes with this change. 23 passed.
This commit is contained in:
@@ -697,3 +697,11 @@ def test_trim_oldest_when_still_over_cap(ledger_env, monkeypatch):
|
||||
assert compactions == [], "an append under the cap must not re-run the sweep"
|
||||
assert skill_ledger.ledger_path().stat().st_size <= 8192
|
||||
assert "{not json at all" in skill_ledger.ledger_path().read_text(encoding="utf-8")
|
||||
# U+2028 inside a row (ensure_ascii=False leaves it unescaped) is not a row boundary for the
|
||||
# trim: a cap that fits only the newest row keeps that row byte-for-byte, not its second half.
|
||||
u_row = (json.dumps({"id": "u2028", "skill": "my-skill", "action": "edit",
|
||||
"evidence": {"note": "line one\u2028line two"}}, ensure_ascii=False) + "\n").encode("utf-8")
|
||||
with open(skill_ledger.ledger_path(), "ab") as fh:
|
||||
fh.write(u_row)
|
||||
assert skill_ledger._trim_oldest(len(u_row) + 8) >= 1
|
||||
assert skill_ledger.ledger_path().read_bytes() == u_row, "a retained row containing U+2028 survives intact"
|
||||
|
||||
@@ -366,17 +366,21 @@ def _trim_oldest(max_bytes: int) -> int:
|
||||
raw = _read_ledger("trim skipped")
|
||||
if raw is None:
|
||||
return 0
|
||||
lines = raw.decode("utf-8").splitlines()
|
||||
kept: List[str] = []
|
||||
# Split on the physical row terminator only: ``str.splitlines`` would also split on
|
||||
# U+2028/U+2029/U+0085, which ``ensure_ascii=False`` rows may legitimately contain.
|
||||
lines = raw.split(b"\n")
|
||||
if lines and lines[-1] == b"":
|
||||
lines.pop()
|
||||
kept: List[bytes] = []
|
||||
size = 0
|
||||
for line in reversed(lines): # the first iteration always keeps lines[-1]: the newest entry
|
||||
addition = len(line.encode("utf-8")) + 1
|
||||
addition = len(line) + 1
|
||||
if kept and size + addition > max_bytes:
|
||||
break
|
||||
kept.append(line)
|
||||
size += addition
|
||||
kept.reverse()
|
||||
data = ("\n".join(kept) + "\n").encode("utf-8") if kept else b""
|
||||
data = b"\n".join(kept) + b"\n" if kept else b""
|
||||
dropped = len(lines) - len(kept)
|
||||
if dropped <= 0:
|
||||
return 0
|
||||
|
||||
Reference in New Issue
Block a user