fix: ranged @file read bails at the char budget mid-line (review follow-up)

_next_line collected every readline(line_cap) piece until the newline, so a
one-line giant (minified JSON) was still fully materialized before the
total_chars > char_budget gate. Pass the remaining budget into the helper and
stop collecting as soon as it is exceeded; the caller returns the oversized
block and never needs the rest of the line.
This commit is contained in:
teknium1
2026-09-20 01:03:55 -07:00
committed by Teknium
parent 0accda1f76
commit 6d1341a13b
2 changed files with 21 additions and 5 deletions

View File

@@ -307,8 +307,11 @@ def _expand_path_reference(ref: ContextReference, cwd: Path, *, allowed_root: Pa
char_budget = None if max_inline_tokens is None else max_inline_tokens * CHARS_PER_TOKEN
line_cap = None if char_budget is None else char_budget + 1
def _next_line(fh, collect: bool) -> str | None:
pieces, seen = [], False
def _next_line(fh, collect: bool, remaining: int | None = None) -> str | None:
"""One line ("" when skipping), None at EOF. While collecting, stop as soon as the
pieces exceed ``remaining``: the caller returns the oversized block and never needs
the rest of a giant line, so it is never materialized."""
pieces, seen, collected = [], False, 0
while True:
piece = fh.readline() if line_cap is None else fh.readline(line_cap)
if not piece:
@@ -316,6 +319,9 @@ def _expand_path_reference(ref: ContextReference, cwd: Path, *, allowed_root: Pa
seen = True
if collect:
pieces.append(piece)
collected += len(piece)
if remaining is not None and collected > remaining:
break
if piece.endswith("\n"):
break
return ("".join(pieces) if collect else "") if seen else None
@@ -326,7 +332,7 @@ def _expand_path_reference(ref: ContextReference, cwd: Path, *, allowed_root: Pa
if _next_line(fh, collect=False) is None:
break
for _ in range((ref.line_end or ref.line_start) - ref.line_start + 1):
line = _next_line(fh, collect=True)
line = _next_line(fh, collect=True, remaining=None if char_budget is None else char_budget - total_chars)
if line is None:
break
total_chars += len(line)

View File

@@ -246,7 +246,7 @@ def test_line_range_bounds_reads_on_single_line_file(tmp_path: Path, monkeypatch
payload = tmp_path / "oneline.txt"
payload.write_text("x" * 100_000, encoding="utf-8") # one giant line, no newline
read_sizes = []
read_sizes, returned = [], []
orig_open = Path.open
def _counting_open(self, *args, **kwargs):
@@ -254,7 +254,14 @@ def test_line_range_bounds_reads_on_single_line_file(tmp_path: Path, monkeypatch
mode = args[0] if args else kwargs.get("mode", "r")
if "b" not in mode:
orig_readline = fh.readline
fh.readline = lambda *a, **k: (read_sizes.append(a[0] if a else -1), orig_readline(*a, **k))[1]
def _readline(*a, **k):
read_sizes.append(a[0] if a else -1)
piece = orig_readline(*a, **k)
returned.append(len(piece))
return piece
fh.readline = _readline
return fh
monkeypatch.setattr(Path, "open", _counting_open)
@@ -264,6 +271,9 @@ def test_line_range_bounds_reads_on_single_line_file(tmp_path: Path, monkeypatch
assert "too large to inline safely" in result.message
# hard_limit = 500 tokens -> char budget 2000 -> each readline bounded at 2001
assert read_sizes and max(read_sizes) <= 2001
# ... and the giant line is never materialized: reading stops once the budget is exceeded
# instead of collecting every 2001-char piece up to the newline (review follow-up).
assert sum(returned) <= 2 * 2001
def test_run_quiet_caps_child_output(tmp_path: Path):