Validation treated every read_file_raw error as "the path is free" for an Add target and a Move destination; only _apply_add checked not_found. With no byte transport (native reads off, no base64 or od) an `Add new-src` then `Move new-src -> dst` validated, and the mv replaced the existing dst while the patch reported success. _occupied() frees a path only on not_found (overlay-aware), and _apply_move re-checks the destination right before mv, since validation's answer is stale once earlier ops of the patch have applied.
453 lines
22 KiB
Python
453 lines
22 KiB
Python
"""V4A patch parser/applier (codex, cline). ``*** Begin Patch``/``*** End Patch`` wrap ops:
|
|
``*** Update File: p`` + hunks (``@@ hint @@``, `` ctx``, ``-old``, ``+new``); ``*** Add File: n``
|
|
+ ``+`` lines; ``*** Delete File: o``; ``*** Move File: a -> b``. Entry points:
|
|
``parse_v4a_patch(text) -> (ops, error)`` and ``apply_v4a_operations(ops, file_ops)``."""
|
|
|
|
import contextlib
|
|
import difflib
|
|
import inspect
|
|
import re
|
|
from dataclasses import dataclass, field
|
|
from enum import Enum
|
|
from typing import TYPE_CHECKING, Any, Callable, Dict, List, Optional, Tuple
|
|
|
|
if TYPE_CHECKING: # annotations only; the real import is per-call in apply_v4a_operations
|
|
from tools.file_operations_common import PatchResult
|
|
|
|
from tools.file_operations_common import PatchResult
|
|
|
|
|
|
class OperationType(Enum):
|
|
ADD = "add"
|
|
UPDATE = "update"
|
|
DELETE = "delete"
|
|
MOVE = "move"
|
|
|
|
|
|
@dataclass
|
|
class HunkLine:
|
|
prefix: str # ' ', '-', or '+'
|
|
content: str
|
|
|
|
|
|
@dataclass
|
|
class Hunk:
|
|
context_hint: Optional[str] = None
|
|
lines: List[HunkLine] = field(default_factory=list)
|
|
|
|
|
|
@dataclass
|
|
class PatchOperation:
|
|
operation: OperationType
|
|
file_path: str
|
|
new_path: Optional[str] = None # MOVE only
|
|
hunks: List[Hunk] = field(default_factory=list)
|
|
|
|
|
|
# Markers must occupy the whole line at column 0 so content lines that merely
|
|
# mention the format ("+*** End Patch") can't truncate or reset the patch.
|
|
_BEGIN_MARKER = re.compile(r'^\*\*\*\s*Begin\s+Patch\s*$')
|
|
_END_MARKER = re.compile(r'^\*\*\*\s*End\s+Patch\s*$')
|
|
_OP_MARKERS: List[Tuple[OperationType, re.Pattern]] = [
|
|
(OperationType.UPDATE, re.compile(r'\*\*\*\s*Update\s+File:\s*(.+)')),
|
|
(OperationType.ADD, re.compile(r'\*\*\*\s*Add\s+File:\s*(.+)')),
|
|
(OperationType.DELETE, re.compile(r'\*\*\*\s*Delete\s+File:\s*(.+)')),
|
|
(OperationType.MOVE, re.compile(r'\*\*\*\s*Move\s+File:\s*(.+?)\s*->\s*(.+)'))]
|
|
_HINT_RE = re.compile(r'@@\s*(.+?)\s*@@')
|
|
|
|
|
|
def parse_v4a_patch(patch_content: str) -> Tuple[List[PatchOperation], Optional[str]]:
|
|
"""-> ``(operations, None)`` (empty patch = ``[]``, no error) or ``([], "Parse error: …")``."""
|
|
# Tolerate CRLF: a stray ``\r`` would land in every HunkLine.content and defeat the markers.
|
|
lines = [ln[:-1] if ln.endswith('\r') else ln for ln in patch_content.split('\n')]
|
|
start_idx = -1 # parse from the top when no Begin marker is present
|
|
end_idx = len(lines)
|
|
for i, line in enumerate(lines):
|
|
if _BEGIN_MARKER.match(line):
|
|
start_idx = i
|
|
elif _END_MARKER.match(line):
|
|
end_idx = i
|
|
break
|
|
operations: List[PatchOperation] = []
|
|
current_op: Optional[PatchOperation] = None
|
|
current_hunk: Optional[Hunk] = None
|
|
|
|
def _flush_hunk() -> None:
|
|
if current_op and current_hunk and current_hunk.lines:
|
|
current_op.hunks.append(current_hunk)
|
|
|
|
def _flush() -> None:
|
|
if current_op:
|
|
_flush_hunk()
|
|
operations.append(current_op)
|
|
|
|
for line in lines[start_idx + 1:end_idx]:
|
|
op_match = next(((kind, m) for kind, rx in _OP_MARKERS if (m := rx.match(line))), None)
|
|
if op_match:
|
|
kind, m = op_match
|
|
_flush()
|
|
current_op = PatchOperation(
|
|
operation=kind,
|
|
file_path=m.group(1).strip(),
|
|
new_path=m.group(2).strip() if kind is OperationType.MOVE else None)
|
|
# UPDATE hunks start lazily ('@@' or first hunk line); ADD collects all '+' lines
|
|
# into one hunk; DELETE/MOVE are complete.
|
|
current_hunk = Hunk() if kind is OperationType.ADD else None
|
|
if kind in (OperationType.DELETE, OperationType.MOVE):
|
|
operations.append(current_op)
|
|
current_op = None
|
|
elif line.startswith('@@'):
|
|
if current_op:
|
|
_flush_hunk()
|
|
hint_match = _HINT_RE.match(line)
|
|
current_hunk = Hunk(context_hint=hint_match.group(1) if hint_match else None)
|
|
elif current_op and line:
|
|
if current_hunk is None:
|
|
current_hunk = Hunk()
|
|
if line[0] in '+- ':
|
|
current_hunk.lines.append(HunkLine(line[0], line[1:]))
|
|
elif line[0] != '\\': # "\ No newline at end of file" marker is skipped
|
|
current_hunk.lines.append(HunkLine(' ', line)) # implicit context line
|
|
_flush()
|
|
parse_errors: List[str] = []
|
|
for op in operations:
|
|
if not op.file_path:
|
|
parse_errors.append("Operation with empty file path")
|
|
if op.operation is OperationType.UPDATE and not op.hunks:
|
|
parse_errors.append(f"UPDATE {op.file_path!r}: no hunks found")
|
|
if op.operation is OperationType.MOVE and not op.new_path:
|
|
parse_errors.append(
|
|
f"MOVE {op.file_path!r}: missing destination path (expected 'src -> dst')")
|
|
return ([], "Parse error: " + "; ".join(parse_errors)) if parse_errors else (operations, None)
|
|
|
|
|
|
def _count_occurrences(text: str, pattern: str) -> int:
|
|
"""Count occurrences of *pattern* in *text*, advancing one char per hit (overlaps count)."""
|
|
return sum(1 for i in range(len(text) + 1) if text.startswith(pattern, i))
|
|
|
|
|
|
def _split_hunk(hunk: Hunk) -> Tuple[List[str], List[str]]:
|
|
"""``(search_lines, replace_lines)``: context+removed vs context+added."""
|
|
return ([l.content for l in hunk.lines if l.prefix != '+'],
|
|
[l.content for l in hunk.lines if l.prefix != '-'])
|
|
|
|
|
|
def _no_match_hint(error: Optional[str], search_pattern: str, content: str) -> str:
|
|
"""Best-effort 'Did you mean...' suffix; never lets a hint failure mask the real error."""
|
|
with contextlib.suppress(Exception):
|
|
from tools.fuzzy_match import format_no_match_hint
|
|
return format_no_match_hint(error, 0, search_pattern, content)
|
|
return ""
|
|
|
|
|
|
def _hint_ambiguity(content: str, hint: str, tail: str = "") -> Tuple[int, str]:
|
|
"""(occurrences, error) for an addition-only hunk's context hint; error is '' when unique."""
|
|
n = _count_occurrences(content, hint)
|
|
return n, f"context hint '{hint}' is ambiguous ({n} occurrences){tail}" if n > 1 else ""
|
|
|
|
|
|
def _validate_operations(operations: List[PatchOperation], file_ops: Any) -> List[str]:
|
|
"""Dry-run every operation -> error strings (empty = safe). UPDATE hunks are simulated in
|
|
order so later hunks see post-earlier-hunk content, exactly as apply will."""
|
|
from tools.fuzzy_match import fuzzy_find_and_replace, is_already_applied
|
|
errors: List[str] = []
|
|
real_change_count = 0
|
|
# Overlay so inter-op state validates (a MOVE creating the path a later UPDATE targets).
|
|
pending_content: dict = {}
|
|
removed_paths: set = set()
|
|
|
|
def _read(path: str) -> Tuple[Optional[str], Optional[str]]:
|
|
if path in pending_content:
|
|
return pending_content[path], None
|
|
if path in removed_paths:
|
|
return None, "file not found"
|
|
r = file_ops.read_file_raw(path)
|
|
return (None, r.error) if r.error else (r.content, None)
|
|
|
|
def _occupied(path: str) -> Optional[str]:
|
|
"""Why an Add target or Move destination is not free, or None. Only a read that reports
|
|
the path absent (``not_found``) frees it: a read that FAILED (no byte transport, a
|
|
directory, an unreadable file) says nothing about what is there, and taking it as free
|
|
writes over the file the check exists to protect."""
|
|
if path in pending_content:
|
|
return "exists"
|
|
if path in removed_paths:
|
|
return None
|
|
r = file_ops.read_file_raw(path)
|
|
if not r.error:
|
|
return "exists"
|
|
return None if getattr(r, "not_found", False) else f"could not confirm the path is free — {r.error}"
|
|
|
|
def _validate_update(op: PatchOperation) -> None:
|
|
nonlocal real_change_count
|
|
simulated, read_err = _read(op.file_path)
|
|
if read_err:
|
|
errors.append(f"{op.file_path}: {read_err}")
|
|
return
|
|
for hunk_index, hunk in enumerate(op.hunks, start=1):
|
|
search_lines, replace_lines = _split_hunk(hunk)
|
|
if search_lines == replace_lines:
|
|
# Context-only anchor hunks (models emit these between changes) are inert; identical
|
|
# -/+ lines are skipped by apply as a no-op — neither may fail validation.
|
|
real_change_count += any(l.prefix in '-+' for l in hunk.lines)
|
|
continue
|
|
real_change_count += 1
|
|
if not search_lines: # addition-only: the context hint must be unique
|
|
if hunk.context_hint:
|
|
occurrences, ambiguous = _hint_ambiguity(simulated, hunk.context_hint)
|
|
if occurrences == 0:
|
|
errors.append(f"{op.file_path}: addition-only hunk context hint "
|
|
f"'{hunk.context_hint}' not found")
|
|
elif ambiguous:
|
|
errors.append(f"{op.file_path}: addition-only hunk {ambiguous}")
|
|
continue
|
|
search_pattern, replacement = '\n'.join(search_lines), '\n'.join(replace_lines)
|
|
new_simulated, count, _strategy, match_error = fuzzy_find_and_replace(
|
|
simulated, search_pattern, replacement, replace_all=False)
|
|
if count:
|
|
simulated = new_simulated
|
|
elif not is_already_applied(simulated or "", search_pattern, replacement):
|
|
# Already-applied hunks are no-ops (apply performs the same skip).
|
|
label = f"'{hunk.context_hint}'" if hunk.context_hint else "(no hint)"
|
|
errors.append(
|
|
f"{op.file_path}: hunk {hunk_index} {label} not found"
|
|
+ (f" — {match_error}" if match_error else "")
|
|
+ _no_match_hint(match_error, search_pattern, simulated))
|
|
pending_content[op.file_path] = simulated
|
|
|
|
def _remove(path: str) -> None:
|
|
removed_paths.add(path)
|
|
pending_content.pop(path, None)
|
|
|
|
for op in operations:
|
|
if op.operation == OperationType.UPDATE:
|
|
_validate_update(op)
|
|
continue
|
|
real_change_count += 1
|
|
if op.operation == OperationType.DELETE:
|
|
if _read(op.file_path)[1]:
|
|
errors.append(f"{op.file_path}: file not found for deletion")
|
|
else:
|
|
_remove(op.file_path)
|
|
elif op.operation == OperationType.MOVE:
|
|
if not op.new_path:
|
|
errors.append(f"{op.file_path}: MOVE operation missing destination path")
|
|
continue
|
|
src_content, src_err = _read(op.file_path)
|
|
if src_err:
|
|
errors.append(f"{op.file_path}: source file not found for move")
|
|
dst_taken = _occupied(op.new_path)
|
|
if dst_taken == "exists":
|
|
errors.append(f"{op.new_path}: destination already exists — move would overwrite")
|
|
elif dst_taken:
|
|
errors.append(f"{op.new_path}: {dst_taken}")
|
|
elif not src_err: # only a cleanly-validated move updates the overlay
|
|
pending_content[op.new_path] = src_content if src_content is not None else ""
|
|
_remove(op.file_path)
|
|
elif op.operation == OperationType.ADD:
|
|
# An Add must create a NEW file. If the target already exists, write_file
|
|
# would clobber it with only the patch's '+' lines and report success,
|
|
# silently destroying the original contents (models frequently confuse Add
|
|
# with Update). Reject it here so the two-phase contract holds, mirroring
|
|
# the MOVE destination guard. Overlay-aware: an Add after a Delete of the
|
|
# same path in this patch stays legal, and the added content enters the
|
|
# overlay so later hunks against it validate.
|
|
add_taken = _occupied(op.file_path)
|
|
if add_taken == "exists":
|
|
errors.append(f"{op.file_path}: file already exists — use Update File, not Add File")
|
|
elif add_taken:
|
|
errors.append(f"{op.file_path}: {add_taken}")
|
|
else:
|
|
removed_paths.discard(op.file_path)
|
|
pending_content[op.file_path] = '\n'.join(
|
|
line.content for hunk in op.hunks for line in hunk.lines if line.prefix == '+')
|
|
if not errors and real_change_count == 0:
|
|
errors.append("Patch contains no changes (only context lines were provided)")
|
|
return errors
|
|
|
|
|
|
# Every _apply_* returns (success, diff_or_error, lsp_diagnostics, lint_result).
|
|
ApplyResult = Tuple[bool, str, Optional[str], Optional[dict]]
|
|
|
|
|
|
def _fail(error: str) -> ApplyResult:
|
|
return False, error, None, None
|
|
|
|
|
|
def _written(result: Any, diff: str) -> ApplyResult:
|
|
"""Outcome of a write: its error, else success with LSP/lint propagated from the WriteResult."""
|
|
if result.error:
|
|
return _fail(result.error)
|
|
return True, diff, getattr(result, "lsp_diagnostics", None), getattr(result, "lint", None)
|
|
|
|
|
|
def _unified_diff(path: str, old: str, new: Optional[str]) -> str:
|
|
"""Unified diff ``a/path`` -> ``b/path`` (``new=None`` = deletion, ``/dev/null``)."""
|
|
return ''.join(difflib.unified_diff(
|
|
old.splitlines(keepends=True), [] if new is None else new.splitlines(keepends=True),
|
|
fromfile=f"a/{path}", tofile="/dev/null" if new is None else f"b/{path}"))
|
|
|
|
|
|
def apply_v4a_operations(operations: List[PatchOperation], file_ops: Any) -> PatchResult:
|
|
"""Two-phase: validate everything, then apply (atomic on validation failure). A phase-2
|
|
failure (validate/apply race) carries a ``git diff`` note since state may be inconsistent.
|
|
``file_ops`` needs read_file_raw/write_file/delete_file/move_file."""
|
|
|
|
def _bullets(errs: List[str]) -> str:
|
|
return "\n".join(f" • {e}" for e in errs)
|
|
|
|
if errors := _validate_operations(operations, file_ops):
|
|
return PatchResult(
|
|
success=False,
|
|
error="Patch validation failed (no files were modified):\n" + _bullets(errors))
|
|
files: Dict[str, List[str]] = {"created": [], "deleted": [], "modified": []}
|
|
all_diffs: List[str] = []
|
|
# V4A bypasses write_file's WriteResult plumbing: LSP diagnostics and lint propagate per file.
|
|
lsp_blocks: List[str] = []
|
|
lint_results: Dict[str, dict] = {}
|
|
for op in operations:
|
|
handler, verb, bucket = _APPLY_DISPATCH[op.operation]
|
|
try:
|
|
ok, payload, lsp, lint = handler(op, file_ops)
|
|
except Exception as e:
|
|
ok, payload = None, str(e)
|
|
if not ok:
|
|
prefix = f"Failed to {verb}" if ok is False else "Error processing"
|
|
errors.append(f"{prefix} {op.file_path}: {payload}")
|
|
continue
|
|
is_move = op.operation is OperationType.MOVE
|
|
files[bucket].append(f"{op.file_path} -> {op.new_path}" if is_move else op.file_path)
|
|
all_diffs.append(payload)
|
|
if lsp:
|
|
lsp_blocks.append(lsp)
|
|
if lint:
|
|
lint_results[op.file_path] = lint
|
|
# Each LSP block carries its own <diagnostics file="..."> header; joining keeps attribution.
|
|
return PatchResult(
|
|
success=not errors,
|
|
error=("Apply phase failed (state may be inconsistent — run `git diff` to assess):\n"
|
|
+ _bullets(errors)) if errors else None,
|
|
diff='\n'.join(all_diffs),
|
|
files_modified=files["modified"], files_created=files["created"], files_deleted=files["deleted"],
|
|
lint=lint_results or None, lsp_diagnostics="\n\n".join(lsp_blocks) or None)
|
|
|
|
|
|
def _write_file_accepts_pre_content(file_ops: Any) -> bool:
|
|
"""Whether ``file_ops.write_file`` accepts ``pre_content`` — read from the signature, not by
|
|
catching TypeError around the call, so a TypeError raised *inside* it can't double-write."""
|
|
try:
|
|
params = inspect.signature(file_ops.write_file).parameters
|
|
except (TypeError, ValueError):
|
|
return False
|
|
return "pre_content" in params or any(
|
|
p.kind is inspect.Parameter.VAR_KEYWORD for p in params.values())
|
|
|
|
|
|
def _apply_add(op: PatchOperation, file_ops: Any) -> ApplyResult:
|
|
"""Create a file from the hunks' '+' lines. Fails closed when the target already
|
|
exists: validation confirmed the path was free (or freed by an earlier DELETE in
|
|
this patch, which has already applied by now), so an existing file here is a
|
|
validate/apply race — never clobber."""
|
|
read_back = file_ops.read_file_raw(op.file_path)
|
|
if not read_back.error:
|
|
return _fail(f"{op.file_path}: file already exists — use Update File, not Add File")
|
|
if not getattr(read_back, "not_found", False):
|
|
# The read FAILED; it did not report an absent path. Treating that as "the path is free"
|
|
# writes the Add payload over whatever is actually there.
|
|
return _fail(f"{op.file_path}: could not confirm the path is free — {read_back.error}")
|
|
content_lines = [line.content for hunk in op.hunks for line in hunk.lines if line.prefix == '+']
|
|
result = file_ops.write_file(op.file_path, '\n'.join(content_lines))
|
|
diff = f"--- /dev/null\n+++ b/{op.file_path}\n" + '\n'.join(f"+{line}" for line in content_lines)
|
|
return _written(result, diff)
|
|
|
|
|
|
def _apply_delete(op: PatchOperation, file_ops: Any) -> ApplyResult:
|
|
"""Delete a file, producing a real unified diff of the removed content."""
|
|
read_result = file_ops.read_file_raw(op.file_path) # re-read guards validate/apply races
|
|
if read_result.error:
|
|
return _fail(f"Cannot delete {op.file_path}: file not found")
|
|
result = file_ops.delete_file(op.file_path)
|
|
diff = _unified_diff(op.file_path, read_result.content, None) or f"# Deleted: {op.file_path}"
|
|
return _fail(result.error) if result.error else (True, diff, None, None)
|
|
|
|
|
|
def _apply_move(op: PatchOperation, file_ops: Any) -> ApplyResult:
|
|
"""Move, re-checking the destination first: validation's answer is stale once earlier ops of
|
|
this patch have applied, and ``mv`` replaces whatever is there."""
|
|
dst = file_ops.read_file_raw(op.new_path)
|
|
if not dst.error:
|
|
return _fail(f"{op.new_path}: destination already exists — move would overwrite")
|
|
if not getattr(dst, "not_found", False):
|
|
return _fail(f"{op.new_path}: could not confirm the destination is free — {dst.error}")
|
|
result = file_ops.move_file(op.file_path, op.new_path)
|
|
return _fail(result.error) if result.error else (
|
|
True, f"# Moved: {op.file_path} -> {op.new_path}", None, None)
|
|
|
|
|
|
def _insert_addition_only(new_content: str, hunk: Hunk, insert_text: str) -> Tuple[Optional[str], Optional[str]]:
|
|
"""Place an addition-only hunk after its context hint (or at EOF). Returns (content, error)."""
|
|
if hunk.context_hint:
|
|
occurrences, ambiguous = _hint_ambiguity(
|
|
new_content, hunk.context_hint, " — provide a more unique hint")
|
|
if ambiguous:
|
|
return None, f"Addition-only hunk: {ambiguous}"
|
|
if occurrences == 1:
|
|
eol = new_content.find('\n', new_content.find(hunk.context_hint))
|
|
if eol == -1:
|
|
return new_content + '\n' + insert_text, None
|
|
return new_content[:eol + 1] + insert_text + '\n' + new_content[eol + 1:], None
|
|
# No hint / hint not found — append at end as a safe fallback.
|
|
return new_content.rstrip('\n') + '\n' + insert_text + '\n', None
|
|
|
|
|
|
def _apply_update(op: PatchOperation, file_ops: Any) -> ApplyResult:
|
|
"""Apply each hunk via fuzzy replace, then write once."""
|
|
from tools.fuzzy_match import fuzzy_find_and_replace, is_already_applied
|
|
read_result = file_ops.read_file_raw(op.file_path) # raw: no line numbers / truncation
|
|
if read_result.error:
|
|
return _fail(f"Cannot read file: {read_result.error}")
|
|
current_content = new_content = read_result.content
|
|
for hunk in op.hunks:
|
|
search_lines, replace_lines = _split_hunk(hunk)
|
|
if search_lines and search_lines == replace_lines:
|
|
continue
|
|
search_pattern, replacement = '\n'.join(search_lines), '\n'.join(replace_lines)
|
|
if not search_lines:
|
|
new_content, err = _insert_addition_only(new_content, hunk, replacement)
|
|
if err:
|
|
return _fail(err)
|
|
continue
|
|
new_content, count, _strategy, error = fuzzy_find_and_replace(
|
|
new_content, search_pattern, replacement, replace_all=False)
|
|
if not (error and count == 0):
|
|
continue
|
|
# Retry inside a window around the context hint, if any.
|
|
hint_pos = new_content.find(hunk.context_hint) if hunk.context_hint else -1
|
|
if hint_pos != -1:
|
|
window_start = max(0, hint_pos - 500)
|
|
window_end = min(len(new_content), hint_pos + 2000)
|
|
window_new, count, _strategy, error = fuzzy_find_and_replace(
|
|
new_content[window_start:window_end], search_pattern, replacement, replace_all=False)
|
|
if count > 0:
|
|
new_content = new_content[:window_start] + window_new + new_content[window_end:]
|
|
error = None
|
|
if error:
|
|
# Mirror validation's already-applied skip, else the two phases disagree and fail here.
|
|
if is_already_applied(new_content, search_pattern, replacement):
|
|
continue
|
|
hint = _no_match_hint(error, search_pattern, new_content)
|
|
return _fail(f"Could not apply hunk: {error}" + hint)
|
|
# Pass pre_content to skip a redundant re-read inside write_file when supported.
|
|
extra = {"pre_content": current_content} if _write_file_accepts_pre_content(file_ops) else {}
|
|
write_result = file_ops.write_file(op.file_path, new_content, **extra)
|
|
return _written(write_result, _unified_diff(op.file_path, current_content, new_content))
|
|
|
|
|
|
# operation -> (handler, verb for error text, files_* bucket)
|
|
_APPLY_DISPATCH: Dict[OperationType, Tuple[Callable[[PatchOperation, Any], ApplyResult], str, str]] = {
|
|
OperationType.ADD: (_apply_add, "add", "created"),
|
|
OperationType.DELETE: (_apply_delete, "delete", "deleted"),
|
|
OperationType.MOVE: (_apply_move, "move", "modified"),
|
|
OperationType.UPDATE: (_apply_update, "update", "modified"),
|
|
}
|