refactor(tools): AST-neutral layout compaction of file_tools group

This commit is contained in:
Teknium
2026-09-02 22:51:55 -07:00
parent 8d6232878e
commit 40f2c7951b
6 changed files with 72 additions and 149 deletions

View File

@@ -139,16 +139,14 @@ class FileStateRegistry:
f"{resolved} was modified by sibling subagent "
f"{writer_tid!r} but this agent never read it. "
"Read the file before writing to avoid overwriting "
"the sibling's changes."
)
"the sibling's changes.")
read_ts = stamp[1]
if writer_ts > read_ts:
return (
f"{resolved} was modified by sibling subagent "
f"{writer_tid!r} at {_fmt_ts(writer_ts)} — after "
f"this agent's last read at {_fmt_ts(read_ts)}. "
"Re-read the file before writing."
)
"Re-read the file before writing.")
if stamp is not None:
read_mtime, _read_ts, partial = stamp
@@ -156,20 +154,17 @@ class FileStateRegistry:
return (
f"{resolved} was modified since you last read it "
"on disk (external edit or unrecorded writer). "
"Re-read the file before writing."
)
"Re-read the file before writing.")
if partial:
return (
f"{resolved} was last read with offset/limit pagination "
"(partial view). Re-read the whole file before "
"overwriting it."
)
"overwriting it.")
return None
return (
f"{resolved} was not read by this agent. "
"Read the file first so you can write an informed edit."
)
"Read the file first so you can write an informed edit.")
def writes_since(self, exclude_task_id: str, since_ts: float,
paths: Iterable[str]) -> Dict[str, List[str]]:
@@ -242,5 +237,4 @@ __all__ = [
"check_stale",
"lock_path",
"writes_since",
"known_reads",
]
"known_reads"]

View File

@@ -24,10 +24,7 @@ from pathlib import Path
from agent.file_safety import get_read_block_error
from tools.binary_extensions import has_binary_extension
from tools.file_operations import (
ShellFileOperations,
normalize_read_pagination,
normalize_search_pagination,
)
ShellFileOperations, normalize_read_pagination, normalize_search_pagination)
from tools import file_state
from agent.redact import redact_sensitive_text
from tools.file_tools_paths import ( # noqa: F401 (re-exported)
@@ -44,8 +41,7 @@ from tools.file_tools_paths import ( # noqa: F401 (re-exported)
_resolve_path_for_task,
_sentinel_free_abs_cwd,
_terminal_env_type_for_task,
_uses_container_paths,
)
_uses_container_paths)
from tools.file_tools_write_guards import ( # noqa: F401 (re-exported)
_PROTECTED_INSTRUCTION_BASENAMES,
_READ_DEDUP_STATUS_MESSAGE,
@@ -64,8 +60,7 @@ from tools.file_tools_write_guards import ( # noqa: F401 (re-exported)
_looks_like_read_file_line_numbered_content,
_protected_instruction_config,
_protected_instruction_reason,
_request_protected_instruction_approval,
)
_request_protected_instruction_approval)
from tools.file_tools_read_tracking import ( # noqa: F401 (re-exported)
_DEDUP_CAP,
_NOT_FOUND_CAP,
@@ -88,8 +83,7 @@ from tools.file_tools_read_tracking import ( # noqa: F401 (re-exported)
_task_data,
_update_read_timestamp,
notify_other_tool_call,
reset_file_dedup,
)
reset_file_dedup)
logger = logging.getLogger(__name__)
@@ -155,14 +149,12 @@ def _apply_char_budget(result_dict: dict, content: str, offset: int, total_lines
result_dict["hint"] = (
f"Output truncated at the {max_chars:,}-char read budget after "
f"{lines_kept} line(s) (showing lines {offset}-{next_offset - 1} of "
f"{total_lines}). Use offset={next_offset} to continue."
)
f"{total_lines}). Use offset={next_offset} to continue.")
if len(trimmed.split("\n", 1)[0]) >= max_chars:
result_dict["hint"] += (
" Note: the first line alone exceeded the budget and was "
"clamped mid-line; its remainder is not retrievable via "
"offset."
)
"offset.")
return trimmed
@@ -182,8 +174,7 @@ _BLOCKED_DEVICE_PATHS = frozenset({
_BLOCKED_PROC_SUFFIXES = (
"/fd/0", "/fd/1", "/fd/2", # stdio aliases
"/environ", "/cmdline", "/maps", "/smaps", "/smaps_rollup", "/numa_maps",
"/mem", "/auxv", "/pagemap",
)
"/mem", "/auxv", "/pagemap")
def _file_ops_uses_host_paths(file_ops) -> bool:
@@ -330,8 +321,7 @@ def _create_terminal_env_for_file_ops(raw_task_id: str, task_id: str):
_resolve_task_host_cwd,
_select_image,
get_session_cwd,
resolve_task_overrides,
)
resolve_task_overrides)
config = _get_env_config()
env_type = config["env_type"]
@@ -351,8 +341,7 @@ def _create_terminal_env_for_file_ops(raw_task_id: str, task_id: str):
logger.info(
"Ignoring host/relative cwd override %r for %s backend "
"(won't exist in sandbox). Using %r instead.",
cwd, env_type, config["cwd"],
)
cwd, env_type, config["cwd"])
cwd = config["cwd"]
logger.info("Creating new %s environment for task %s...", env_type, task_id[:8])
terminal_env = _create_configured_env(
@@ -376,8 +365,7 @@ def _get_file_ops(task_id: str = "default") -> ShellFileOperations:
from tools.terminal_tool import (
_active_environments, _env_lock, _last_activity, _start_cleanup_thread,
_creation_locks, _creation_locks_lock, _resolve_container_task_id,
get_session_cwd, record_session_cwd,
)
get_session_cwd, record_session_cwd)
raw_task_id = task_id or "default"
task_id = _resolve_container_task_id(raw_task_id)
@@ -441,8 +429,7 @@ _SPECIAL_FILE_KINDS = (
(stat.S_ISFIFO, "a FIFO (named pipe)"),
(stat.S_ISSOCK, "a socket"),
(stat.S_ISCHR, "a character device"),
(stat.S_ISBLK, "a block device"),
)
(stat.S_ISBLK, "a block device"))
def _special_file_kind(path) -> str | None:
@@ -478,8 +465,7 @@ def _read_extracted_document(path: str, _resolved, offset: int, limit: int, task
MAX_DOCUMENT_BYTES,
ExtractionError,
extract_document_bytes,
is_extractable_document,
)
is_extractable_document)
if not is_extractable_document(str(_resolved)):
return None
@@ -504,8 +490,7 @@ def _read_extracted_document(path: str, _resolved, offset: int, limit: int, task
return tool_error(
f"Cannot read '{path}' ({_doc_ext}): document "
f"extraction failed — {exc}. Use terminal utilities "
"to inspect or convert the file."
)
"to inspect or convert the file.")
return None
lines = extracted_text.splitlines()
@@ -517,13 +502,11 @@ def _read_extracted_document(path: str, _resolved, offset: int, limit: int, task
"total_lines": total_lines,
"file_size": binary.file_size,
"truncated": total_lines > end_line,
"extracted_document": True,
}
"extracted_document": True}
if result_dict["truncated"]:
result_dict["hint"] = (
f"Use offset={end_line + 1} to continue reading "
f"(showing {offset}-{min(end_line, total_lines)} of {total_lines} lines)"
)
f"(showing {offset}-{min(end_line, total_lines)} of {total_lines} lines)")
max_chars = _get_max_read_chars()
if len(result_dict["content"]) > max_chars:
_apply_char_budget(result_dict, result_dict["content"], offset, total_lines, max_chars)
@@ -550,8 +533,7 @@ def _dedup_stub_or_block(task_data: dict, dedup_key: tuple, path: str) -> str:
"still current. Proceed with your task using "
"the information you already have.",
path=path,
already_read=hits + 1,
)
already_read=hits + 1)
return json.dumps({
"status": "unchanged",
@@ -614,8 +596,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str =
if _is_blocked_device(path, base_dir=device_base):
return tool_error(
f"Cannot read '{path}': this is a device file that would "
"block or produce infinite output."
)
"block or produce infinite output.")
_resolved = _resolve_path_for_task(path, task_id)
@@ -629,9 +610,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str =
f"'{path}' is {kind}, not a regular file — reading "
"it would block indefinitely, so no read was "
"attempted. Use terminal utilities if you need to "
"interact with it."
),
})
"interact with it.")})
extracted = _read_extracted_document(path, _resolved, offset, limit, task_id)
if extracted is not None:
@@ -642,8 +621,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str =
if has_binary_extension(str(_resolved)):
return tool_error(
f"Cannot read binary file '{path}' ({_resolved.suffix.lower()}). "
"Use vision_analyze for images, or terminal to inspect binary files."
)
"Use vision_analyze for images, or terminal to inspect binary files.")
# Hermes internal denylist: blocks prompt injection via catalog/hub
# metadata and credential stores under HERMES_HOME. Pass the resolved
@@ -692,8 +670,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str =
if len(result.content or "") > max_chars:
result.content = _apply_char_budget(
result_dict, result.content or "", offset,
result_dict.get("total_lines", "unknown"), max_chars,
)
result_dict.get("total_lines", "unknown"), max_chars)
if result.content:
result.content = redact_sensitive_text(result.content, file_read=True)
result_dict["content"] = result.content
@@ -703,8 +680,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str =
result_dict.setdefault("_hint", (
f"This file is large ({file_size:,} bytes). "
"Consider reading only the section you need with offset and limit "
"to keep context usage efficient."
))
"to keep context usage efficient."))
count = _record_successful_read(task_data, task_id, path, resolved_str, offset, limit,
dedup_key, partial=(offset > 1) or bool(result_dict.get("truncated")))
@@ -714,14 +690,12 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str =
"The content has NOT changed. You already have this information. "
"STOP re-reading and proceed with your task.",
path=path,
already_read=count,
)
already_read=count)
if count >= 3:
result_dict["_warning"] = (
f"You have read this exact file region {count} times consecutively. "
"The content has not changed since your last read. Use the information you already have. "
"If you are stuck in a loop, stop reading and proceed with writing or responding."
)
"If you are stuck in a loop, stop reading and proceed with writing or responding.")
return json.dumps(result_dict, ensure_ascii=False)
except Exception as e:
return tool_error(str(e))
@@ -764,8 +738,7 @@ def _write_precheck_error(paths: list[str], content_paths: list[str], task_id: s
return _first_error(
*(lambda p=p: _check_binary_document_write(p, task_id) for p in content_paths),
lambda: _check_protected_instruction_write(paths, task_id),
lambda: _check_approval_required_write(paths, task_id),
)
lambda: _check_approval_required_write(paths, task_id))
def _edit_warnings(paths: list[str], path_to_resolved: dict, task_id: str) -> list[str]:
@@ -813,8 +786,7 @@ def write_file_tool(path: str, content: str, task_id: str = "default",
"Refusing to write internal read_file display text as file content. "
"Strip read_file line-number prefixes or reconstruct the intended "
"file contents before writing."
) if _is_internal_file_tool_content(content) else None,
)
) if _is_internal_file_tool_content(content) else None)
if err:
return tool_error(err)
try:
@@ -873,8 +845,7 @@ def _collect_v4a_header_paths(patch: str) -> tuple[list[str], list[str]] | str:
f"V4A patch header contains '..' traversal: {v4a_path!r}. "
"Use the agent's cwd-relative path (no '..') or an absolute "
"path in '*** Update File:' / '*** Add File:' / "
"'*** Delete File:' / '*** Move File:' headers."
)
"'*** Delete File:' / '*** Move File:' headers.")
paths.append(v4a_path)
if writes_text:
content_paths.append(v4a_path)
@@ -958,13 +929,11 @@ def patch_tool(mode: str = "replace", path: str = None, old_string: str = None,
"content, (2) use a longer / more unique old_string with "
"surrounding context lines, or (3) use write_file to "
"replace the entire file if the targeted region is hard "
"to anchor."
)
"to anchor.")
elif "Did you mean one of these sections?" not in str(result_dict["error"]):
result_dict["_hint"] = (
"old_string not found. Use read_file to verify the current "
"content, or search_files to locate the text."
)
"content, or search_files to locate the text.")
return json.dumps(result_dict, ensure_ascii=False)
except Exception as e:
return tool_error(str(e))
@@ -983,8 +952,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".",
search_key = ("search", pattern, target, str(path), file_glob or "", limit, offset)
with _read_tracker_lock:
task_data = _read_tracker.setdefault(task_id, {
"last_key": None, "consecutive": 0, "read_history": set(),
})
"last_key": None, "consecutive": 0, "read_history": set()})
count = _bump_consecutive(task_data, search_key)
if count >= 4:
@@ -993,8 +961,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".",
"The results have NOT changed. You already have this information. "
"STOP re-searching and proceed with your task.",
pattern=pattern,
already_searched=count,
)
already_searched=count)
try:
resolved_search_path = str(_resolve_path_for_task(path, task_id))
@@ -1016,8 +983,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".",
result = _get_file_ops(task_id).search(
pattern=pattern, path=path, target=target, file_glob=file_glob,
limit=limit, offset=offset, output_mode=output_mode, context=context
)
limit=limit, offset=offset, output_mode=output_mode, context=context)
omitted = _filter_read_blocked_search_results(result, task_id)
for m in getattr(result, "matches", None) or ():
if getattr(m, "content", None):
@@ -1027,8 +993,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".",
if omitted:
result_dict["_omitted"] = (
f"{omitted} result(s) omitted because they target credential, "
"token, cache, or secret-bearing environment files."
)
"token, cache, or secret-bearing environment files.")
# No early return on a cached miss — same rationale as the read path.
_search_err = result_dict.get("error") or ""
@@ -1038,8 +1003,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".",
if count >= 3:
result_dict["_warning"] = (
f"You have run this exact search {count} times consecutively. "
"The results have not changed. Use the information you already have."
)
"The results have not changed. Use the information you already have.")
result_json = json.dumps(result_dict, ensure_ascii=False)
if result_dict.get("truncated"):

View File

@@ -47,8 +47,7 @@ def _terminal_env_type_for_task(task_id: str = "default") -> str:
"""Best-effort terminal backend type for path-resolution decisions."""
try:
from tools.terminal_tool import (
_active_environments, _env_lock, _get_env_config, _resolve_container_task_id,
)
_active_environments, _env_lock, _get_env_config, _resolve_container_task_id)
try:
container_key = _resolve_container_task_id(task_id)
@@ -147,10 +146,7 @@ def _authoritative_workspace_root(task_id: str = "default") -> str | None:
def _resolve_base_dir(
task_id: str = "default",
*,
container_paths: bool | None = None,
) -> Path | PurePosixPath:
task_id: str = "default", *, container_paths: bool | None = None) -> Path | PurePosixPath:
"""Return the ABSOLUTE base directory for resolving relative paths.
Uses ``_authoritative_workspace_root`` (live cwd → registered override →
@@ -249,7 +245,6 @@ def _path_resolution_warning(filepath: str, resolved: Path, task_id: str = "defa
f"OUTSIDE the active workspace ({str(root)!r}). The edit will land in "
f"a different directory than the terminal's cwd. If this is not "
f"intended (e.g. a git-worktree session writing into the main "
f"checkout), pass an absolute path under the workspace instead."
)
f"checkout), pass an absolute path under the workspace instead.")
except Exception:
return None

View File

@@ -44,8 +44,7 @@ def _task_data(task_id: str) -> dict:
code paths (or injected by tests) may lack the newer containers.
"""
task_data = _read_tracker.setdefault(task_id, {
"last_key": None, "consecutive": 0, "read_history": set(),
})
"last_key": None, "consecutive": 0, "read_history": set()})
for key in ("dedup", "dedup_hits", "read_timestamps"):
task_data.setdefault(key, {})
return task_data
@@ -80,8 +79,7 @@ def _cap_read_tracker_data(task_data: dict) -> None:
("dedup", _DEDUP_CAP),
("dedup_hits", _DEDUP_CAP),
("read_timestamps", _READ_TIMESTAMPS_CAP),
("not_found", _NOT_FOUND_CAP),
):
("not_found", _NOT_FOUND_CAP)):
container = task_data.get(key)
if container is not None and len(container) > cap:
_evict_oldest(container, cap)
@@ -237,8 +235,7 @@ def _check_file_staleness(filepath: str, task_id: str) -> str | None:
return (
f"Warning: {filepath} was modified since you last read it "
"(external edit or concurrent agent). The content you read may be "
"stale. Consider re-reading the file to verify before writing."
)
"stale. Consider re-reading the file to verify before writing.")
return None

View File

@@ -21,8 +21,7 @@ from tools.file_tools_paths import _expand_tilde, _resolve_path_for_task
_SENSITIVE_PATH_PREFIXES = (
"/etc/", "/boot/", "/usr/lib/systemd/",
"/private/etc/",
"/private/var/db/", "/private/var/root/",
)
"/private/var/db/", "/private/var/root/")
_SENSITIVE_EXACT_PATHS = {"/var/run/docker.sock", "/run/docker.sock"}
_hermes_config_resolved: str | None = None
@@ -77,8 +76,7 @@ def _check_sensitive_path(filepath: str, task_id: str = "default") -> str | None
if any(c.startswith(_SENSITIVE_PATH_PREFIXES) or c in _SENSITIVE_EXACT_PATHS for c in candidates):
return (
f"Refusing to write to sensitive system path: {filepath}\n"
"Use the terminal tool with sudo if you need to modify system files."
)
"Use the terminal tool with sudo if you need to modify system files.")
# approvals.mode and other security settings live in config.yaml; a
# prompt-injected agent could silently disable exec approval by editing it.
hermes_config = _get_hermes_config_resolved()
@@ -86,8 +84,7 @@ def _check_sensitive_path(filepath: str, task_id: str = "default") -> str | None
return (
f"Refusing to write to Hermes config file: {filepath}\n"
"Agent cannot modify security-sensitive configuration. "
"Edit ~/.hermes/config.yaml directly or use 'hermes config' instead."
)
"Edit ~/.hermes/config.yaml directly or use 'hermes config' instead.")
return None
@@ -100,8 +97,7 @@ def _check_sensitive_path(filepath: str, task_id: str = "default") -> str | None
# human approval — even under --yolo — and fail closed without a human channel.
# Basenames match in ANY directory and case-insensitively (loaders probe variants).
_PROTECTED_INSTRUCTION_BASENAMES = frozenset({
"agents.md", "claude.md", "soul.md", ".cursorrules",
})
"agents.md", "claude.md", "soul.md", ".cursorrules"})
def _protected_instruction_config() -> tuple[bool, list[str]]:
@@ -182,15 +178,13 @@ def _request_protected_instruction_approval(reasons: list[str], task_id: str = "
description = (
f"Write to protected agent-instruction file(s): {targets}. "
"These files steer future agent behavior; approval is always "
"required (not bypassed by auto-approve)."
)
"required (not bypassed by auto-approve).")
display = f"<write to {targets}>"
blocked = (
f"BLOCKED: write to protected agent-instruction file(s) ({targets}) "
"{why} The user has NOT consented to this write. Do NOT retry it or "
"attempt the same edit via another path (terminal, execute_code, "
"etc.)."
)
"etc.).")
timed_out = blocked.format(why="approval prompt timed out without a user response. Silence is not consent.")
denied = blocked.format(why="was denied by the user.")
@@ -215,8 +209,7 @@ def _request_protected_instruction_approval(reasons: list[str], task_id: str = "
"pattern_keys": ["protected_instruction_file"],
"description": description,
"allow_permanent": False,
"allow_session": False,
}
"allow_session": False}
decision = _approval._await_gateway_decision(session_key, notify_cb, approval_data, surface="gateway")
if decision.get("notify_failed"):
return blocked.format(why="requires approval but the approval request could not be delivered.")
@@ -282,13 +275,11 @@ def _check_approval_required_write(paths: list[str], task_id: str = "default") -
description = (
f"Write to SSH client config file(s): {display_targets}. "
"The SSH config can carry ProxyCommand / Match exec directives that "
"run commands, so writes require your approval."
)
"run commands, so writes require your approval.")
blocked = (
f"BLOCKED: write to SSH config file(s) ({display_targets}) "
"{why} Do NOT retry it via another path (terminal, execute_code) "
"without the user's explicit consent."
)
"without the user's explicit consent.")
try:
import tools.approval as _approval
@@ -307,8 +298,7 @@ def _check_approval_required_write(paths: list[str], task_id: str = "default") -
"approve in config.yaml."),
autoapprove_log_prefix="ssh_config_write",
fail_closed_when_no_human=True,
no_human_block_message=blocked.format(why=_NO_HUMAN),
)
no_human_block_message=blocked.format(why=_NO_HUMAN))
if result.get("approved"):
return None
return result.get("message") or blocked.format(why="was denied.")
@@ -318,8 +308,7 @@ def _get_container_mirror_prefix_for_task(task_id: str = "default") -> str | Non
"""Return the container-side Hermes mirror prefix for persistent Docker file tools."""
try:
from tools.terminal_tool import (
_active_environments, _env_lock, _get_env_config, _resolve_container_task_id,
)
_active_environments, _env_lock, _get_env_config, _resolve_container_task_id)
container_key = _resolve_container_task_id(task_id)
with _env_lock:
env = _active_environments.get(container_key) or _active_environments.get(task_id)
@@ -372,8 +361,7 @@ def _check_binary_document_write(filepath: str, task_id: str = "default") -> str
"corrupt the file (read_file showed you EXTRACTED text, not the real "
"bytes). Use the docx/xlsx/powerpoint skills or a library like "
"python-docx/openpyxl/python-pptx via the terminal to create or edit "
"this document."
)
"this document.")
if is_pdf_path(filepath):
try:
resolved = Path(_resolve_path_for_task(filepath, task_id))
@@ -386,8 +374,7 @@ def _check_binary_document_write(filepath: str, task_id: str = "default") -> str
"read_file showed you EXTRACTED text, not the real bytes — writing "
"text back would destroy the document. Use the pdf skill or a PDF "
"library via the terminal to modify it. (Creating a NEW .pdf file "
"is allowed.)"
)
"is allowed.)")
except OSError:
pass
return None
@@ -399,8 +386,7 @@ def _check_binary_document_write(filepath: str, task_id: str = "default") -> str
_READ_DEDUP_STATUS_MESSAGE = (
"File unchanged since last read. The content from "
"the earlier read_file result in this conversation is "
"still current — refer to that instead of re-reading."
)
"still current — refer to that instead of re-reading.")
def _is_internal_file_status_text(content: str) -> bool:

View File

@@ -18,8 +18,7 @@ Span = tuple[int, int]
IDENTICAL_STRINGS_ERROR = (
"No edit was applied because old_string and new_string are identical. "
"Provide the existing text to replace in old_string and the changed "
"replacement text in new_string."
)
"replacement text in new_string.")
UNICODE_MAP = {
"\u201c": '"', "\u201d": '"', # smart double quotes
@@ -32,8 +31,7 @@ UNICODE_MAP = {
"\u2000": " ", "\u2001": " ", "\u2002": " ", "\u2003": " ",
"\u2004": " ", "\u2005": " ", "\u2006": " ", "\u2007": " ",
"\u2008": " ", "\u2009": " ", "\u200a": " ", "\u202f": " ",
"\u205f": " ", "\u3000": " ",
}
"\u205f": " ", "\u3000": " "}
def _unicode_normalize(text: str) -> str:
@@ -58,8 +56,7 @@ def _window_spans(content: str, content_lines: list[str], n: int,
"""Spans of every ``n``-line window starting at ``i`` for which ``accept(i)``."""
return [
_calculate_line_positions(content_lines, i, i + n, len(content))
for i in range(len(content_lines) - n + 1) if accept(i)
]
for i in range(len(content_lines) - n + 1) if accept(i)]
def _match_transformed_lines(content: str, pattern: str,
@@ -257,8 +254,7 @@ def _strategy_block_anchor(content: str, pattern: str) -> list[Span]:
potential_matches = {
i for i in range(len(norm_content_lines) - n + 1)
if norm_content_lines[i].strip() == first_line
and norm_content_lines[i + n - 1].strip() == last_line
}
and norm_content_lines[i + n - 1].strip() == last_line}
# Looser thresholds (0.10/0.30) matched unrelated blocks; these are the safe floor.
threshold = 0.50 if len(potential_matches) == 1 else 0.70
pattern_middle = '\n'.join(pattern_lines[1:-1])
@@ -300,8 +296,7 @@ def _strategy_context_aware(content: str, pattern: str) -> list[Span]:
return False
return all(
not p_line.strip() or _sim(p_line.strip(), c_line.strip()) >= 0.80
for p_line, c_line in zip(pattern_lines, block_lines)
)
for p_line, c_line in zip(pattern_lines, block_lines))
return _window_spans(content, content_lines, n, accept)
@@ -316,8 +311,7 @@ STRATEGIES: list[tuple[str, Callable[[str, str], list[Span]]]] = [
("trimmed_boundary", _strategy_trimmed_boundary),
("unicode_normalized", _strategy_unicode_normalized),
("block_anchor", _strategy_block_anchor),
("context_aware", _strategy_context_aware),
]
("context_aware", _strategy_context_aware)]
# Matches from these only *approximately* resemble old_string — fine for one
# unique replacement, never safe under replace_all.
@@ -386,15 +380,13 @@ def fuzzy_find_and_replace(content: str, old_string: str, new_string: str,
return content, 0, None, (
f"Found {len(matches)} matches for old_string. "
f"Provide more context to make it unique, or use replace_all=True. "
f"Matches:\n{locations}"
)
f"Matches:\n{locations}")
if replace_all and len(matches) > 1 and strategy_name in SIMILARITY_STRATEGIES:
return content, 0, None, (
f"Found {len(matches)} approximate matches via the "
f"'{strategy_name}' strategy; replace_all only applies to exact "
f"matches. Provide the precise text (whitespace included) so an "
f"exact/line-trimmed match can be made."
)
f"exact/line-trimmed match can be made.")
# Non-exact matches came through some normalization, so new_string may
# carry serialization drift the file doesn't have.
@@ -408,8 +400,7 @@ def fuzzy_find_and_replace(content: str, old_string: str, new_string: str,
effective_new = _preserve_unicode_in_replacement(content, matches, old_string, effective_new)
new_content = _apply_replacements(
content, matches, effective_new,
old_string=old_string if strategy_name != "exact" else None,
)
old_string=old_string if strategy_name != "exact" else None)
return new_content, len(matches), strategy_name, None
return content, 0, None, "Could not find a match for old_string in the file"
@@ -441,8 +432,7 @@ def _detect_escape_drift(content: str, matches: list[Span],
f"serialization artifact where an apostrophe or quote got "
f"prefixed with a spurious backslash. Re-read the file with "
f"read_file and pass old_string/new_string without "
f"backslash-escaping {plain!r} characters."
)
f"backslash-escaping {plain!r} characters.")
return _detect_backslash_doubling(matched_regions, old_string, new_string)
@@ -476,8 +466,7 @@ def _detect_backslash_doubling(matched_regions: str, old_string: str,
"were JSON-escaped one extra time; applying new_string verbatim would "
"double every backslash in the file. Re-read the file with read_file "
"and resend old_string/new_string with the backslash counts exactly "
"as they appear in the file."
)
"as they appear in the file.")
def _maybe_unescape_new_string(new_string: str, content: str, matches: list[Span]) -> str:
@@ -626,8 +615,7 @@ def find_closest_lines(old_string: str, content: str, context_lines: int = 2, ma
continue
seen_ranges.add((start, end))
parts.append("\n".join(
f"{start + j + 1:4d}| {content_lines[start + j]}" for j in range(end - start)
))
f"{start + j + 1:4d}| {content_lines[start + j]}" for j in range(end - start)))
if not parts:
return ""
result = "\n---\n".join(parts)
@@ -640,8 +628,7 @@ def find_closest_lines(old_string: str, content: str, context_lines: int = 2, ma
"\n\nWhitespace difference detected (→ = tab, · = space):\n"
f" file has: {_visualize_whitespace(best_line)}\n"
f" you sent: {_visualize_whitespace(old_lines[0])}\n"
"Use the exact whitespace shown in 'file has'."
)
"Use the exact whitespace shown in 'file has'.")
return result