From 40f2c7951b70ec540f6ade5bc65181ebd869e3c4 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 22:51:55 -0700 Subject: [PATCH] refactor(tools): AST-neutral layout compaction of file_tools group --- tools/file_state.py | 18 ++---- tools/file_tools.py | 102 ++++++++++-------------------- tools/file_tools_paths.py | 11 +--- tools/file_tools_read_tracking.py | 9 +-- tools/file_tools_write_guards.py | 42 ++++-------- tools/fuzzy_match.py | 39 ++++-------- 6 files changed, 72 insertions(+), 149 deletions(-) diff --git a/tools/file_state.py b/tools/file_state.py index 99f3e2d069..36dd1da640 100644 --- a/tools/file_state.py +++ b/tools/file_state.py @@ -139,16 +139,14 @@ class FileStateRegistry: f"{resolved} was modified by sibling subagent " f"{writer_tid!r} but this agent never read it. " "Read the file before writing to avoid overwriting " - "the sibling's changes." - ) + "the sibling's changes.") read_ts = stamp[1] if writer_ts > read_ts: return ( f"{resolved} was modified by sibling subagent " f"{writer_tid!r} at {_fmt_ts(writer_ts)} — after " f"this agent's last read at {_fmt_ts(read_ts)}. " - "Re-read the file before writing." - ) + "Re-read the file before writing.") if stamp is not None: read_mtime, _read_ts, partial = stamp @@ -156,20 +154,17 @@ class FileStateRegistry: return ( f"{resolved} was modified since you last read it " "on disk (external edit or unrecorded writer). " - "Re-read the file before writing." - ) + "Re-read the file before writing.") if partial: return ( f"{resolved} was last read with offset/limit pagination " "(partial view). Re-read the whole file before " - "overwriting it." - ) + "overwriting it.") return None return ( f"{resolved} was not read by this agent. " - "Read the file first so you can write an informed edit." - ) + "Read the file first so you can write an informed edit.") def writes_since(self, exclude_task_id: str, since_ts: float, paths: Iterable[str]) -> Dict[str, List[str]]: @@ -242,5 +237,4 @@ __all__ = [ "check_stale", "lock_path", "writes_since", - "known_reads", -] + "known_reads"] diff --git a/tools/file_tools.py b/tools/file_tools.py index 7c24dae8a4..29bf67fb47 100644 --- a/tools/file_tools.py +++ b/tools/file_tools.py @@ -24,10 +24,7 @@ from pathlib import Path from agent.file_safety import get_read_block_error from tools.binary_extensions import has_binary_extension from tools.file_operations import ( - ShellFileOperations, - normalize_read_pagination, - normalize_search_pagination, -) + ShellFileOperations, normalize_read_pagination, normalize_search_pagination) from tools import file_state from agent.redact import redact_sensitive_text from tools.file_tools_paths import ( # noqa: F401 (re-exported) @@ -44,8 +41,7 @@ from tools.file_tools_paths import ( # noqa: F401 (re-exported) _resolve_path_for_task, _sentinel_free_abs_cwd, _terminal_env_type_for_task, - _uses_container_paths, -) + _uses_container_paths) from tools.file_tools_write_guards import ( # noqa: F401 (re-exported) _PROTECTED_INSTRUCTION_BASENAMES, _READ_DEDUP_STATUS_MESSAGE, @@ -64,8 +60,7 @@ from tools.file_tools_write_guards import ( # noqa: F401 (re-exported) _looks_like_read_file_line_numbered_content, _protected_instruction_config, _protected_instruction_reason, - _request_protected_instruction_approval, -) + _request_protected_instruction_approval) from tools.file_tools_read_tracking import ( # noqa: F401 (re-exported) _DEDUP_CAP, _NOT_FOUND_CAP, @@ -88,8 +83,7 @@ from tools.file_tools_read_tracking import ( # noqa: F401 (re-exported) _task_data, _update_read_timestamp, notify_other_tool_call, - reset_file_dedup, -) + reset_file_dedup) logger = logging.getLogger(__name__) @@ -155,14 +149,12 @@ def _apply_char_budget(result_dict: dict, content: str, offset: int, total_lines result_dict["hint"] = ( f"Output truncated at the {max_chars:,}-char read budget after " f"{lines_kept} line(s) (showing lines {offset}-{next_offset - 1} of " - f"{total_lines}). Use offset={next_offset} to continue." - ) + f"{total_lines}). Use offset={next_offset} to continue.") if len(trimmed.split("\n", 1)[0]) >= max_chars: result_dict["hint"] += ( " Note: the first line alone exceeded the budget and was " "clamped mid-line; its remainder is not retrievable via " - "offset." - ) + "offset.") return trimmed @@ -182,8 +174,7 @@ _BLOCKED_DEVICE_PATHS = frozenset({ _BLOCKED_PROC_SUFFIXES = ( "/fd/0", "/fd/1", "/fd/2", # stdio aliases "/environ", "/cmdline", "/maps", "/smaps", "/smaps_rollup", "/numa_maps", - "/mem", "/auxv", "/pagemap", -) + "/mem", "/auxv", "/pagemap") def _file_ops_uses_host_paths(file_ops) -> bool: @@ -330,8 +321,7 @@ def _create_terminal_env_for_file_ops(raw_task_id: str, task_id: str): _resolve_task_host_cwd, _select_image, get_session_cwd, - resolve_task_overrides, - ) + resolve_task_overrides) config = _get_env_config() env_type = config["env_type"] @@ -351,8 +341,7 @@ def _create_terminal_env_for_file_ops(raw_task_id: str, task_id: str): logger.info( "Ignoring host/relative cwd override %r for %s backend " "(won't exist in sandbox). Using %r instead.", - cwd, env_type, config["cwd"], - ) + cwd, env_type, config["cwd"]) cwd = config["cwd"] logger.info("Creating new %s environment for task %s...", env_type, task_id[:8]) terminal_env = _create_configured_env( @@ -376,8 +365,7 @@ def _get_file_ops(task_id: str = "default") -> ShellFileOperations: from tools.terminal_tool import ( _active_environments, _env_lock, _last_activity, _start_cleanup_thread, _creation_locks, _creation_locks_lock, _resolve_container_task_id, - get_session_cwd, record_session_cwd, - ) + get_session_cwd, record_session_cwd) raw_task_id = task_id or "default" task_id = _resolve_container_task_id(raw_task_id) @@ -441,8 +429,7 @@ _SPECIAL_FILE_KINDS = ( (stat.S_ISFIFO, "a FIFO (named pipe)"), (stat.S_ISSOCK, "a socket"), (stat.S_ISCHR, "a character device"), - (stat.S_ISBLK, "a block device"), -) + (stat.S_ISBLK, "a block device")) def _special_file_kind(path) -> str | None: @@ -478,8 +465,7 @@ def _read_extracted_document(path: str, _resolved, offset: int, limit: int, task MAX_DOCUMENT_BYTES, ExtractionError, extract_document_bytes, - is_extractable_document, - ) + is_extractable_document) if not is_extractable_document(str(_resolved)): return None @@ -504,8 +490,7 @@ def _read_extracted_document(path: str, _resolved, offset: int, limit: int, task return tool_error( f"Cannot read '{path}' ({_doc_ext}): document " f"extraction failed — {exc}. Use terminal utilities " - "to inspect or convert the file." - ) + "to inspect or convert the file.") return None lines = extracted_text.splitlines() @@ -517,13 +502,11 @@ def _read_extracted_document(path: str, _resolved, offset: int, limit: int, task "total_lines": total_lines, "file_size": binary.file_size, "truncated": total_lines > end_line, - "extracted_document": True, - } + "extracted_document": True} if result_dict["truncated"]: result_dict["hint"] = ( f"Use offset={end_line + 1} to continue reading " - f"(showing {offset}-{min(end_line, total_lines)} of {total_lines} lines)" - ) + f"(showing {offset}-{min(end_line, total_lines)} of {total_lines} lines)") max_chars = _get_max_read_chars() if len(result_dict["content"]) > max_chars: _apply_char_budget(result_dict, result_dict["content"], offset, total_lines, max_chars) @@ -550,8 +533,7 @@ def _dedup_stub_or_block(task_data: dict, dedup_key: tuple, path: str) -> str: "still current. Proceed with your task using " "the information you already have.", path=path, - already_read=hits + 1, - ) + already_read=hits + 1) return json.dumps({ "status": "unchanged", @@ -614,8 +596,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str = if _is_blocked_device(path, base_dir=device_base): return tool_error( f"Cannot read '{path}': this is a device file that would " - "block or produce infinite output." - ) + "block or produce infinite output.") _resolved = _resolve_path_for_task(path, task_id) @@ -629,9 +610,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str = f"'{path}' is {kind}, not a regular file — reading " "it would block indefinitely, so no read was " "attempted. Use terminal utilities if you need to " - "interact with it." - ), - }) + "interact with it.")}) extracted = _read_extracted_document(path, _resolved, offset, limit, task_id) if extracted is not None: @@ -642,8 +621,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str = if has_binary_extension(str(_resolved)): return tool_error( f"Cannot read binary file '{path}' ({_resolved.suffix.lower()}). " - "Use vision_analyze for images, or terminal to inspect binary files." - ) + "Use vision_analyze for images, or terminal to inspect binary files.") # Hermes internal denylist: blocks prompt injection via catalog/hub # metadata and credential stores under HERMES_HOME. Pass the resolved @@ -692,8 +670,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str = if len(result.content or "") > max_chars: result.content = _apply_char_budget( result_dict, result.content or "", offset, - result_dict.get("total_lines", "unknown"), max_chars, - ) + result_dict.get("total_lines", "unknown"), max_chars) if result.content: result.content = redact_sensitive_text(result.content, file_read=True) result_dict["content"] = result.content @@ -703,8 +680,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str = result_dict.setdefault("_hint", ( f"This file is large ({file_size:,} bytes). " "Consider reading only the section you need with offset and limit " - "to keep context usage efficient." - )) + "to keep context usage efficient.")) count = _record_successful_read(task_data, task_id, path, resolved_str, offset, limit, dedup_key, partial=(offset > 1) or bool(result_dict.get("truncated"))) @@ -714,14 +690,12 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str = "The content has NOT changed. You already have this information. " "STOP re-reading and proceed with your task.", path=path, - already_read=count, - ) + already_read=count) if count >= 3: result_dict["_warning"] = ( f"You have read this exact file region {count} times consecutively. " "The content has not changed since your last read. Use the information you already have. " - "If you are stuck in a loop, stop reading and proceed with writing or responding." - ) + "If you are stuck in a loop, stop reading and proceed with writing or responding.") return json.dumps(result_dict, ensure_ascii=False) except Exception as e: return tool_error(str(e)) @@ -764,8 +738,7 @@ def _write_precheck_error(paths: list[str], content_paths: list[str], task_id: s return _first_error( *(lambda p=p: _check_binary_document_write(p, task_id) for p in content_paths), lambda: _check_protected_instruction_write(paths, task_id), - lambda: _check_approval_required_write(paths, task_id), - ) + lambda: _check_approval_required_write(paths, task_id)) def _edit_warnings(paths: list[str], path_to_resolved: dict, task_id: str) -> list[str]: @@ -813,8 +786,7 @@ def write_file_tool(path: str, content: str, task_id: str = "default", "Refusing to write internal read_file display text as file content. " "Strip read_file line-number prefixes or reconstruct the intended " "file contents before writing." - ) if _is_internal_file_tool_content(content) else None, - ) + ) if _is_internal_file_tool_content(content) else None) if err: return tool_error(err) try: @@ -873,8 +845,7 @@ def _collect_v4a_header_paths(patch: str) -> tuple[list[str], list[str]] | str: f"V4A patch header contains '..' traversal: {v4a_path!r}. " "Use the agent's cwd-relative path (no '..') or an absolute " "path in '*** Update File:' / '*** Add File:' / " - "'*** Delete File:' / '*** Move File:' headers." - ) + "'*** Delete File:' / '*** Move File:' headers.") paths.append(v4a_path) if writes_text: content_paths.append(v4a_path) @@ -958,13 +929,11 @@ def patch_tool(mode: str = "replace", path: str = None, old_string: str = None, "content, (2) use a longer / more unique old_string with " "surrounding context lines, or (3) use write_file to " "replace the entire file if the targeted region is hard " - "to anchor." - ) + "to anchor.") elif "Did you mean one of these sections?" not in str(result_dict["error"]): result_dict["_hint"] = ( "old_string not found. Use read_file to verify the current " - "content, or search_files to locate the text." - ) + "content, or search_files to locate the text.") return json.dumps(result_dict, ensure_ascii=False) except Exception as e: return tool_error(str(e)) @@ -983,8 +952,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".", search_key = ("search", pattern, target, str(path), file_glob or "", limit, offset) with _read_tracker_lock: task_data = _read_tracker.setdefault(task_id, { - "last_key": None, "consecutive": 0, "read_history": set(), - }) + "last_key": None, "consecutive": 0, "read_history": set()}) count = _bump_consecutive(task_data, search_key) if count >= 4: @@ -993,8 +961,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".", "The results have NOT changed. You already have this information. " "STOP re-searching and proceed with your task.", pattern=pattern, - already_searched=count, - ) + already_searched=count) try: resolved_search_path = str(_resolve_path_for_task(path, task_id)) @@ -1016,8 +983,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".", result = _get_file_ops(task_id).search( pattern=pattern, path=path, target=target, file_glob=file_glob, - limit=limit, offset=offset, output_mode=output_mode, context=context - ) + limit=limit, offset=offset, output_mode=output_mode, context=context) omitted = _filter_read_blocked_search_results(result, task_id) for m in getattr(result, "matches", None) or (): if getattr(m, "content", None): @@ -1027,8 +993,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".", if omitted: result_dict["_omitted"] = ( f"{omitted} result(s) omitted because they target credential, " - "token, cache, or secret-bearing environment files." - ) + "token, cache, or secret-bearing environment files.") # No early return on a cached miss — same rationale as the read path. _search_err = result_dict.get("error") or "" @@ -1038,8 +1003,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".", if count >= 3: result_dict["_warning"] = ( f"You have run this exact search {count} times consecutively. " - "The results have not changed. Use the information you already have." - ) + "The results have not changed. Use the information you already have.") result_json = json.dumps(result_dict, ensure_ascii=False) if result_dict.get("truncated"): diff --git a/tools/file_tools_paths.py b/tools/file_tools_paths.py index d25b0f6247..dda9a90573 100644 --- a/tools/file_tools_paths.py +++ b/tools/file_tools_paths.py @@ -47,8 +47,7 @@ def _terminal_env_type_for_task(task_id: str = "default") -> str: """Best-effort terminal backend type for path-resolution decisions.""" try: from tools.terminal_tool import ( - _active_environments, _env_lock, _get_env_config, _resolve_container_task_id, - ) + _active_environments, _env_lock, _get_env_config, _resolve_container_task_id) try: container_key = _resolve_container_task_id(task_id) @@ -147,10 +146,7 @@ def _authoritative_workspace_root(task_id: str = "default") -> str | None: def _resolve_base_dir( - task_id: str = "default", - *, - container_paths: bool | None = None, -) -> Path | PurePosixPath: + task_id: str = "default", *, container_paths: bool | None = None) -> Path | PurePosixPath: """Return the ABSOLUTE base directory for resolving relative paths. Uses ``_authoritative_workspace_root`` (live cwd → registered override → @@ -249,7 +245,6 @@ def _path_resolution_warning(filepath: str, resolved: Path, task_id: str = "defa f"OUTSIDE the active workspace ({str(root)!r}). The edit will land in " f"a different directory than the terminal's cwd. If this is not " f"intended (e.g. a git-worktree session writing into the main " - f"checkout), pass an absolute path under the workspace instead." - ) + f"checkout), pass an absolute path under the workspace instead.") except Exception: return None diff --git a/tools/file_tools_read_tracking.py b/tools/file_tools_read_tracking.py index de50abe7af..00435e4caf 100644 --- a/tools/file_tools_read_tracking.py +++ b/tools/file_tools_read_tracking.py @@ -44,8 +44,7 @@ def _task_data(task_id: str) -> dict: code paths (or injected by tests) may lack the newer containers. """ task_data = _read_tracker.setdefault(task_id, { - "last_key": None, "consecutive": 0, "read_history": set(), - }) + "last_key": None, "consecutive": 0, "read_history": set()}) for key in ("dedup", "dedup_hits", "read_timestamps"): task_data.setdefault(key, {}) return task_data @@ -80,8 +79,7 @@ def _cap_read_tracker_data(task_data: dict) -> None: ("dedup", _DEDUP_CAP), ("dedup_hits", _DEDUP_CAP), ("read_timestamps", _READ_TIMESTAMPS_CAP), - ("not_found", _NOT_FOUND_CAP), - ): + ("not_found", _NOT_FOUND_CAP)): container = task_data.get(key) if container is not None and len(container) > cap: _evict_oldest(container, cap) @@ -237,8 +235,7 @@ def _check_file_staleness(filepath: str, task_id: str) -> str | None: return ( f"Warning: {filepath} was modified since you last read it " "(external edit or concurrent agent). The content you read may be " - "stale. Consider re-reading the file to verify before writing." - ) + "stale. Consider re-reading the file to verify before writing.") return None diff --git a/tools/file_tools_write_guards.py b/tools/file_tools_write_guards.py index f17db4ee23..9ce02b4a08 100644 --- a/tools/file_tools_write_guards.py +++ b/tools/file_tools_write_guards.py @@ -21,8 +21,7 @@ from tools.file_tools_paths import _expand_tilde, _resolve_path_for_task _SENSITIVE_PATH_PREFIXES = ( "/etc/", "/boot/", "/usr/lib/systemd/", "/private/etc/", - "/private/var/db/", "/private/var/root/", -) + "/private/var/db/", "/private/var/root/") _SENSITIVE_EXACT_PATHS = {"/var/run/docker.sock", "/run/docker.sock"} _hermes_config_resolved: str | None = None @@ -77,8 +76,7 @@ def _check_sensitive_path(filepath: str, task_id: str = "default") -> str | None if any(c.startswith(_SENSITIVE_PATH_PREFIXES) or c in _SENSITIVE_EXACT_PATHS for c in candidates): return ( f"Refusing to write to sensitive system path: {filepath}\n" - "Use the terminal tool with sudo if you need to modify system files." - ) + "Use the terminal tool with sudo if you need to modify system files.") # approvals.mode and other security settings live in config.yaml; a # prompt-injected agent could silently disable exec approval by editing it. hermes_config = _get_hermes_config_resolved() @@ -86,8 +84,7 @@ def _check_sensitive_path(filepath: str, task_id: str = "default") -> str | None return ( f"Refusing to write to Hermes config file: {filepath}\n" "Agent cannot modify security-sensitive configuration. " - "Edit ~/.hermes/config.yaml directly or use 'hermes config' instead." - ) + "Edit ~/.hermes/config.yaml directly or use 'hermes config' instead.") return None @@ -100,8 +97,7 @@ def _check_sensitive_path(filepath: str, task_id: str = "default") -> str | None # human approval — even under --yolo — and fail closed without a human channel. # Basenames match in ANY directory and case-insensitively (loaders probe variants). _PROTECTED_INSTRUCTION_BASENAMES = frozenset({ - "agents.md", "claude.md", "soul.md", ".cursorrules", -}) + "agents.md", "claude.md", "soul.md", ".cursorrules"}) def _protected_instruction_config() -> tuple[bool, list[str]]: @@ -182,15 +178,13 @@ def _request_protected_instruction_approval(reasons: list[str], task_id: str = " description = ( f"Write to protected agent-instruction file(s): {targets}. " "These files steer future agent behavior; approval is always " - "required (not bypassed by auto-approve)." - ) + "required (not bypassed by auto-approve).") display = f"" blocked = ( f"BLOCKED: write to protected agent-instruction file(s) ({targets}) " "{why} The user has NOT consented to this write. Do NOT retry it or " "attempt the same edit via another path (terminal, execute_code, " - "etc.)." - ) + "etc.).") timed_out = blocked.format(why="approval prompt timed out without a user response. Silence is not consent.") denied = blocked.format(why="was denied by the user.") @@ -215,8 +209,7 @@ def _request_protected_instruction_approval(reasons: list[str], task_id: str = " "pattern_keys": ["protected_instruction_file"], "description": description, "allow_permanent": False, - "allow_session": False, - } + "allow_session": False} decision = _approval._await_gateway_decision(session_key, notify_cb, approval_data, surface="gateway") if decision.get("notify_failed"): return blocked.format(why="requires approval but the approval request could not be delivered.") @@ -282,13 +275,11 @@ def _check_approval_required_write(paths: list[str], task_id: str = "default") - description = ( f"Write to SSH client config file(s): {display_targets}. " "The SSH config can carry ProxyCommand / Match exec directives that " - "run commands, so writes require your approval." - ) + "run commands, so writes require your approval.") blocked = ( f"BLOCKED: write to SSH config file(s) ({display_targets}) " "{why} Do NOT retry it via another path (terminal, execute_code) " - "without the user's explicit consent." - ) + "without the user's explicit consent.") try: import tools.approval as _approval @@ -307,8 +298,7 @@ def _check_approval_required_write(paths: list[str], task_id: str = "default") - "approve in config.yaml."), autoapprove_log_prefix="ssh_config_write", fail_closed_when_no_human=True, - no_human_block_message=blocked.format(why=_NO_HUMAN), - ) + no_human_block_message=blocked.format(why=_NO_HUMAN)) if result.get("approved"): return None return result.get("message") or blocked.format(why="was denied.") @@ -318,8 +308,7 @@ def _get_container_mirror_prefix_for_task(task_id: str = "default") -> str | Non """Return the container-side Hermes mirror prefix for persistent Docker file tools.""" try: from tools.terminal_tool import ( - _active_environments, _env_lock, _get_env_config, _resolve_container_task_id, - ) + _active_environments, _env_lock, _get_env_config, _resolve_container_task_id) container_key = _resolve_container_task_id(task_id) with _env_lock: env = _active_environments.get(container_key) or _active_environments.get(task_id) @@ -372,8 +361,7 @@ def _check_binary_document_write(filepath: str, task_id: str = "default") -> str "corrupt the file (read_file showed you EXTRACTED text, not the real " "bytes). Use the docx/xlsx/powerpoint skills or a library like " "python-docx/openpyxl/python-pptx via the terminal to create or edit " - "this document." - ) + "this document.") if is_pdf_path(filepath): try: resolved = Path(_resolve_path_for_task(filepath, task_id)) @@ -386,8 +374,7 @@ def _check_binary_document_write(filepath: str, task_id: str = "default") -> str "read_file showed you EXTRACTED text, not the real bytes — writing " "text back would destroy the document. Use the pdf skill or a PDF " "library via the terminal to modify it. (Creating a NEW .pdf file " - "is allowed.)" - ) + "is allowed.)") except OSError: pass return None @@ -399,8 +386,7 @@ def _check_binary_document_write(filepath: str, task_id: str = "default") -> str _READ_DEDUP_STATUS_MESSAGE = ( "File unchanged since last read. The content from " "the earlier read_file result in this conversation is " - "still current — refer to that instead of re-reading." -) + "still current — refer to that instead of re-reading.") def _is_internal_file_status_text(content: str) -> bool: diff --git a/tools/fuzzy_match.py b/tools/fuzzy_match.py index e0c4e98f07..5ab92b8cb8 100644 --- a/tools/fuzzy_match.py +++ b/tools/fuzzy_match.py @@ -18,8 +18,7 @@ Span = tuple[int, int] IDENTICAL_STRINGS_ERROR = ( "No edit was applied because old_string and new_string are identical. " "Provide the existing text to replace in old_string and the changed " - "replacement text in new_string." -) + "replacement text in new_string.") UNICODE_MAP = { "\u201c": '"', "\u201d": '"', # smart double quotes @@ -32,8 +31,7 @@ UNICODE_MAP = { "\u2000": " ", "\u2001": " ", "\u2002": " ", "\u2003": " ", "\u2004": " ", "\u2005": " ", "\u2006": " ", "\u2007": " ", "\u2008": " ", "\u2009": " ", "\u200a": " ", "\u202f": " ", - "\u205f": " ", "\u3000": " ", -} + "\u205f": " ", "\u3000": " "} def _unicode_normalize(text: str) -> str: @@ -58,8 +56,7 @@ def _window_spans(content: str, content_lines: list[str], n: int, """Spans of every ``n``-line window starting at ``i`` for which ``accept(i)``.""" return [ _calculate_line_positions(content_lines, i, i + n, len(content)) - for i in range(len(content_lines) - n + 1) if accept(i) - ] + for i in range(len(content_lines) - n + 1) if accept(i)] def _match_transformed_lines(content: str, pattern: str, @@ -257,8 +254,7 @@ def _strategy_block_anchor(content: str, pattern: str) -> list[Span]: potential_matches = { i for i in range(len(norm_content_lines) - n + 1) if norm_content_lines[i].strip() == first_line - and norm_content_lines[i + n - 1].strip() == last_line - } + and norm_content_lines[i + n - 1].strip() == last_line} # Looser thresholds (0.10/0.30) matched unrelated blocks; these are the safe floor. threshold = 0.50 if len(potential_matches) == 1 else 0.70 pattern_middle = '\n'.join(pattern_lines[1:-1]) @@ -300,8 +296,7 @@ def _strategy_context_aware(content: str, pattern: str) -> list[Span]: return False return all( not p_line.strip() or _sim(p_line.strip(), c_line.strip()) >= 0.80 - for p_line, c_line in zip(pattern_lines, block_lines) - ) + for p_line, c_line in zip(pattern_lines, block_lines)) return _window_spans(content, content_lines, n, accept) @@ -316,8 +311,7 @@ STRATEGIES: list[tuple[str, Callable[[str, str], list[Span]]]] = [ ("trimmed_boundary", _strategy_trimmed_boundary), ("unicode_normalized", _strategy_unicode_normalized), ("block_anchor", _strategy_block_anchor), - ("context_aware", _strategy_context_aware), -] + ("context_aware", _strategy_context_aware)] # Matches from these only *approximately* resemble old_string — fine for one # unique replacement, never safe under replace_all. @@ -386,15 +380,13 @@ def fuzzy_find_and_replace(content: str, old_string: str, new_string: str, return content, 0, None, ( f"Found {len(matches)} matches for old_string. " f"Provide more context to make it unique, or use replace_all=True. " - f"Matches:\n{locations}" - ) + f"Matches:\n{locations}") if replace_all and len(matches) > 1 and strategy_name in SIMILARITY_STRATEGIES: return content, 0, None, ( f"Found {len(matches)} approximate matches via the " f"'{strategy_name}' strategy; replace_all only applies to exact " f"matches. Provide the precise text (whitespace included) so an " - f"exact/line-trimmed match can be made." - ) + f"exact/line-trimmed match can be made.") # Non-exact matches came through some normalization, so new_string may # carry serialization drift the file doesn't have. @@ -408,8 +400,7 @@ def fuzzy_find_and_replace(content: str, old_string: str, new_string: str, effective_new = _preserve_unicode_in_replacement(content, matches, old_string, effective_new) new_content = _apply_replacements( content, matches, effective_new, - old_string=old_string if strategy_name != "exact" else None, - ) + old_string=old_string if strategy_name != "exact" else None) return new_content, len(matches), strategy_name, None return content, 0, None, "Could not find a match for old_string in the file" @@ -441,8 +432,7 @@ def _detect_escape_drift(content: str, matches: list[Span], f"serialization artifact where an apostrophe or quote got " f"prefixed with a spurious backslash. Re-read the file with " f"read_file and pass old_string/new_string without " - f"backslash-escaping {plain!r} characters." - ) + f"backslash-escaping {plain!r} characters.") return _detect_backslash_doubling(matched_regions, old_string, new_string) @@ -476,8 +466,7 @@ def _detect_backslash_doubling(matched_regions: str, old_string: str, "were JSON-escaped one extra time; applying new_string verbatim would " "double every backslash in the file. Re-read the file with read_file " "and resend old_string/new_string with the backslash counts exactly " - "as they appear in the file." - ) + "as they appear in the file.") def _maybe_unescape_new_string(new_string: str, content: str, matches: list[Span]) -> str: @@ -626,8 +615,7 @@ def find_closest_lines(old_string: str, content: str, context_lines: int = 2, ma continue seen_ranges.add((start, end)) parts.append("\n".join( - f"{start + j + 1:4d}| {content_lines[start + j]}" for j in range(end - start) - )) + f"{start + j + 1:4d}| {content_lines[start + j]}" for j in range(end - start))) if not parts: return "" result = "\n---\n".join(parts) @@ -640,8 +628,7 @@ def find_closest_lines(old_string: str, content: str, context_lines: int = 2, ma "\n\nWhitespace difference detected (→ = tab, · = space):\n" f" file has: {_visualize_whitespace(best_line)}\n" f" you sent: {_visualize_whitespace(old_lines[0])}\n" - "Use the exact whitespace shown in 'file has'." - ) + "Use the exact whitespace shown in 'file has'.") return result