From ec49ae7c0fe012523f205d715083251ab83bb62e Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 10:06:41 -0700 Subject: [PATCH] refactor(tools): wire file_operations to extracted common/lint/search modules; restore literal security pins in lazy_deps --- .../test_file_safety_container_mirror.py | 2 +- tests/tools/test_accretion_caps.py | 10 +- tools/file_operations.py | 1853 +---------------- tools/file_operations_search.py | 2 +- tools/lazy_deps.py | 20 +- 5 files changed, 76 insertions(+), 1811 deletions(-) diff --git a/tests/agent/test_file_safety_container_mirror.py b/tests/agent/test_file_safety_container_mirror.py index bf42a5d882..196e4bb966 100644 --- a/tests/agent/test_file_safety_container_mirror.py +++ b/tests/agent/test_file_safety_container_mirror.py @@ -72,7 +72,7 @@ class TestFileToolIntegration: """file_tools must catch the mirror path before creating DockerEnvironment.""" def test_guard_uses_current_docker_config_before_env_exists(self, monkeypatch): - import tools.file_tools as file_tools + import tools.file_tools_write_guards as file_tools monkeypatch.setattr( file_tools, diff --git a/tests/tools/test_accretion_caps.py b/tests/tools/test_accretion_caps.py index 123c051622..73e3fc77fe 100644 --- a/tests/tools/test_accretion_caps.py +++ b/tests/tools/test_accretion_caps.py @@ -30,8 +30,9 @@ class TestReadTrackerCaps: def test_read_history_capped(self, monkeypatch): """read_history set is bounded by _READ_HISTORY_CAP.""" from tools import file_tools as ft + from tools import file_tools_read_tracking as rt - monkeypatch.setattr(ft, "_READ_HISTORY_CAP", 10) + monkeypatch.setattr(rt, "_READ_HISTORY_CAP", 10) task_data = { "last_key": None, "consecutive": 0, @@ -46,10 +47,11 @@ class TestReadTrackerCaps: def test_live_cap_applied_after_read_add(self, tmp_path, monkeypatch): """Live read_file path enforces caps.""" from tools import file_tools as ft + from tools import file_tools_read_tracking as rt - monkeypatch.setattr(ft, "_READ_HISTORY_CAP", 3) - monkeypatch.setattr(ft, "_DEDUP_CAP", 3) - monkeypatch.setattr(ft, "_READ_TIMESTAMPS_CAP", 3) + monkeypatch.setattr(rt, "_READ_HISTORY_CAP", 3) + monkeypatch.setattr(rt, "_DEDUP_CAP", 3) + monkeypatch.setattr(rt, "_READ_TIMESTAMPS_CAP", 3) # Create 10 distinct files and read each once. for i in range(10): diff --git a/tools/file_operations.py b/tools/file_operations.py index fbfd06cab8..5e5fa28677 100644 --- a/tools/file_operations.py +++ b/tools/file_operations.py @@ -35,8 +35,8 @@ import hashlib import json import unicodedata from abc import ABC, abstractmethod -from dataclasses import dataclass, field -from typing import Optional, List, Dict, Any, ClassVar + +from typing import Optional, Dict from pathlib import Path from tools.binary_extensions import BINARY_EXTENSIONS @@ -46,6 +46,56 @@ from agent.file_safety import ( get_write_denied_error, is_write_denied as _shared_is_write_denied, ) +from tools.file_operations_common import ( # noqa: F401 (re-exported) + DEFAULT_READ_LIMIT, + DEFAULT_READ_OFFSET, + DEFAULT_SEARCH_LIMIT, + DEFAULT_SEARCH_OFFSET, + ExecuteResult, + LintResult, + PatchResult, + ReadResult, + SearchMatch, + SearchResult, + WriteResult, + _FENCE_MARKER_RE, + _OSC_SEQUENCE_RE, + _UTF8_BOM, + _coerce_int, + _detect_line_ending, + _has_bom, + _normalize_line_endings, + _strip_bom, + _strip_terminal_fence_leaks, + normalize_read_pagination, + normalize_search_pagination, +) +from tools.file_operations_lint import ( # noqa: F401 (re-exported) + LINTERS, + LintMixin, + _FAIL_CLOSED_INPROC_EXTS, + _LINTER_UNUSABLE_PATTERNS, + _SHELL_LINTER_LSP_REDUNDANT, + _lint_json_inproc, + _lint_python_inproc, + _lint_toml_inproc, + _lint_yaml_inproc, + _looks_like_linter_unusable, +) +from tools.file_operations_search import ( # noqa: F401 (re-exported) + SearchMixin, + _MACOS_TCC_PROTECTED_HOME_DIRS, + _REGEX_NEWLINE_ESCAPE_RE, + _SEARCH_OUTPUT_RE, + _SEARCH_TIMEOUT_MARKER_RE, + _is_line_oriented_newline_error, + _macos_protected_search_exclusions, + _maybe_warn_line_oriented_newline_pattern, + _parse_search_context_line, + _pattern_has_regex_newline, + _search_stdout_and_limit, + _split_tool_diagnostics, +) # --------------------------------------------------------------------------- @@ -54,148 +104,12 @@ from agent.file_safety import ( _HOME = str(Path.home()) -_MACOS_TCC_PROTECTED_HOME_DIRS = ( - "Desktop", - "Documents", - "Downloads", - "Library", - "Movies", - "Music", - "Pictures", -) - - -def _macos_protected_search_exclusions( - path: str, - *, - cwd: Optional[str] = None, - home: Optional[str] = None, - platform: Optional[str] = None, -) -> List[str]: - """Return protected home directories below a broad macOS search root. - - Direct searches inside a protected directory remain allowed. Only an - ancestor search (for example ``$HOME`` or ``/Users``) receives exclusions, - preventing recursive tools from triggering unattended TCC prompts. - """ - if (platform or sys.platform) != "darwin": - return [] - - home_path = Path(home or Path.home()).expanduser() - root = Path(path).expanduser() - if not root.is_absolute(): - root = Path(cwd or os.getcwd()) / root - root = Path(os.path.normpath(str(root))) - home_path = Path(os.path.normpath(str(home_path))) - - exclusions: List[str] = [] - for dirname in _MACOS_TCC_PROTECTED_HOME_DIRS: - protected = home_path / dirname - try: - relative = protected.relative_to(root) - except ValueError: - continue - if relative.parts: - exclusions.append(relative.as_posix()) - return exclusions - WRITE_DENIED_PATHS = build_write_denied_paths(_HOME) WRITE_DENIED_PREFIXES = build_write_denied_prefixes(_HOME) -_OSC_SEQUENCE_RE = re.compile(r"\x1b\][^\x07\x1b]*(?:\x07|\x1b\\)") -_FENCE_MARKER_RE = re.compile(r"'?\x07?__HERMES_FENCE_[A-Za-z0-9]+__\x07?'?") - - -def _strip_terminal_fence_leaks(text: str) -> str: - """Strip leaked terminal fence wrappers from file read output.""" - if not text: - return text - - cleaned_lines: List[str] = [] - for line in text.splitlines(keepends=True): - had_terminal_wrapper = "__HERMES_FENCE_" in line or "\x1b]" in line - cleaned = _OSC_SEQUENCE_RE.sub("", line) - cleaned = _FENCE_MARKER_RE.sub("", cleaned) - cleaned = cleaned.replace("\x07", "") - if had_terminal_wrapper and cleaned.strip("'\r\n\t ") == "": - continue - cleaned_lines.append(cleaned) - return "".join(cleaned_lines) - - -def _detect_line_ending(sample: str) -> Optional[str]: - """Return the dominant line ending in ``sample`` or None if undetermined. - - Looks at the first few line breaks and picks ``\\r\\n`` if any are - present (Windows / DOS), otherwise ``\\n`` (Unix). Returns ``None`` - for empty / single-line content where we can't tell. Used to - preserve the file's original line endings across write_file and - patch operations — without this the agent's bare-LF tool args - silently normalize Windows-line-ending files, and patch produces - mixed endings when only a substituted region changes. - """ - if not sample: - return None - # Look at the first chunk — enough to tell, cheap to scan. - head = sample[:4096] - if "\r\n" in head: - return "\r\n" - if "\n" in head: - return "\n" - return None - - -def _normalize_line_endings(text: str, target: str) -> str: - """Convert all line endings in ``text`` to ``target`` (``\\n`` or ``\\r\\n``). - - Idempotent: ``_normalize_line_endings(_normalize_line_endings(x, "\\r\\n"), "\\r\\n") == _normalize_line_endings(x, "\\r\\n")``. - Strips lone ``\\r`` characters as well, so mixed-ending content is - homogenized in a single pass. - """ - # First collapse to LF (handle CRLF and lone CR), then expand if target - # is CRLF. Order matters: doing the replacements separately would - # double-convert a CRLF -> LFLF. - lf_normalized = text.replace("\r\n", "\n").replace("\r", "\n") - if target == "\n": - return lf_normalized - if target == "\r\n": - return lf_normalized.replace("\n", "\r\n") - return text - - -# UTF-8 byte order mark. Some Windows editors (Notepad, older Visual Studio, -# some PowerShell redirects) prepend this invisible 3-byte marker -# (EF BB BF == U+FEFF) to UTF-8 text files. It renders as nothing but is a -# real character at the start of the decoded string, so without handling it: -# - read_file would surface a stray U+FEFF as the first character (the -# model sees a phantom char before `import ...`), and -# - patch matches against the true first line would miss, and write_file -# would silently drop or double the marker on rewrite. -# We strip it on read so the model sees clean content, and restore it on -# write when the original file had one — exactly mirroring the line-ending -# preservation above (detect on disk, preserve across the edit). -_UTF8_BOM = "\ufeff" - - -def _strip_bom(text: str) -> tuple[str, bool]: - """Return (text-without-leading-BOM, had_bom). - - Only a single leading BOM is stripped; a BOM appearing mid-content is - left alone (it's legitimate data there, not a file marker). - """ - if text and text.startswith(_UTF8_BOM): - return text[len(_UTF8_BOM):], True - return text, False - - -def _has_bom(text: Optional[str]) -> bool: - """True if ``text`` begins with a UTF-8 BOM.""" - return bool(text) and text.startswith(_UTF8_BOM) - - def _is_write_denied(path: str) -> bool: """Return True if path is on the write deny list.""" return _shared_is_write_denied(path) @@ -205,293 +119,6 @@ def _is_write_denied(path: str) -> bool: # Result Data Classes # ============================================================================= -@dataclass -class ReadResult: - """Result from reading a file.""" - content: str = "" - total_lines: int = 0 - file_size: int = 0 - truncated: bool = False - hint: Optional[str] = None - is_binary: bool = False - is_image: bool = False - base64_content: Optional[str] = None - mime_type: Optional[str] = None - dimensions: Optional[str] = None # For images: "WIDTHxHEIGHT" - error: Optional[str] = None - similar_files: List[str] = field(default_factory=list) - - def to_dict(self) -> dict: - return {k: v for k, v in self.__dict__.items() if v is not None and v != []} - - -@dataclass -class WriteResult: - """Result from writing a file.""" - bytes_written: int = 0 - dirs_created: bool = False - # True when the on-disk sha256 matched the intended content after the - # write (post-write verification). None when the backend couldn't - # verify (no sha256sum). A mismatch never reaches the caller as a - # flag — it becomes a hard error. - verified: Optional[bool] = None - lint: Optional[Dict[str, Any]] = None - # Semantic diagnostics from the LSP layer, when applicable. Kept in - # its own field (not folded into ``lint``) so the model and any - # downstream parsers can read syntax errors and semantic errors as - # separate signals. ``None`` when LSP is disabled, when the file - # isn't in a git workspace, or when no diagnostics were introduced - # by this edit. - lsp_diagnostics: Optional[str] = None - error: Optional[str] = None - warning: Optional[str] = None - - def to_dict(self) -> dict: - return {k: v for k, v in self.__dict__.items() if v is not None} - - -@dataclass -class PatchResult: - """Result from patching a file.""" - success: bool = False - diff: str = "" - files_modified: List[str] = field(default_factory=list) - files_created: List[str] = field(default_factory=list) - files_deleted: List[str] = field(default_factory=list) - lint: Optional[Dict[str, Any]] = None - # See :class:`WriteResult.lsp_diagnostics`. - lsp_diagnostics: Optional[str] = None - error: Optional[str] = None - # Set on success-shaped no-ops: the requested edit was already present - # in the file, so nothing was written. Carries a short note for the - # model explaining why no diff is included. - no_change: bool = False - note: Optional[str] = None - - def to_dict(self) -> dict: - result: Dict[str, Any] = {"success": self.success} - if self.no_change: - result["no_change"] = True - if self.note: - result["note"] = self.note - if self.diff: - result["diff"] = self.diff - if self.files_modified: - result["files_modified"] = self.files_modified - if self.files_created: - result["files_created"] = self.files_created - if self.files_deleted: - result["files_deleted"] = self.files_deleted - if self.lint: - result["lint"] = self.lint - if self.lsp_diagnostics: - result["lsp_diagnostics"] = self.lsp_diagnostics - if self.error: - result["error"] = self.error - return result - - -@dataclass -class SearchMatch: - """A single search match.""" - path: str - line_number: int - content: str - mtime: float = 0.0 # Modification time for sorting - - -@dataclass -class SearchResult: - """Result from searching.""" - matches: List[SearchMatch] = field(default_factory=list) - files: List[str] = field(default_factory=list) - counts: Dict[str, int] = field(default_factory=dict) - total_count: int = 0 - truncated: bool = False - limit_reason: Optional[str] = None - warning: Optional[str] = None - error: Optional[str] = None - - # Densify content-mode matches into a path-grouped text block above this - # many matches. Below it, the verbose array is already compact enough that - # the path-grouping header costs more than it saves. - _DENSIFY_MIN_MATCHES: ClassVar[int] = 5 - - def _densify_matches(self) -> Optional[str]: - """Render content-mode matches as a compact, path-grouped text block. - - The verbose form repeats the ``{"path","line","content"}`` keys and the - full path string for every match. This groups consecutive matches by - path (path printed once, then `` : `` rows), which is - lossless — every path, line number, and content byte is preserved — and - readable by the model without any decode step. - - Returns ``None`` when densification is not worthwhile (too few matches), - so the caller falls back to the verbose array. - """ - if len(self.matches) < self._DENSIFY_MIN_MATCHES: - return None - # ripgrep emits matches path-ordered (all hits in a file are - # consecutive), so grouping on path change collapses each file to a - # single header without reordering results. - lines: list[str] = [] - current_path: Optional[str] = None - for m in self.matches: - if m.path != current_path: - lines.append(m.path) - current_path = m.path - # rstrip trailing whitespace only; leading indentation in code is - # meaningful and preserved verbatim after the ": " prefix. - lines.append(f" {m.line_number}: {m.content.rstrip()}") - return "\n".join(lines) - - def to_dict(self, densify: bool = False) -> dict: - result: dict[str, object] = {"total_count": self.total_count} - if self.matches: - dense = self._densify_matches() if densify else None - if dense is not None: - # Self-describing: the format key tells the model how to read - # the block so it never has to guess the shape. - result["matches_format"] = ( - "path-grouped: each file path on its own line, followed by " - "indented ': ' rows for matches in that file" - ) - result["matches_text"] = dense - else: - result["matches"] = [ - {"path": m.path, "line": m.line_number, "content": m.content} - for m in self.matches - ] - if self.files: - result["files"] = self.files - if self.counts: - result["counts"] = self.counts - if self.truncated: - result["truncated"] = True - if self.limit_reason: - result["limit_reason"] = self.limit_reason - if self.warning: - result["warning"] = self.warning - if self.error: - result["error"] = self.error - return result - - -@dataclass -class LintResult: - """Result from linting a file.""" - success: bool = True - skipped: bool = False - output: str = "" - message: str = "" - - def to_dict(self) -> dict: - if self.skipped: - return {"status": "skipped", "message": self.message} - result = {"status": "ok" if self.success else "error", "output": self.output} - if self.message: - result["message"] = self.message - return result - - -@dataclass -class ExecuteResult: - """Result from executing a shell command.""" - stdout: str = "" - exit_code: int = 0 - - -_SEARCH_TIMEOUT_MARKER_RE = re.compile(r"\n?\[Command timed out after \d+s\]\s*$") - - -def _search_stdout_and_limit(result: ExecuteResult) -> tuple[str, Optional[str]]: - """Return stdout cleaned for parsing and a limit reason for search timeouts.""" - if result.exit_code == 124: - return _SEARCH_TIMEOUT_MARKER_RE.sub("", result.stdout), "search_timeout" - return result.stdout, None - - -def _split_tool_diagnostics(output: str) -> tuple[str, str]: - """Separate rg/grep diagnostic lines from real match output. - - ``_exec`` runs commands with ``stderr=subprocess.STDOUT``, so error and - warning text from ``rg``/``grep`` is interleaved with match lines in a - single stream. Diagnostics must not be parsed as matches, and on a hard - failure they are the error message to surface. - - Returns ``(diagnostics, payload)`` where ``payload`` contains only lines - that look like real search output — a match line (``file:line:content``), - a files-only path, a count line, or a context line/separator. Everything - else (tool-prefixed errors, rg's multi-line ``regex parse error`` block - with its indented carets, blank lines) is folded into ``diagnostics``. - - Classifying by *shape* rather than by error prefix is what lets the - exit-2 guard distinguish a pure failure (no usable payload → surface the - error) from a partial failure (some files matched, one was unreadable → - keep the matches). It also means error text can never be mis-parsed as a - match, a latent bug that predates the exit-code fix. - """ - diagnostics: list[str] = [] - payload: list[str] = [] - for line in output.split('\n'): - if not line.strip(): - continue - # Tool diagnostics always carry the ": " prefix (e.g. - # "rg: : Permission denied", "grep: Invalid regular - # expression", "rg: regex parse error:"). Check this first: a real - # match path can legitimately contain "-" (e.g. a tmp dir like - # ".../pytest-686/..."), which the shape regex would otherwise treat - # as a match line. - stripped = line.lstrip() - if stripped.startswith("rg: ") or stripped.startswith("grep: "): - diagnostics.append(line) - continue - # Otherwise classify by output shape. rg's regex-parse-error block - # also emits an indented caret line and a trailing "error: ..." line - # with no tool prefix; neither matches a search-output shape, so they - # fall through to diagnostics. - # match / count : ":<...>" (has a colon; rg -c uses path:count) - # files_only : "" (no whitespace, no leading colon) - # context line : "--" or the "--" group separator - if line == "--" or _SEARCH_OUTPUT_RE.match(line): - payload.append(line) - else: - diagnostics.append(line) - return '\n'.join(diagnostics), '\n'.join(payload) - - -# A real rg/grep output line starts with a path token and is followed by a -# ``:`` (match/count), a ``-`` (context), or nothing (files_only). Tool -# diagnostics ("rg: ...", "grep: ...", "error: ...", indented carets) never -# match because the path token forbids whitespace and a leading tool prefix -# like "rg" is followed by ": " (space) which the negated class rejects. -_SEARCH_OUTPUT_RE = re.compile(r'^([A-Za-z]:)?[^\s:][^\n]*?[:\-]\d|^[^\s:][^\s]*$') - - -def _parse_search_context_line(line: str) -> tuple[str, int, str] | None: - """Parse grep/rg context output in ``path-line-content`` format. - - Context lines are ambiguous because filenames may legitimately contain - ``--`` segments. Prefer the rightmost numeric separator so a path - like ``dir/file-12-name.py-8-context`` resolves to - ``dir/file-12-name.py`` line ``8`` instead of truncating at ``file``. - """ - if not line or line == "--": - return None - - match = None - for candidate in re.finditer(r'-(\d+)-', line): - match = candidate - - if match is None: - return None - - path = line[:match.start()] - if not path: - return None - - return path, int(match.group(1)), line[match.end():] - # ============================================================================= # Abstract Interface @@ -635,177 +262,6 @@ class FileOperations(ABC): # Image extensions (subset of binary that we can return as base64) IMAGE_EXTENSIONS = {'.png', '.jpg', '.jpeg', '.gif', '.webp', '.bmp', '.ico'} -# Shell-based linters by file extension. Invoked via _exec() with the -# filesystem path. Cover languages where a compile/type check needs an -# external toolchain (py_compile, node, tsc, go vet, rustfmt). -LINTERS = { - '.py': 'python -m py_compile {file} 2>&1', - '.js': 'node --check {file} 2>&1', - '.ts': 'npx tsc --noEmit {file} 2>&1', - '.go': 'go vet {file} 2>&1', - '.rs': 'rustfmt --check {file} 2>&1', -} - -# Extensions where the per-file shell linter is structurally weaker than -# a real LSP server AND produces phantom errors on real-world projects: -# -# - ``.ts``: ``tsc --noEmit FILE.ts`` ignores ``tsconfig.json`` and -# defaults to no-lib / ES5, so every ES2015+ stdlib reference -# (``Promise``, ``Map``, ``Set``, ``ReadonlySet``, ``Iterable``, -# ``Math.imul``, ``Number.isFinite``, etc.) reports as missing. This -# floods the agent's lint field with 20K+ tokens of false positives on -# every edit. No supported tsc flag fixes the single-file invocation; -# the canonical replacement is ``tsserver`` via LSP, which respects -# tsconfig and gives true diagnostics. -# -# ``.tsx`` is intentionally NOT in ``LINTERS`` (and therefore not -# here): it has no shell linter entry, so it falls through to the -# ``ext not in LINTERS`` skip case unchanged. Pre-PR behavior: -# ``.tsx`` was implicitly ``skipped``. Keeping it that way means -# ``.tsx`` edits with LSP disabled get no per-file syntax check -# (same as before this PR) instead of the broken ``tsc`` invocation -# that ``.ts`` used to get. When LSP is enabled, ``.tsx`` is covered -# by the LSP tier via ``_maybe_lsp_diagnostics`` exactly as ``.ts``. -# -# - ``.go``: ``go vet FILE.go`` fails outside a module / GOPATH with -# "cannot find package" — already partially handled by -# ``_LINTER_UNUSABLE_PATTERNS`` but only when the package error is the -# ONLY output; mixed real+phantom output still leaks through. -# ``gopls`` is the canonical replacement. -# -# - ``.rs``: ``rustfmt --check FILE.rs`` is style, not type-checking, and -# rejects non-Cargo project files. ``rust-analyzer`` is the canonical -# replacement. -# -# When the LSP service is configured AND ``enabled_for(path)`` for this -# extension's file, ``_check_lint`` skips the shell linter for these -# extensions — the ``lsp_diagnostics`` channel carries the real signal. -# Everything else in ``LINTERS`` (Python ``py_compile``, ``node --check``) -# is fast, file-local, and correct, so it runs unconditionally. -_SHELL_LINTER_LSP_REDUNDANT = frozenset({'.ts', '.go', '.rs'}) - - -# Patterns that indicate the linter base command exists on PATH but -# couldn't actually run — e.g. ``npx tsc`` when tsc isn't installed in -# node_modules, or rustfmt complaining there's no Cargo project. When -# any of these substrings appears in the linter output, ``_check_lint`` -# returns ``skipped`` instead of ``error`` so: -# -# 1. The write isn't flagged for a tooling problem the agent can't fix. -# 2. The LSP semantic tier still runs (it gates on success/skipped). -# -# Patterns are matched case-insensitively against linter stdout. -_LINTER_UNUSABLE_PATTERNS = { - 'npx': ( - # npx prints this banner when the package isn't installed locally - # AND it can't auto-install (no internet, registry off, etc.) or - # when the binary it tried to run is the wrong one. - 'this is not the tsc command you are looking for', - # npx with --no-install resolution failures - 'could not determine executable to run', - 'not found in npm registry', - ), - 'rustfmt': ( - # rustfmt outside a Cargo project - 'no input filename given', - 'error: not a workspace', - ), - 'go': ( - # ``go vet`` on a file outside a module / GOPATH - 'cannot find package', - 'go: cannot find main module', - ), -} - - -def _looks_like_linter_unusable(base_cmd: str, output: str) -> bool: - """Return True iff ``output`` from ``base_cmd`` indicates the linter - itself couldn't run (a tooling gap), as opposed to a real lint error - in the file being checked. - - ``base_cmd`` is the first word of the linter command line (``npx``, - ``rustfmt``, ``go``, ...). ``output`` is the stdout/stderr captured - from running it. - """ - patterns = _LINTER_UNUSABLE_PATTERNS.get(base_cmd) - if not patterns: - return False - lower = output.lower() - return any(p in lower for p in patterns) - - -def _lint_json_inproc(content: str) -> tuple[bool, str]: - """In-process JSON syntax check. Returns (ok, error_message).""" - import json as _json - try: - _json.loads(content) - return True, "" - except _json.JSONDecodeError as e: - return False, f"JSONDecodeError: {e.msg} (line {e.lineno}, column {e.colno})" - except Exception as e: # noqa: BLE001 — any parse failure is a lint failure - return False, f"{type(e).__name__}: {e}" - - -def _lint_yaml_inproc(content: str) -> tuple[bool, str]: - """In-process YAML syntax check. Returns (ok, error_message). - - Skipped gracefully if PyYAML isn't installed — YAML parsing is optional. - - Deliberately a *syntax-only* scan (``yaml.parse``), not ``safe_load``: - loading rejects perfectly valid YAML that merely isn't a single plain - document — multi-document streams (``---``-separated Kubernetes - manifests raise ``ComposerError``) and application-defined tags - (CloudFormation ``!Sub``/``!Ref``, Ansible ``!vault`` raise - ``ConstructorError``). Those are content conventions for whatever - consumes the file, not syntax errors, and this linter's verdict is - used as a fail-closed WRITE gate in ``write_file`` — a false positive - here refuses a legitimate write outright. ``yaml.parse`` still - catches real scanner/parser failures (unclosed quotes, bad - indentation, tab-mangled block maps). - """ - try: - import yaml as _yaml - except ImportError: - # PyYAML not available — skip silently, caller treats as no linter. - return True, "__SKIP__" - try: - for _event in _yaml.parse(content): - pass - return True, "" - except _yaml.YAMLError as e: - return False, f"YAMLError: {e}" - except Exception as e: # noqa: BLE001 - return False, f"{type(e).__name__}: {e}" - - -def _lint_toml_inproc(content: str) -> tuple[bool, str]: - """In-process TOML syntax check (stdlib tomllib, Python 3.11+).""" - import tomllib as _toml - - try: - _toml.loads(content) - return True, "" - except Exception as e: # tomllib raises TOMLDecodeError, a ValueError subclass - return False, f"{type(e).__name__}: {e}" - - -def _lint_python_inproc(content: str) -> tuple[bool, str]: - """In-process Python syntax check via ast.parse. - - Catches SyntaxError, IndentationError, and everything else the - ast module rejects — matching py_compile's scope but with no - subprocess overhead and no dependency on a ``python`` in PATH. - """ - import ast as _ast - try: - _ast.parse(content) - return True, "" - except SyntaxError as e: - loc = f" (line {e.lineno}, column {e.offset})" if e.lineno else "" - return False, f"{type(e).__name__}: {e.msg}{loc}" - except Exception as e: # noqa: BLE001 - return False, f"{type(e).__name__}: {e}" - # In-process linters by file extension. Preferred over shell linters when # present — no subprocess overhead, microseconds per call. Each callable @@ -820,110 +276,18 @@ LINTERS_INPROC = { '.toml': _lint_toml_inproc, } -# Subset of LINTERS_INPROC that the pre-write fail-closed gate in -# ``write_file`` (see below) refuses on, rather than merely reporting. -# Deliberately excludes ``.py``: unlike JSON/YAML/TOML (atomic structured -# data blobs where "doesn't parse" always means "corrupt"), ``.py`` is -# used throughout this codebase's own test fixtures as a generic -# stand-in extension for arbitrary non-Python text content (e.g. -# ``tests/tools/test_file_operations.py``'s -# ``TestPatchReplacePostWriteVerification`` writes "hello world" / -# "hi world" through a ``*.py`` path purely to exercise write-mechanics, -# not Python validity). Hard-refusing on invalid Python would treat that -# established, exercised pattern as an error and break it. Python source -# keeps the existing (unchanged) post-write lint-delta *report* — still -# visible to the caller, just not a write-blocking refusal. -_FAIL_CLOSED_INPROC_EXTS = frozenset({'.json', '.yaml', '.yml', '.toml'}) # Max limits for read operations MAX_LINES = 2000 MAX_LINE_LENGTH = 2000 MAX_FILE_SIZE = 50 * 1024 # 50KB -DEFAULT_READ_OFFSET = 1 -DEFAULT_READ_LIMIT = 2000 -DEFAULT_SEARCH_OFFSET = 0 -DEFAULT_SEARCH_LIMIT = 50 # Echoed by the size probe when the path exists but is not a regular file. # `wc -c` prints only digits, so this can never collide with a real size. NOT_REGULAR_SENTINEL = "__hermes_not_regular__" -def _coerce_int(value: Any, default: int) -> int: - """Best-effort integer coercion for tool pagination inputs.""" - try: - return int(value) - except (TypeError, ValueError): - return default - - -def normalize_read_pagination(offset: Any = DEFAULT_READ_OFFSET, - limit: Any = DEFAULT_READ_LIMIT) -> tuple[int, int]: - """Return safe read_file pagination bounds. - - Tool schemas declare minimum/maximum values, but not every caller or - provider enforces schemas before dispatch. Clamp here so invalid values - cannot leak into sed ranges like ``0,-1p``. - - The upper bound on ``limit`` comes from ``tool_output.max_lines`` in - config.yaml (defaults to the module-level ``MAX_LINES`` constant). - """ - from tools.tool_output_limits import get_max_lines - max_lines = get_max_lines() - normalized_offset = max(1, _coerce_int(offset, DEFAULT_READ_OFFSET)) - normalized_limit = _coerce_int(limit, DEFAULT_READ_LIMIT) - normalized_limit = max(1, min(normalized_limit, max_lines)) - return normalized_offset, normalized_limit - - -def normalize_search_pagination(offset: Any = DEFAULT_SEARCH_OFFSET, - limit: Any = DEFAULT_SEARCH_LIMIT) -> tuple[int, int]: - """Return safe search pagination bounds for shell head/tail pipelines.""" - normalized_offset = max(0, _coerce_int(offset, DEFAULT_SEARCH_OFFSET)) - normalized_limit = max(1, _coerce_int(limit, DEFAULT_SEARCH_LIMIT)) - return normalized_offset, normalized_limit - - -_REGEX_NEWLINE_ESCAPE_RE = re.compile(r"(? bool: - """Return True when a content-search regex tries to match a newline. - - ``search_files`` runs rg/grep in line-oriented mode, not rg - ``-U``/``--multiline`` mode, so newline regexes cannot match across - lines. Detect both a literal newline already decoded into the tool - argument and a regex ``\n`` escape (odd number of backslashes before - ``n``). Even backslashes, e.g. ``\\n``, mean a literal backslash+n - search and should not warn. - """ - return "\n" in pattern or bool(_REGEX_NEWLINE_ESCAPE_RE.search(pattern)) - - -def _is_line_oriented_newline_error(error: Optional[str]) -> bool: - """Return True for rg's hard error when multiline mode is required.""" - if not error: - return False - return "literal \"\\n\" is not allowed" in error and "--multiline" in error - - -def _maybe_warn_line_oriented_newline_pattern(result: SearchResult, pattern: str) -> SearchResult: - """Attach a newline-regex warning only when search found no usable results.""" - if result.total_count != 0 or not _pattern_has_regex_newline(pattern): - return result - if result.error and not _is_line_oriented_newline_error(result.error): - return result - result.error = None - result.warning = ( - "0 results found. Note: search_files content search is line-oriented " - "and does not run ripgrep with -U/--multiline, so `\\n` in the regex " - "does not match line breaks. Use context=N to inspect neighboring " - "lines, or escape as `\\\\n` when searching for a literal backslash+n." - ) - return result - - -class ShellFileOperations(FileOperations): +class ShellFileOperations(LintMixin, SearchMixin, FileOperations): """ File operations implemented via shell commands. @@ -2410,398 +1774,8 @@ class ShellFileOperations(FileOperations): result = apply_v4a_operations(operations, self) return result - def _check_lint(self, path: str, content: Optional[str] = None) -> LintResult: - """ - Run syntax check on a file after editing. - Prefers the in-process linter for structured formats (JSON, YAML, - TOML) when possible — those parse via the Python stdlib in - microseconds and don't require a subprocess. Falls back to the - shell linter table for compiled/type-checked languages - (py_compile, node --check, tsc, go vet, rustfmt). - Args: - path: File path (used to select the linter + for shell invocation). - content: Optional file content. If provided AND an in-process - linter matches the extension, we lint the content - directly without re-reading the file from disk. Ignored - for shell linters. - - Returns: - LintResult with status and any errors. - """ - ext = os.path.splitext(path)[1].lower() - - # Prefer in-process linter when available. - inproc = LINTERS_INPROC.get(ext) - if inproc is not None: - # Need content — either passed in or read from disk. - if content is None: - read_cmd = f"cat {self._escape_shell_arg(path)} 2>/dev/null" - read_result = self._exec(read_cmd) - if read_result.exit_code != 0: - return LintResult(skipped=True, message=f"Failed to read {path} for lint") - content = read_result.stdout - ok, err = inproc(content) - if err == "__SKIP__": - return LintResult(skipped=True, message=f"No linter available for {ext} (missing dependency)") - return LintResult(success=ok, output="" if ok else err) - - # Fall back to shell linter. - if ext not in LINTERS: - return LintResult(skipped=True, message=f"No linter for {ext} files") - - # A per-file `tsc --noEmit ` cannot read the project's - # tsconfig.json, so for any .ts that belongs to a TS project it floods - # phantom errors — unresolved path aliases (`@/…` → TS2307) and ambient - # globals (`Window.hermesDesktop` → TS2339) that are defined by the - # config it never loads. The delta filter then reports the misleading - # "pre-existing lint errors … the file is still broken", which carries - # no signal and wastes the caller's turns. When an ancestor - # tsconfig.json exists, skip the shell tsc entirely; real diagnostics - # come from the LSP tier (below) or an explicit `tsc -p tsconfig.json` - # the caller runs deliberately. (.tsx already returns above via the - # `ext not in LINTERS` branch.) - if ext == '.ts' and self._has_ancestor_tsconfig(path): - return LintResult( - skipped=True, - message=( - "Project tsconfig.json detected — per-file tsc skipped " - "(single-file tsc can't resolve project aliases/globals; " - "use the LSP tier or `tsc -p tsconfig.json` for real " - "diagnostics)." - ), - ) - - # If a real LSP server is active and claims this file, skip the - # shell linter for extensions whose per-file shell invocation is - # structurally weaker / floods phantom errors. See - # ``_SHELL_LINTER_LSP_REDUNDANT`` above for the rationale per ext. - # The LSP tier runs separately via ``_maybe_lsp_diagnostics`` and - # carries the real diagnostics in ``lsp_diagnostics`` on the - # WriteResult / PatchResult. - if ext in _SHELL_LINTER_LSP_REDUNDANT and self._lsp_will_handle(path): - return LintResult( - skipped=True, - message=f"LSP server handles {ext} — shell linter skipped", - ) - - linter_cmd = LINTERS[ext] - # Extract the base command (first word) - base_cmd = linter_cmd.split()[0] - - if not self._has_command(base_cmd): - return LintResult(skipped=True, message=f"{base_cmd} not available") - - # Run linter. Linters (python, node, tsc, go, rustfmt) are native - # Windows binaries on Windows hosts: they need the C:/... path form. - # The MSYS /c/... form makes node resolve the file as C:\c\Users\... - # (double-prefixed) — every .js write then reports a phantom ENOENT - # lint failure. Native form works for MSYS builds too. - cmd = linter_cmd.replace("{file}", self._escape_native_tool_arg(path)) - result = self._exec(cmd, timeout=30) - - if result.exit_code != 0 and _looks_like_linter_unusable(base_cmd, result.stdout): - # The linter command exists on PATH but couldn't actually run - # (e.g. ``npx tsc`` when tsc isn't in node_modules; ``rustfmt - # --check`` without a Cargo project). This is a tooling gap, - # not a real lint failure — surface it as ``skipped`` so the - # write doesn't get flagged AND so the LSP tier still runs. - from tools.ansi_strip import strip_ansi - cleaned = strip_ansi(result.stdout).strip() - # Collapse to a single line — the npx banner is multi-line ASCII. - first_line = next( - (ln.strip() for ln in cleaned.splitlines() if ln.strip()), - cleaned[:120], - ) - return LintResult( - skipped=True, - message=f"{base_cmd} not usable: {first_line[:200]}", - ) - - return LintResult( - success=result.exit_code == 0, - output=result.stdout.strip() if result.stdout.strip() else "" - ) - - def _check_lint_delta(self, path: str, pre_content: Optional[str], - post_content: Optional[str] = None) -> LintResult: - """ - Run post-write syntax lint with pre-write baseline comparison. - - Two-tier strategy: - - 1. **Syntax check** (in-process or shell-based, microseconds). - Catches the bug class that motivated this layer: corrupt - writes, mashed quotes, truncated output. Hot path. - - 2. **Delta refinement against pre-write content** when the - syntax tier reports errors. Filter out errors that already - existed pre-edit so the agent isn't distracted by inherited - state. - - Semantic diagnostics from the LSP layer are fetched separately - via :meth:`_maybe_lsp_diagnostics` and surfaced in the - ``lsp_diagnostics`` field on :class:`WriteResult` / - :class:`PatchResult`. Keeping the two channels separate lets - the agent (and any downstream parsers) read syntax errors and - semantic errors as independent signals. - - Args: - path: File path (for linter selection). - pre_content: File content BEFORE the write. Pass None for new - files or when the pre-state isn't available — the - delta refinement is skipped and all post errors - are returned. - post_content: File content AFTER the write. Optional; if None, - the shell linter reads from disk (same as - _check_lint). - - Returns: - LintResult. ``output`` contains either the full post-lint - errors (no pre-state) or just the new-error lines (delta - refinement applied). - """ - post = self._check_lint(path, content=post_content) - - # Hot path: clean post-write syntactically. - if post.success or post.skipped: - return post - - # Post-write has syntax errors. If we have pre-content, run the - # delta refinement to filter out pre-existing errors. - if pre_content is None: - return post - - pre = self._check_lint(path, content=pre_content) - if pre.success or pre.skipped or not pre.output: - # Pre-write was clean (or we couldn't lint it) — post errors - # are all new. Return the full post output. - return post - - # Both pre- and post-write had errors. Compute the set-difference - # on non-empty stripped lines. Caveat: single-error parsers - # (ast.parse, json.loads) stop at the first error and don't report - # later ones — if the pre-existing error blocks parsing before - # reaching the edit region, we can't prove the edit is clean. So - # if every post error also appeared pre-edit, we report the file - # as still broken but annotate that this edit introduced nothing - # new on top — the agent knows it's inherited state, not fresh - # damage, without silently dropping the error. - pre_lines = {ln.strip() for ln in pre.output.splitlines() if ln.strip()} - post_lines = [ln for ln in post.output.splitlines() if ln.strip() and ln.strip() not in pre_lines] - - if not post_lines: - # Every error in post was also in pre — this edit didn't make - # anything obviously worse, but the file remains broken and - # the agent should know. - return LintResult( - success=False, - output=post.output, - message="Pre-existing lint errors — this edit didn't introduce new ones but the file is still broken.", - ) - - return LintResult( - success=False, - output=( - "New lint errors introduced by this edit " - "(pre-existing errors filtered out):\n" + "\n".join(post_lines) - ) - ) - - def _lsp_local_only(self) -> bool: - """Return True iff this FileOperations is wired to a local backend. - - LSP servers run on the host process — they need access to the - files they're linting. Remote/sandboxed backends (Docker, - Modal, SSH, Daytona) keep files inside the sandbox where the - host-side LSP server can't reach them, so we skip the LSP - path for those entirely. - """ - env = getattr(self, "env", None) - if env is None: - # Defensive: some tests construct ShellFileOperations via - # ``__new__`` without going through ``__init__``, so - # ``self.env`` may be missing. No env = no LSP path. - return False - try: - from tools.environments.local import LocalEnvironment - except Exception: # noqa: BLE001 - return False - return isinstance(env, LocalEnvironment) - - def _lsp_handles_extension(self, ext: str) -> bool: - """Return True iff some registered LSP server claims this extension. - - Used to decide whether to capture pre-write content for the - line-shift map. Capturing is cheap (one ``cat`` on the host) - but pointless if no LSP would ever look at the file. - - Safe to call on remote backends — the registry is purely - in-process metadata; we still gate the actual LSP path on - :meth:`_lsp_local_only`. - """ - if not ext: - return False - try: - from agent.lsp.servers import SERVERS - except Exception: # noqa: BLE001 - return False - ext_lower = ext.lower() - for srv in SERVERS: - if ext_lower in srv.extensions: - return True - return False - - def _has_ancestor_tsconfig(self, path: str) -> bool: - """True iff a tsconfig.json exists in *path*'s directory or any ancestor. - - A single-file ``tsc`` invocation can't read that config, so its - diagnostics for such a file are pure noise (unresolved aliases / - ambient globals). Used by :meth:`_check_lint` to skip the per-file - shell tsc for project TypeScript files. - - Best-effort and local-host only: a host-side ``os.path`` walk. On a - remote/sandboxed backend the project tree isn't on this host, so the - walk returns False and the shell linter runs exactly as before — never - suppress lint based on a probe that couldn't answer. - """ - if not self._lsp_local_only(): - return False - try: - d = os.path.dirname(os.path.abspath(path)) - while True: - if os.path.isfile(os.path.join(d, "tsconfig.json")): - return True - parent = os.path.dirname(d) - if parent == d: - return False - d = parent - except Exception: # noqa: BLE001 - return False - - def _lsp_will_handle(self, path: str) -> bool: - """Return True iff the LSP service is active AND will lint this file. - - Stronger than :meth:`_lsp_handles_extension` — that one only checks - the static server registry. This one additionally requires the - LSP service to be configured/enabled and the file to pass - :meth:`agent.lsp.manager.LSPService.enabled_for` (which gates on - workspace detection, disabled-server set, and the broken-pair - short-circuit). - - Used by :meth:`_check_lint` to decide whether to skip the per-file - shell linter for extensions in ``_SHELL_LINTER_LSP_REDUNDANT``. - - Best-effort: any failure path returns False so the shell linter - runs as before — never suppress lint based on an LSP probe that - couldn't actually answer the question. - """ - if not self._lsp_local_only(): - return False - try: - from agent.lsp import get_service - except Exception: # noqa: BLE001 - return False - try: - svc = get_service() - except Exception: # noqa: BLE001 - return False - if svc is None: - return False - try: - return bool(svc.enabled_for(path)) - except Exception: # noqa: BLE001 - return False - - def _snapshot_lsp_baseline(self, path: str) -> None: - """Capture pre-edit LSP diagnostics so the post-write delta is correct. - - Best-effort. Silent on every failure path — LSP is an - enrichment layer and must never break a write. - - Skipped entirely on non-local backends (Docker, Modal, SSH, - etc.) — the server can't see files inside the sandbox. - """ - if not self._lsp_local_only(): - return - try: - from agent.lsp import get_service - svc = get_service() - except Exception: # noqa: BLE001 - return - if svc is None: - return - try: - svc.snapshot_baseline(path) - except Exception: # noqa: BLE001 - pass - - def _maybe_lsp_diagnostics( - self, - path: str, - *, - pre_content: Optional[str] = None, - post_content: Optional[str] = None, - ) -> str: - """Best-effort LSP semantic diagnostics for ``path``. - - Returns a formatted ```` block, or empty string - when LSP is unavailable / disabled / produced no errors. - - When both ``pre_content`` and ``post_content`` are provided, - a line-shift map is built and passed to the LSPService so - baseline diagnostics are remapped into post-edit coordinates - before the set-difference. Without this, edits that delete - or insert lines surface every pre-existing diagnostic below - the edit point as "introduced by this edit". - - Wraps everything in a try/except so a misbehaving LSP server - can't break a write. This intentionally swallows all errors - — the calling tier already returned a clean syntax result, so - ``""`` here just means "no extra info to add". - - Skipped entirely on non-local backends (Docker, Modal, SSH, - etc.) — same reasoning as ``_snapshot_lsp_baseline``. - """ - if not self._lsp_local_only(): - return "" - try: - from agent.lsp import get_service - except Exception: # noqa: BLE001 - return "" - try: - svc = get_service() - except Exception: # noqa: BLE001 - return "" - if svc is None or not svc.enabled_for(path): - return "" - - # Build a line-shift map when we have both pre and post — it - # remaps baseline diagnostics into post-edit coordinates so - # the strict (range-aware) delta key matches correctly. - line_shift = None - if pre_content is not None and post_content is not None and pre_content != post_content: - try: - from agent.lsp.range_shift import build_line_shift - line_shift = build_line_shift(pre_content, post_content) - except Exception: # noqa: BLE001 - line_shift = None - - try: - diagnostics = svc.get_diagnostics_sync(path, delta=True, line_shift=line_shift) - except Exception: # noqa: BLE001 - return "" - if not diagnostics: - return "" - try: - from agent.lsp.reporter import report_for_file, truncate - block = report_for_file(path, diagnostics) - if not block: - return "" - return truncate("LSP diagnostics introduced by this edit:\n" + block) - except Exception: # noqa: BLE001 - return "" # ========================================================================= # SEARCH Implementation @@ -2832,8 +1806,7 @@ class ShellFileOperations(FileOperations): path = self._expand_path(path) # Validate that the path exists before searching - check = self._exec(f"test -e {self._escape_shell_arg(path)} && echo exists || echo not_found") - if "not_found" in check.stdout: + if "not_found" in self._path_exists_probe(path): # Multi-path recovery: models frequently pass several paths in # one string ("dir1 dir2 dir3" or comma-separated). Instead of # failing the whole call, split, search every path that exists, @@ -2843,41 +1816,10 @@ class ShellFileOperations(FileOperations): ) if multi is not None: return multi - # Try to suggest nearby paths - parent = os.path.dirname(path) or "." - basename_query = os.path.basename(path) - hint_parts = [f"Path not found: {path}"] - # Check if parent directory exists and list similar entries - parent_check = self._exec( - f"test -d {self._escape_shell_arg(parent)} && echo yes || echo no" - ) - if "yes" in parent_check.stdout and basename_query: - ls_result = self._exec( - f"ls -1 {self._escape_shell_arg(parent)} 2>/dev/null | head -20" - ) - if ls_result.exit_code == 0 and ls_result.stdout.strip(): - lower_q = basename_query.lower() - candidates = [] - for entry in ls_result.stdout.strip().split('\n'): - if not entry: - continue - le = entry.lower() - if lower_q in le or le in lower_q or le.startswith(lower_q[:3]): - candidates.append(os.path.join(parent, entry)) - if candidates: - hint_parts.append( - "Similar paths: " + ", ".join(candidates[:5]) - ) - return SearchResult( - error=". ".join(hint_parts), - total_count=0 - ) - - if target == "files": - result = self._search_files(pattern, path, limit, offset) - else: - result = self._search_content(pattern, path, file_glob, limit, offset, - output_mode, context) + return self._path_not_found_result(path) + + result = self._dispatch_search(pattern, path, target, file_glob, limit, offset, + output_mode, context) exclusions = self._macos_search_exclusions(path) if exclusions and not result.error: @@ -2889,690 +1831,11 @@ class ShellFileOperations(FileOperations): ) return result - def _macos_search_exclusions(self, path: str) -> List[str]: - """Protected descendants to prune for this search root, if any. - - Gated on ``env.is_local``: ``sys.platform``/``Path.home()`` describe - the CONTROLLER, but search commands execute on ``self.env``'s host — a - macOS controller driving a Linux container/SSH backend must not prune - the remote's (unprotected) Downloads, and TCC doesn't exist there - anyway. A Linux controller driving a macOS SSH host keeps today's - behavior (no pruning); detecting the remote OS is out of scope here. - Environments without the flag (test fakes, plugins) default to local - semantics — pruning is a warning-carrying skip, never data loss. - """ - env = getattr(self, "env", None) - if env is not None and getattr(env, "is_local", True) is False: - return [] - cwd = getattr(self.env, "cwd", None) or self.cwd - return _macos_protected_search_exclusions( - path, cwd=cwd, home=_HOME, platform=sys.platform - ) - def _try_multi_path_search(self, pattern: str, path: str, target: str, - file_glob: Optional[str], limit: int, offset: int, - output_mode: str, context: int) -> Optional[SearchResult]: - """Recover a not-found ``path`` that is really several paths in one string. - Production trajectories show models passing "dir1 dir2 dir3" (or - comma-separated lists) as ``path``. Split on whitespace/commas; when - at least one candidate exists and at least two candidates were given, - search every existing path, merge results, and note skipped parts. - Returns None when this doesn't look like a multi-path string. - """ - parts = [p for chunk in path.split(",") for p in chunk.split() if p.strip()] - if len(parts) < 2: - return None - existing, missing = [], [] - for p in parts: - expanded = self._expand_path(p) - chk = self._exec( - f"test -e {self._escape_shell_arg(expanded)} && echo exists || echo not_found" - ) - (existing if "exists" in chk.stdout else missing).append(expanded) - if not existing: - return None - merged = SearchResult() - for p in existing: - if target == "files": - sub = self._search_files(pattern, p, limit, offset) - else: - sub = self._search_content(pattern, p, file_glob, limit, offset, - output_mode, context) - if sub.error: - continue - merged.matches.extend(sub.matches) - merged.files.extend(sub.files) - merged.counts.update(sub.counts) - merged.total_count += sub.total_count - merged.truncated = merged.truncated or sub.truncated - # Respect the caller's limit across the merged set. - merged.matches = merged.matches[:limit] - merged.files = merged.files[:limit] - note = f"path contained {len(parts)} entries; searched {len(existing)} that exist" - if missing: - note += "; skipped missing: " + ", ".join(missing[:3]) - if len(missing) > 3: - note += f" (+{len(missing) - 3} more)" - merged.warning = note - return merged - - def _zero_match_probe(self, pattern: str, path: str, - file_glob: Optional[str]) -> Optional[str]: - """Return a hint for a 0-match content search, or None. - - 13.9% of production content searches return zero matches and give - the model nothing to steer by. Run ONE cheap case-insensitive count - probe; if it hits, say so. If the pattern contains regex - metacharacters, also probe it as a fixed string. Bounded: two rg - invocations max, count-only output. - """ - if not self._has_command('rg'): - return None - - def _tally(stdout: str): - """Parse ``path:count`` lines from rg --count-matches.""" - total = 0 - per_file = [] - for line in (stdout or "").strip().splitlines(): - p, _sep, n = line.rpartition(":") - if n.isdigit(): - total += int(n) - per_file.append(p) - return total, per_file - - def _paths_note(per_file, cap: int = 5) -> str: - shown = ", ".join(per_file[:cap]) - extra = len(per_file) - cap - return shown + (f" (+{extra} more)" if extra > 0 else "") - - glob_expr = f" --glob {self._escape_shell_arg(file_glob)}" if file_glob else "" - probe = self._exec( - f"rg -i --count-matches{glob_expr} " - f"{self._escape_shell_arg(pattern)} {self._escape_native_tool_arg(path)} " - f"2>/dev/null | head -50", - timeout=30, - ) - ci_total, ci_paths = _tally(probe.stdout) - if ci_total > 0: - return ( - f"0 exact matches, but {ci_total} case-insensitive match(es) " - f"in {len(ci_paths)} file(s): {_paths_note(ci_paths)} — " - "the pattern's casing may be wrong." - ) - # Hidden/ignored probe: rg skips dotdirs and .gitignore'd files by - # default. When the pattern exists only there, say so instead of - # returning a bare zero (bench case: match in .hidden/ silently - # missing from results). - hidden = self._exec( - f"rg --hidden --no-ignore --count-matches{glob_expr} " - f"{self._escape_shell_arg(pattern)} {self._escape_native_tool_arg(path)} " - f"2>/dev/null | head -50", - timeout=30, - ) - h_total, h_paths = _tally(hidden.stdout) - if h_total > 0: - return ( - f"0 matches in visible files, but {h_total} match(es) in " - f"{len(h_paths)} hidden or gitignored file(s): " - f"{_paths_note(h_paths)} — these are excluded by default." - ) - if re.search(r"[.\[\](){}?*+^$\\|]", pattern): - fixed = self._exec( - f"rg -F --count-matches{glob_expr} " - f"{self._escape_shell_arg(pattern)} {self._escape_native_tool_arg(path)} " - f"2>/dev/null | head -50", - timeout=30, - ) - f_total, f_paths = _tally(fixed.stdout) - if f_total > 0: - return ( - f"0 regex matches, but {f_total} literal match(es) in " - f"{len(f_paths)} file(s): {_paths_note(f_paths)} — the " - "pattern contains regex metacharacters that likely need " - "escaping (or pass a simpler substring)." - ) - return None - - def _search_files(self, pattern: str, path: str, limit: int, offset: int) -> SearchResult: - """Search for files by name pattern (glob-like).""" - # Auto-prepend **/ for recursive search if not already present - if not pattern.startswith('**/') and '/' not in pattern: - search_pattern = pattern - else: - search_pattern = pattern.split('/')[-1] - - search_root = Path(path) - has_hidden_path_ancestor = any( - part not in {".", ".."} and part.startswith(".") - for part in search_root.parts - ) - - # Prefer ripgrep: respects .gitignore, excludes hidden dirs by - # default, and has parallel directory traversal (~200x faster than - # find on wide trees). Mirrors _search_content which already uses rg. - if self._has_command('rg'): - return self._search_files_rg(search_pattern, path, limit, offset) - - # Fallback: find (slower, no .gitignore awareness) - if not self._has_command('find'): - return SearchResult( - error="File search requires 'rg' (ripgrep) or 'find'. " - "Install ripgrep for best results: " - "https://github.com/BurntSushi/ripgrep#installation" - ) - - # Exclude hidden directories (matching ripgrep's default behavior). - hidden_exclude = "-not -path '*/.*'" if not has_hidden_path_ancestor else "" - hidden_filter_expr = f" {hidden_exclude}" if hidden_exclude else "" - - # Use shell pagination for standard roots. For hidden roots, gather full - # output so we can re-apply hidden-descendant filtering while allowing - # explicit hidden-root searches. - pagination_expr = "" - if not has_hidden_path_ancestor: - pagination_expr = f" | tail -n +{offset + 1} | head -n {limit}" - - # Prune protected directories before traversal so macOS never receives - # an access attempt (filtering matched paths after descent is too late). - protected_paths = [ - os.path.normpath(os.path.join(path, item)) - for item in self._macos_search_exclusions(path) - ] - prune_expr = "" - if protected_paths: - prune_terms = " -o ".join( - f"-path {self._escape_shell_arg(item)}" for item in protected_paths - ) - prune_expr = f" \\( {prune_terms} \\) -prune -o" - - cmd = f"find {self._escape_shell_arg(path)}{prune_expr}{hidden_filter_expr} -type f -name {self._escape_shell_arg(search_pattern)} " \ - f"-printf '%T@ %p\\n' 2>/dev/null | sort -rn{pagination_expr}" - - result = self._exec(cmd, timeout=60) - stdout, limit_reason = _search_stdout_and_limit(result) - - if not stdout.strip() and not limit_reason: - # Try without -printf (BSD find compatibility -- macOS) - cmd_simple = f"find {self._escape_shell_arg(path)}{prune_expr}{hidden_filter_expr} -type f -name {self._escape_shell_arg(search_pattern)} " \ - f"2>/dev/null | sort -rn{pagination_expr}" - result = self._exec(cmd_simple, timeout=60) - stdout, limit_reason = _search_stdout_and_limit(result) - - files = [] - for line in stdout.strip().split('\n'): - if not line: - continue - parts = line.split(' ', 1) - if len(parts) == 2 and parts[0].replace('.', '').isdigit(): - files.append(parts[1]) - else: - files.append(line) - - # For explicit hidden roots, find's path-based filtering excludes every - # file under the hidden path. Apply descendant filtering after command - # execution so only the explicit root ancestry is bypassed. - if has_hidden_path_ancestor: - normalized_root = search_root.resolve() - filtered_files = [] - for file_path in files: - try: - rel_parts = Path(file_path).resolve().relative_to(normalized_root).parts - except ValueError: - rel_parts = Path(file_path).parts - if any(part not in {".", ".."} and part.startswith(".") for part in rel_parts): - continue - filtered_files.append(file_path) - files = filtered_files[offset:offset + limit] - # pagination for standard roots is already applied in shell - - return SearchResult( - files=files, - total_count=len(files), - truncated=bool(limit_reason), - limit_reason=limit_reason, - ) - - def _search_files_rg(self, pattern: str, path: str, limit: int, offset: int) -> SearchResult: - """Search for files by name using ripgrep's --files mode. - - rg --files respects .gitignore and excludes hidden directories by - default, and uses parallel directory traversal for ~200x speedup - over find on wide trees. Results are sorted by modification time - (most recently edited first) when rg >= 13.0 supports --sortr. - """ - # rg --files -g uses glob patterns; wrap bare names so they match - # at any depth (equivalent to find -name). - if '/' not in pattern and not pattern.startswith('*'): - glob_pattern = f"*{pattern}" - else: - glob_pattern = pattern - - fetch_limit = limit + offset - exclusion_globs = " ".join( - f"--glob {self._escape_shell_arg(f'!{item}/**')}" - for item in self._macos_search_exclusions(path) - ) - exclusion_args = f" {exclusion_globs}" if exclusion_globs else "" - # Try mtime-sorted first (rg 13+); fall back to unsorted if not supported. - cmd_sorted = ( - f"rg --files --sortr=modified -g {self._escape_shell_arg(glob_pattern)}" - f"{exclusion_args} " - f"{self._escape_native_tool_arg(path)} 2>/dev/null " - f"| head -n {fetch_limit}" - ) - result = self._exec(cmd_sorted, timeout=60) - stdout, limit_reason = _search_stdout_and_limit(result) - all_files = [f for f in stdout.strip().split('\n') if f] - - if not all_files and not limit_reason: - # --sortr may have failed on older rg; retry without it. - cmd_plain = ( - f"rg --files -g {self._escape_shell_arg(glob_pattern)}" - f"{exclusion_args} " - f"{self._escape_native_tool_arg(path)} 2>/dev/null " - f"| head -n {fetch_limit}" - ) - result = self._exec(cmd_plain, timeout=60) - stdout, limit_reason = _search_stdout_and_limit(result) - all_files = [f for f in stdout.strip().split('\n') if f] - - page = all_files[offset:offset + limit] - - return SearchResult( - files=page, - total_count=len(all_files), - truncated=len(all_files) >= fetch_limit or bool(limit_reason), - limit_reason=limit_reason, - ) - def _search_content(self, pattern: str, path: str, file_glob: Optional[str], - limit: int, offset: int, output_mode: str, context: int) -> SearchResult: - """Search for content inside files (grep-like).""" - # Try ripgrep first (fast), fallback to grep (slower but works) - used_rg = False - if self._has_command('rg'): - used_rg = True - result = self._search_with_rg(pattern, path, file_glob, limit, offset, - output_mode, context) - elif self._has_command('grep'): - result = self._search_with_grep(pattern, path, file_glob, limit, offset, - output_mode, context) - else: - # Neither rg nor grep available (Windows without Git Bash, etc.) - return SearchResult( - error="Content search requires ripgrep (rg) or grep. " - "Install ripgrep: https://github.com/BurntSushi/ripgrep#installation" - ) - - # Zero-match steering: a 0-match result with no guidance is a dead - # turn. Probe cheaply for near-misses (wrong casing, hidden-only - # matches, unescaped regex metacharacters) and attach the finding - # as a warning. Runs for BOTH engines. - if (not result.error and result.total_count == 0 - and not result.matches and not result.files and not result.counts): - try: - hint = self._zero_match_probe(pattern, path, file_glob) - except Exception: - hint = None - if hint: - result.warning = hint if not result.warning else f"{result.warning} {hint}" - - # rg auto-enables --multiline for \n patterns, so the line-oriented - # explanation only applies to the grep fallback engine. - if used_rg: - return result - return _maybe_warn_line_oriented_newline_pattern(result, pattern) - def _search_with_rg(self, pattern: str, path: str, file_glob: Optional[str], - limit: int, offset: int, output_mode: str, context: int) -> SearchResult: - """Search using ripgrep.""" - cmd_parts = ["rg", "--line-number", "--no-heading", "--with-filename"] - - # Auto-multiline: a regex `\n` (or a literal newline in the pattern) - # cannot match in rg's default line-oriented mode — it used to hard - # error ("the literal \"\\n\" is not allowed") and burn a turn. When - # the pattern clearly wants to cross lines, enable -U/--multiline - # up front and note it in the result. - multiline = _pattern_has_regex_newline(pattern) - if multiline: - cmd_parts.append("--multiline") - - # Add context if requested - if context > 0: - cmd_parts.extend(["-C", str(context)]) - - # Exclude macOS TCC-protected descendants during broad searches. - for item in self._macos_search_exclusions(path): - cmd_parts.extend(["--glob", self._escape_shell_arg(f"!{item}/**")]) - - # Add file glob filter (must be quoted to prevent shell expansion) - if file_glob: - cmd_parts.extend(["--glob", self._escape_shell_arg(file_glob)]) - - # Output mode handling - if output_mode == "files_only": - cmd_parts.append("-l") # Files only - elif output_mode == "count": - cmd_parts.append("-c") # Count per file - - # Add pattern and path - cmd_parts.append(self._escape_shell_arg(pattern)) - # rg is a native Windows binary when installed via winget/cargo/choco: - # it needs the C:/... path form, not the MSYS /c/... form (which - # nothing converts back — Hermes sets MSYS_NO_PATHCONV for its bash). - cmd_parts.append(self._escape_native_tool_arg(path)) - - # Fetch extra rows so we can report the true total before slicing. - # For context mode, rg emits separator lines ("--") between groups, - # so we grab generously and filter in Python. - fetch_limit = limit + offset + 200 if context > 0 else limit + offset - cmd_parts.extend(["|", "head", "-n", str(fetch_limit)]) - - # `set -o pipefail` so rg's exit status propagates through `| head`. - # Without it the pipeline reports head's status (0), masking rg's - # error code (2) and making the guard below unreachable. rg handles a - # truncating head cleanly (exit 0 on SIGPIPE), so pipefail does not - # introduce false errors on a successful-but-truncated search. - cmd = "set -o pipefail; " + " ".join(cmd_parts) - result = self._exec(cmd, timeout=60) - stdout, limit_reason = _search_stdout_and_limit(result) - - # _exec merges stderr into stdout (stderr=subprocess.STDOUT), so rg's - # diagnostic lines ("rg: : ", "rg: regex parse error:") - # are interleaved with match output. Split them out: diagnostics must - # not be parsed as matches, and on a hard error they ARE the message. - diagnostics, payload = _split_tool_diagnostics(stdout) - - # rg exit codes: 0=matches found, 1=no matches, 2=error. rg returns 2 - # even on partial errors (e.g. one unreadable file in a tree that - # otherwise matched), so only surface an error when exit==2 AND no - # usable match payload remains. Otherwise we keep the real matches. - if result.exit_code == 2 and not payload.strip(): - error_msg = diagnostics.strip() or result.stdout.strip() or "Search error" - return SearchResult(error=f"Search failed: {error_msg}", total_count=0) - - # Parse the diagnostic-free payload so error text never becomes a match. - stdout = payload - _ml_note = ( - "Pattern contains \\n — multiline mode (-U) was enabled automatically " - "so the regex can match across line boundaries." - ) if multiline else None - # Parse results based on output mode - if output_mode == "files_only": - all_files = [f for f in stdout.strip().split('\n') if f] - total = len(all_files) - page = all_files[offset:offset + limit] - return SearchResult( - files=page, - total_count=total, - truncated=bool(limit_reason), - limit_reason=limit_reason, - warning=_ml_note, - ) - - elif output_mode == "count": - counts = {} - for line in stdout.strip().split('\n'): - if ':' in line: - parts = line.rsplit(':', 1) - if len(parts) == 2: - try: - counts[parts[0]] = int(parts[1]) - except ValueError: - pass - return SearchResult( - counts=counts, - total_count=sum(counts.values()), - truncated=bool(limit_reason), - limit_reason=limit_reason, - ) - - else: - # Parse content matches and context lines. - # rg match lines: "file:lineno:content" (colon separator) - # rg context lines: "file-lineno-content" (dash separator) - # rg group seps: "--" - # Note: on Windows, paths contain drive letters (e.g. C:\path), - # so naive split(":") breaks. Use regex to handle both platforms. - _match_re = re.compile(r'^([A-Za-z]:)?(.*?):(\d+):(.*)$') - matches = [] - for line in stdout.strip().split('\n'): - if not line or line == "--": - continue - - # Try match line first (colon-separated: file:line:content) - m = _match_re.match(line) - if m: - matches.append(SearchMatch( - path=(m.group(1) or '') + m.group(2), - line_number=int(m.group(3)), - content=m.group(4)[:500] - )) - continue - - # Try context line (dash-separated: file-line-content) - # Only attempt if context was requested to avoid false positives - if context > 0: - parsed = _parse_search_context_line(line) - if parsed: - matches.append(SearchMatch( - path=parsed[0], - line_number=parsed[1], - content=parsed[2][:500] - )) - - total = len(matches) - page = matches[offset:offset + limit] - return SearchResult( - matches=page, - total_count=total, - truncated=total > offset + limit or bool(limit_reason), - limit_reason=limit_reason, - warning=_ml_note, - ) - def _search_with_grep(self, pattern: str, path: str, file_glob: Optional[str], - limit: int, offset: int, output_mode: str, context: int) -> SearchResult: - """Fallback search using grep.""" - cmd_parts = ["grep", "-rnHE"] # -H forces filenames; -E matches rg regex behavior - - # Exclude hidden directories (matching ripgrep's default behavior). - # This prevents searching inside .hub/index-cache/, .git/, etc. - cmd_parts.append("--exclude-dir='.*'") - # Protected-dir pruning CANNOT use --exclude-dir here: grep matches - # exclude-dir globs against BASENAMES anywhere in the tree, so - # --exclude-dir=Downloads would silently skip every nested directory - # named Downloads (a repo's own Downloads/ folder included), not just - # the protected home child. When exclusions apply (darwin broad-home - # search on a local backend), route through find's path-scoped -prune - # instead — same traversal-prevention the find backend uses. - protected_paths = [ - os.path.normpath(os.path.join(path, item)) - for item in self._macos_search_exclusions(path) - ] - if protected_paths: - return self._search_with_grep_pruned( - pattern, path, file_glob, limit, offset, output_mode, context, - protected_paths, - ) - - # Add context if requested - if context > 0: - cmd_parts.extend(["-C", str(context)]) - - # Add file pattern filter (must be quoted to prevent shell expansion) - if file_glob: - cmd_parts.extend(["--include", self._escape_shell_arg(file_glob)]) - - # Output mode handling - if output_mode == "files_only": - cmd_parts.append("-l") - elif output_mode == "count": - cmd_parts.append("-c") - - # Add pattern and path. grep applies --exclude-dir to the command-line - # search root too, so passing the default relative root ``.`` causes - # ``.*`` to exclude the entire search. Anchor relative paths at the - # shell's live cwd; quoting $PWD separately keeps user paths escaped - # while working across local, container, and remote backends. - cmd_parts.append(self._escape_shell_arg(pattern)) - is_absolute = path.startswith(("/", "\\\\")) or bool( - re.match(r"^[A-Za-z]:[\\/]", path) - ) - if is_absolute: - search_root = self._escape_shell_arg(path) - else: - relative_path = path[2:] if path.startswith("./") else path - search_root = '"$PWD"' - if relative_path not in {"", "."}: - search_root += f"/{self._escape_shell_arg(relative_path)}" - cmd_parts.append(search_root) - - # Fetch generously so we can compute total before slicing - fetch_limit = limit + offset + (200 if context > 0 else 0) - cmd_parts.extend(["|", "head", "-n", str(fetch_limit)]) - - # `set -o pipefail` so grep's exit status propagates through `| head` - # (without it the pipeline reports head's 0, masking grep's error 2). - # A truncating head makes grep exit 141 (SIGPIPE) on an otherwise - # successful search; the strict `== 2` guard below ignores that, so - # pipefail does not turn truncated results into false errors. - cmd = "set -o pipefail; " + " ".join(cmd_parts) - result = self._exec(cmd, timeout=60) - return self._parse_grep_search_output(result, output_mode, limit, offset, context) - def _search_with_grep_pruned(self, pattern: str, path: str, file_glob: Optional[str], - limit: int, offset: int, output_mode: str, context: int, - protected_paths: List[str]) -> SearchResult: - """grep fallback with PATH-scoped protected-dir pruning. - - Files are enumerated by ``find`` with the same ``-path ... -prune`` - expression the find backend uses (traversal never enters the protected - dirs, so macOS never sees an access attempt), then handed to grep via - ``-exec {} +``. This exists because grep's own ``--exclude-dir`` - matches basenames anywhere in the tree — it cannot express "only the - home-level Downloads". Hidden directories are pruned to mirror the - plain path's ``--exclude-dir='.*'``. Trade-off: with ``-exec {} +`` - find folds grep's exit code into its own generic non-zero, so a hard - grep error surfaces as an empty result rather than exit 2 — acceptable - for this darwin-local-broad-search-only branch. - """ - grep_parts = ["grep", "-nHE"] - if context > 0: - grep_parts.extend(["-C", str(context)]) - if output_mode == "files_only": - grep_parts.append("-l") - elif output_mode == "count": - grep_parts.append("-c") - grep_parts.append(self._escape_shell_arg(pattern)) - - prune_terms = " -o ".join( - f"-path {self._escape_shell_arg(item)}" for item in protected_paths - ) - find_parts = [ - "find", self._escape_shell_arg(path or "."), - f"\\( {prune_terms} \\) -prune", "-o", - "\\( -type d -name '.*' \\) -prune", "-o", - "-type f", - ] - if file_glob: - find_parts.extend(["-name", self._escape_shell_arg(file_glob)]) - find_parts.extend(["-exec", *grep_parts, "{}", "+"]) - fetch_limit = limit + offset + (200 if context > 0 else 0) - cmd = ( - "set -o pipefail; " + " ".join(find_parts) - + f" 2>/dev/null | head -n {fetch_limit}" - ) - result = self._exec(cmd, timeout=60) - return self._parse_grep_search_output(result, output_mode, limit, offset, context) - - def _parse_grep_search_output(self, result, output_mode: str, limit: int, - offset: int, context: int) -> SearchResult: - """Shared grep output parsing for the plain and pruned variants.""" - stdout, limit_reason = _search_stdout_and_limit(result) - - # _exec merges stderr into stdout, so grep's diagnostic lines - # ("grep: : ") are interleaved with matches. Split them - # out so they're never parsed as matches and so a hard error has a - # clean message. - diagnostics, payload = _split_tool_diagnostics(stdout) - - # grep exit codes: 0=matches found, 1=no matches, 2=error. grep - # returns 2 on partial errors (e.g. an unreadable file) even when - # other files matched, so only surface an error when exit==2 AND no - # usable match payload remains. - if result.exit_code == 2 and not payload.strip(): - error_msg = diagnostics.strip() or result.stdout.strip() or "Search error" - return SearchResult(error=f"Search failed: {error_msg}", total_count=0) - - stdout = payload - if output_mode == "files_only": - all_files = [f for f in stdout.strip().split('\n') if f] - total = len(all_files) - page = all_files[offset:offset + limit] - return SearchResult( - files=page, - total_count=total, - truncated=bool(limit_reason), - limit_reason=limit_reason, - ) - - elif output_mode == "count": - counts = {} - for line in stdout.strip().split('\n'): - if ':' in line: - parts = line.rsplit(':', 1) - if len(parts) == 2: - try: - counts[parts[0]] = int(parts[1]) - except ValueError: - pass - return SearchResult( - counts=counts, - total_count=sum(counts.values()), - truncated=bool(limit_reason), - limit_reason=limit_reason, - ) - - else: - # grep match lines: "file:lineno:content" (colon) - # grep context lines: "file-lineno-content" (dash) - # grep group seps: "--" - # Note: on Windows, paths contain drive letters (e.g. C:\path), - # so naive split(":") breaks. Use regex to handle both platforms. - _match_re = re.compile(r'^([A-Za-z]:)?(.*?):(\d+):(.*)$') - matches = [] - for line in stdout.strip().split('\n'): - if not line or line == "--": - continue - - m = _match_re.match(line) - if m: - matches.append(SearchMatch( - path=(m.group(1) or '') + m.group(2), - line_number=int(m.group(3)), - content=m.group(4)[:500] - )) - continue - - if context > 0: - parsed = _parse_search_context_line(line) - if parsed: - matches.append(SearchMatch( - path=parsed[0], - line_number=parsed[1], - content=parsed[2][:500] - )) - - - total = len(matches) - page = matches[offset:offset + limit] - return SearchResult( - matches=page, - total_count=total, - truncated=total > offset + limit or bool(limit_reason), - limit_reason=limit_reason, - ) diff --git a/tools/file_operations_search.py b/tools/file_operations_search.py index cfbbfb9ef2..b61603f5f8 100644 --- a/tools/file_operations_search.py +++ b/tools/file_operations_search.py @@ -308,7 +308,7 @@ class SearchMixin: existing, missing = [], [] for p in parts: expanded = self._expand_path(p) - (existing if self._path_exists(expanded) else missing).append(expanded) + (existing if "exists" in self._path_exists_probe(expanded) else missing).append(expanded) if not existing: return None diff --git a/tools/lazy_deps.py b/tools/lazy_deps.py index f2ef410e29..cfc8f755cf 100644 --- a/tools/lazy_deps.py +++ b/tools/lazy_deps.py @@ -45,10 +45,10 @@ logger = logging.getLogger(__name__) # Allowlist: "namespace.backend" -> pip specs matching the pyproject extra. # Pins are exact (no ranges, security posture); bump here AND in pyproject. -# Shared patched floors (prior CVEs + GHSA-cq5v-8q36-5273/GHSA-mfx4-hv73-q22v/ -# GHSA-mq44-7p77-q5h7; CVE-2026-48710 BadHost) — keep in sync with pyproject. -_AIOHTTP_PIN = "aiohttp==3.14.3" -_STARLETTE_PIN = "starlette==1.3.1" +# Shared patched floors, spelled out as literals in every feature because +# tests/test_packaging_metadata.py checks them by AST: aiohttp==3.14.3 (prior +# CVEs + GHSA-cq5v-8q36-5273/GHSA-mfx4-hv73-q22v/GHSA-mq44-7p77-q5h7) and +# starlette==1.3.1 (CVE-2026-48710 BadHost) — keep in sync with pyproject. LAZY_DEPS: dict[str, tuple[str, ...]] = { # ─── Inference providers ─────────────────────────────────────────────── @@ -140,19 +140,19 @@ LAZY_DEPS: dict[str, tuple[str, ...]] = { "platform.discord": ( "discord.py[voice]==2.7.1", "brotlicffi==1.2.0.1", - _AIOHTTP_PIN, + "aiohttp==3.14.3", ), "platform.slack": ( "slack-bolt==1.30.0", "slack-sdk==3.43.0", - _AIOHTTP_PIN, + "aiohttp==3.14.3", ), "platform.matrix": ( "mautrix[encryption]==0.21.1", "aiosqlite==0.22.1", "asyncpg==0.31.0", "aiohttp-socks==0.11.0", - _AIOHTTP_PIN, + "aiohttp==3.14.3", ), "platform.dingtalk": ( "dingtalk-stream==0.24.3", @@ -166,7 +166,7 @@ LAZY_DEPS: dict[str, tuple[str, ...]] = { # WeCom callback adapter parses untrusted XML POST bodies -> defusedxml. "platform.wecom_callback": ("defusedxml==0.7.1",), # Teams pulls a heavy tree (msal, dependency-injector); also the `teams` extra. - "platform.teams": ("microsoft-teams-apps==2.0.13.4", _AIOHTTP_PIN), + "platform.teams": ("microsoft-teams-apps==2.0.13.4", "aiohttp==3.14.3"), # ─── Terminal backends ───────────────────────────────────────────────── "terminal.modal": ("modal==1.3.4",), @@ -191,7 +191,7 @@ LAZY_DEPS: dict[str, tuple[str, ...]] = { "tool.dashboard": ( "fastapi==0.133.1", "uvicorn[standard]==0.41.0", - _STARLETTE_PIN, + "starlette==1.3.1", "python-multipart==0.0.32", # FastAPI UploadFile/Form streaming uploads ), # Pillow and firecrawl-anydoc are CORE deps; these entries are the self-heal @@ -204,7 +204,7 @@ LAZY_DEPS: dict[str, tuple[str, ...]] = { "tool.computer_use": ( "mcp==2.0.0", "httpx2==2.7.0", # mcp 2.x HTTP stack — sync with pyproject [computer-use] - _STARLETTE_PIN, + "starlette==1.3.1", ), # huggingface-hub is SHARED with transformers (>=1.5.0,<2 via Hindsight) and # active_features() marks it active on mere presence, so `hermes update`