refactor(tools): compact file_operations tiers — shared cat/head/python-snippet helpers, one search pipeline runner, drop dead re-exports

This commit is contained in:
Teknium
2026-09-02 22:19:23 -07:00
parent 113f04616b
commit e86180b446
5 changed files with 446 additions and 986 deletions

File diff suppressed because it is too large Load Diff

View File

@@ -111,12 +111,9 @@ class SearchResult:
_DENSIFY_MIN_MATCHES: ClassVar[int] = 5
def _densify_matches(self) -> Optional[str]:
"""Render content matches as a lossless, path-grouped text block.
Path printed once, then `` <line>: <content>`` rows. Relies on rg/grep
emitting a file's hits consecutively, so grouping on path change needs
no reordering. Returns None when too few matches to be worth it.
"""
"""Lossless path-grouped text block: path once, then `` <line>: <content>``
rows. Relies on rg/grep emitting a file's hits consecutively. None when
too few matches to be worth it."""
if len(self.matches) < self._DENSIFY_MIN_MATCHES:
return None
lines: list[str] = []
@@ -142,8 +139,7 @@ class SearchResult:
result["matches_text"] = dense
else:
result["matches"] = [
{"path": m.path, "line": m.line_number, "content": m.content}
for m in self.matches
{"path": m.path, "line": m.line_number, "content": m.content} for m in self.matches
]
if self.files:
result["files"] = self.files
@@ -198,9 +194,7 @@ def _strip_terminal_fence_leaks(text: str) -> str:
cleaned_lines: List[str] = []
for line in text.splitlines(keepends=True):
had_terminal_wrapper = "__HERMES_FENCE_" in line or "\x1b]" in line
cleaned = _OSC_SEQUENCE_RE.sub("", line)
cleaned = _FENCE_MARKER_RE.sub("", cleaned)
cleaned = cleaned.replace("\x07", "")
cleaned = _FENCE_MARKER_RE.sub("", _OSC_SEQUENCE_RE.sub("", line)).replace("\x07", "")
if had_terminal_wrapper and cleaned.strip("'\r\n\t ") == "":
continue
cleaned_lines.append(cleaned)
@@ -209,15 +203,10 @@ def _strip_terminal_fence_leaks(text: str) -> str:
def _detect_line_ending(sample: str) -> Optional[str]:
"""Dominant line ending of ``sample`` (``\\r\\n`` if any CRLF in the first 4KB,
else ``\\n``), or None for empty/single-line content.
Used to preserve a file's endings across write_file/patch: the agent's bare-LF
tool args would otherwise silently normalize CRLF files, and patch would
produce mixed endings when only the substituted region changes.
"""
if not sample:
return None
head = sample[:4096]
else ``\\n``), or None for empty/single-line content. Preserves a file's
endings across write_file/patch: bare-LF tool args would otherwise silently
normalize CRLF files, and patch would produce mixed endings."""
head = sample[:4096] if sample else ""
if "\r\n" in head:
return "\r\n"
if "\n" in head:
@@ -239,15 +228,14 @@ def _normalize_line_endings(text: str, target: str) -> str:
# UTF-8 BOM (EF BB BF == U+FEFF), prepended by some Windows editors. Stripped on
# read so the model never sees a phantom first character (and patch's first-line
# match works), restored on write when the on-disk file had one — mirroring the
# line-ending preservation above.
# match works), restored on write when the on-disk file had one.
_UTF8_BOM = "\ufeff"
def _strip_bom(text: str) -> tuple[str, bool]:
"""Return (text-without-leading-BOM, had_bom). Only a leading BOM is
stripped; mid-content U+FEFF is legitimate data."""
if text and text.startswith(_UTF8_BOM):
if _has_bom(text):
return text[len(_UTF8_BOM):], True
return text, False
@@ -281,16 +269,12 @@ def normalize_read_pagination(offset: Any = DEFAULT_READ_OFFSET,
like ``0,-1p`` (schemas declare bounds, but not every caller enforces them).
The ``limit`` ceiling is ``tool_output.max_lines`` from config.yaml."""
from tools.tool_output_limits import get_max_lines
max_lines = get_max_lines()
normalized_offset = max(1, _coerce_int(offset, DEFAULT_READ_OFFSET))
normalized_limit = _coerce_int(limit, DEFAULT_READ_LIMIT)
normalized_limit = max(1, min(normalized_limit, max_lines))
normalized_limit = max(1, min(_coerce_int(limit, DEFAULT_READ_LIMIT), get_max_lines()))
return normalized_offset, normalized_limit
def normalize_search_pagination(offset: Any = DEFAULT_SEARCH_OFFSET,
limit: Any = DEFAULT_SEARCH_LIMIT) -> tuple[int, int]:
"""Return safe search pagination bounds for shell head/tail pipelines."""
normalized_offset = max(0, _coerce_int(offset, DEFAULT_SEARCH_OFFSET))
normalized_limit = max(1, _coerce_int(limit, DEFAULT_SEARCH_LIMIT))
return normalized_offset, normalized_limit
return max(0, _coerce_int(offset, DEFAULT_SEARCH_OFFSET)), max(1, _coerce_int(limit, DEFAULT_SEARCH_LIMIT))

View File

@@ -1,8 +1,7 @@
"""Syntax-lint and LSP-diagnostics tier for ``tools.file_operations``.
Extracted from ``ShellFileOperations`` as ``LintMixin``; the class inherits it
so every ``self._check_lint(...)`` call resolves unchanged via the MRO. Module
constants are re-imported into ``tools.file_operations`` for back-compat.
``ShellFileOperations`` inherits ``LintMixin``; module constants and in-process
linters are pure and re-imported into ``tools.file_operations`` for back-compat.
"""
import ast
@@ -26,16 +25,14 @@ LINTERS = {
# Extensions whose per-file shell linter is structurally weaker than a real LSP
# server and floods phantom errors on real projects: single-file ``tsc`` ignores
# tsconfig (no-lib/ES5 → every ES2015+ stdlib name "missing"), ``go vet`` fails
# outside a module, ``rustfmt --check`` is style-only and rejects non-Cargo files.
# tsconfig, ``go vet`` fails outside a module, ``rustfmt --check`` is style-only.
# When an LSP server claims the file, ``_check_lint`` skips the shell linter for
# these; py_compile / node --check are file-local and correct so always run.
_SHELL_LINTER_LSP_REDUNDANT = frozenset({'.ts', '.go', '.rs'})
# Output substrings meaning the linter binary exists but could not actually run
# (tooling gap, not a lint failure). ``_check_lint`` then returns ``skipped`` so
# the write isn't flagged and the LSP tier (which gates on ok/skipped) still runs.
# Matched case-insensitively.
# (tooling gap, not a lint failure) → ``skipped`` so the write isn't flagged and
# the LSP tier (which gates on ok/skipped) still runs. Matched case-insensitively.
_LINTER_UNUSABLE_PATTERNS = {
'npx': (
'this is not the tsc command you are looking for', # tsc not installed locally
@@ -77,10 +74,9 @@ def _lint_yaml_inproc(content: str) -> tuple[bool, str]:
"""In-process YAML syntax check; ``__SKIP__`` when PyYAML is missing.
Syntax-only (``yaml.parse``), NOT ``safe_load``: loading rejects valid YAML
that isn't one plain document — multi-doc ``---`` streams (ComposerError) and
app tags like CloudFormation ``!Sub`` / Ansible ``!vault`` (ConstructorError).
This verdict is a fail-closed WRITE gate, so a false positive refuses a
legitimate write; ``parse`` still catches real scanner/parser errors.
that isn't one plain document — multi-doc ``---`` streams and app tags like
CloudFormation ``!Sub`` / Ansible ``!vault``. This verdict is a fail-closed
WRITE gate, so a false positive refuses a legitimate write.
"""
try:
import yaml as _yaml
@@ -144,7 +140,6 @@ class LintMixin:
"""Syntax-check ``path``: in-process linter when one matches the
extension (``content`` avoids a re-read), else the shell linter table."""
ext = os.path.splitext(path)[1].lower()
inproc = LINTERS_INPROC.get(ext)
if inproc is not None:
if content is None:
@@ -156,94 +151,60 @@ class LintMixin:
if err == "__SKIP__":
return LintResult(skipped=True, message=f"No linter available for {ext} (missing dependency)")
return LintResult(success=ok, output="" if ok else err)
if ext not in LINTERS:
return LintResult(skipped=True, message=f"No linter for {ext} files")
# Single-file tsc can't read the project's tsconfig.json, so for project
# .ts files it floods phantom TS2307/TS2339 errors the delta filter then
# misreports as "pre-existing"; skip and let the LSP tier speak.
if ext == '.ts' and self._has_ancestor_tsconfig(path):
return LintResult(
skipped=True,
message=(
"Project tsconfig.json detected — per-file tsc skipped "
"(single-file tsc can't resolve project aliases/globals; "
"use the LSP tier or `tsc -p tsconfig.json` for real "
"diagnostics)."
),
)
return LintResult(skipped=True, message=(
"Project tsconfig.json detected — per-file tsc skipped "
"(single-file tsc can't resolve project aliases/globals; "
"use the LSP tier or `tsc -p tsconfig.json` for real "
"diagnostics)."
))
if ext in _SHELL_LINTER_LSP_REDUNDANT and self._lsp_will_handle(path):
return LintResult(
skipped=True,
message=f"LSP server handles {ext} — shell linter skipped",
)
return LintResult(skipped=True, message=f"LSP server handles {ext} — shell linter skipped")
linter_cmd = LINTERS[ext]
base_cmd = linter_cmd.split()[0]
if not self._has_command(base_cmd):
return LintResult(skipped=True, message=f"{base_cmd} not available")
# Linters are native Windows binaries on Windows: they need the C:/...
# form, not MSYS /c/... (node would resolve it as C:\c\Users\... → phantom ENOENT).
cmd = linter_cmd.replace("{file}", self._escape_native_tool_arg(path))
result = self._exec(cmd, timeout=30)
result = self._exec(linter_cmd.replace("{file}", self._escape_native_tool_arg(path)), timeout=30)
if result.exit_code != 0 and _looks_like_linter_unusable(base_cmd, result.stdout):
from tools.ansi_strip import strip_ansi
cleaned = strip_ansi(result.stdout).strip()
# Collapse to one line — the npx banner is multi-line ASCII art.
first_line = next(
(ln.strip() for ln in cleaned.splitlines() if ln.strip()),
cleaned[:120],
)
return LintResult(
skipped=True,
message=f"{base_cmd} not usable: {first_line[:200]}",
)
return LintResult(
success=result.exit_code == 0,
output=result.stdout.strip() if result.stdout.strip() else ""
)
first_line = next((ln.strip() for ln in cleaned.splitlines() if ln.strip()), cleaned[:120])
return LintResult(skipped=True, message=f"{base_cmd} not usable: {first_line[:200]}")
return LintResult(success=result.exit_code == 0, output=result.stdout.strip())
def _check_lint_delta(self, path: str, pre_content: Optional[str],
post_content: Optional[str] = None) -> LintResult:
"""Post-write lint; when it fails and ``pre_content`` is known, report
only errors this edit introduced (pre-existing lines filtered out).
Semantic (LSP) diagnostics are a separate channel — see
``_maybe_lsp_diagnostics`` — so syntax and semantic signals stay distinct.
"""
"""Post-write lint; when it fails and ``pre_content`` is known, report only
errors this edit introduced (pre-existing lines filtered out). Semantic
(LSP) diagnostics are a separate channel — see ``_maybe_lsp_diagnostics``."""
post = self._check_lint(path, content=post_content)
if post.success or post.skipped or pre_content is None:
return post
pre = self._check_lint(path, content=pre_content)
if pre.success or pre.skipped or not pre.output:
return post # pre-write was clean (or unlintable): all post errors are new
# Single-error parsers (ast.parse, json.loads) stop at the first error, so
# if every post error already existed we can't prove the edit is clean —
# report the file as still broken but say nothing new was introduced.
pre_lines = {ln.strip() for ln in pre.output.splitlines() if ln.strip()}
post_lines = [ln for ln in post.output.splitlines() if ln.strip() and ln.strip() not in pre_lines]
if not post_lines:
return LintResult(
success=False,
output=post.output,
success=False, output=post.output,
message="Pre-existing lint errors — this edit didn't introduce new ones but the file is still broken.",
)
return LintResult(
success=False,
output=(
"New lint errors introduced by this edit "
"(pre-existing errors filtered out):\n" + "\n".join(post_lines)
)
)
return LintResult(success=False, output=(
"New lint errors introduced by this edit "
"(pre-existing errors filtered out):\n" + "\n".join(post_lines)
))
def _lsp_local_only(self) -> bool:
"""True iff wired to a local backend. LSP servers run on the host and
@@ -259,10 +220,7 @@ class LintMixin:
def _lsp_service(self):
"""The active LSPService, or None on a non-local backend / any failure.
Shared best-effort probe: LSP is an enrichment layer and must never
break a write, so every failure path collapses to None.
"""
LSP is an enrichment layer and must never break a write."""
if not self._lsp_local_only():
return None
try:
@@ -327,13 +285,8 @@ class LintMixin:
except Exception: # noqa: BLE001
pass
def _maybe_lsp_diagnostics(
self,
path: str,
*,
pre_content: Optional[str] = None,
post_content: Optional[str] = None,
) -> str:
def _maybe_lsp_diagnostics(self, path: str, *, pre_content: Optional[str] = None,
post_content: Optional[str] = None) -> str:
"""Formatted LSP diagnostics introduced by this edit, or "" when LSP is
unavailable/disabled/clean. Never raises past the service probe.
@@ -344,7 +297,6 @@ class LintMixin:
svc = self._lsp_service()
if svc is None or not svc.enabled_for(path):
return ""
line_shift = None
if pre_content is not None and post_content is not None and pre_content != post_content:
try:
@@ -352,7 +304,6 @@ class LintMixin:
line_shift = build_line_shift(pre_content, post_content)
except Exception: # noqa: BLE001
line_shift = None
try:
diagnostics = svc.get_diagnostics_sync(path, delta=True, line_shift=line_shift)
except Exception: # noqa: BLE001

View File

@@ -1,8 +1,7 @@
"""Content/file search tier for ``tools.file_operations``.
Extracted from ``ShellFileOperations`` as ``SearchMixin``; the class inherits it
so every ``self._search_*`` call resolves unchanged via the MRO. Module-level
helpers are re-imported into ``tools.file_operations`` for back-compat.
``ShellFileOperations`` inherits ``SearchMixin``; module-level helpers are pure
(no I/O) and re-imported into ``tools.file_operations`` for back-compat.
"""
import os
@@ -19,11 +18,7 @@ _MACOS_TCC_PROTECTED_HOME_DIRS = (
def _macos_protected_search_exclusions(
path: str,
*,
cwd: Optional[str] = None,
home: Optional[str] = None,
platform: Optional[str] = None,
path: str, *, cwd: Optional[str] = None, home: Optional[str] = None, platform: Optional[str] = None,
) -> List[str]:
"""Protected home dirs (relative to ``path``) below a broad macOS search root.
@@ -33,14 +28,11 @@ def _macos_protected_search_exclusions(
"""
if (platform or sys.platform) != "darwin":
return []
home_path = Path(home or Path.home()).expanduser()
root = Path(path).expanduser()
if not root.is_absolute():
root = Path(cwd or os.getcwd()) / root
root = Path(os.path.normpath(str(root)))
home_path = Path(os.path.normpath(str(home_path)))
home_path = Path(os.path.normpath(str(Path(home or Path.home()).expanduser())))
exclusions: List[str] = []
for dirname in _MACOS_TCC_PROTECTED_HOME_DIRS:
try:
@@ -70,13 +62,13 @@ _SEARCH_OUTPUT_RE = re.compile(r'^([A-Za-z]:)?[^\s:][^\n]*?[:\-]\d|^[^\s:][^\s]*
def _split_tool_diagnostics(output: str) -> tuple[str, str]:
"""Separate rg/grep diagnostic lines from real match output.
"""Separate rg/grep diagnostic lines from real match output → ``(diagnostics, payload)``.
``_exec`` merges stderr into stdout, so tool errors interleave with matches.
Returns ``(diagnostics, payload)``. Classifying by SHAPE (not error prefix)
lets the exit-2 guard tell a pure failure (no payload → surface the error)
from a partial one (one unreadable file, others matched → keep matches), and
guarantees error text is never parsed as a match.
Classifying by SHAPE (not error prefix) lets the exit-2 guard tell a pure
failure (no payload → surface the error) from a partial one (one unreadable
file, others matched → keep matches), and guarantees error text is never
parsed as a match.
"""
diagnostics: list[str] = []
payload: list[str] = []
@@ -85,11 +77,9 @@ def _split_tool_diagnostics(output: str) -> tuple[str, str]:
continue
# Prefix check first: a real match path can contain "-<digit>" (e.g.
# ".../pytest-686/..."), which the shape regex would accept as a match.
stripped = line.lstrip()
if stripped.startswith("rg: ") or stripped.startswith("grep: "):
if line.lstrip().startswith(("rg: ", "grep: ")):
diagnostics.append(line)
continue
if line == "--" or _SEARCH_OUTPUT_RE.match(line):
elif line == "--" or _SEARCH_OUTPUT_RE.match(line):
payload.append(line)
else:
diagnostics.append(line)
@@ -97,22 +87,17 @@ def _split_tool_diagnostics(output: str) -> tuple[str, str]:
def _parse_search_context_line(line: str) -> tuple[str, int, str] | None:
"""Parse a ``path-line-content`` context line.
Filenames may contain ``-<digits>-`` segments, so use the RIGHTMOST numeric
separator: ``dir/file-12-name.py-8-context`` → (``dir/file-12-name.py``, 8).
"""
"""Parse a ``path-line-content`` context line using the RIGHTMOST numeric
separator (filenames may contain ``-<digits>-`` segments):
``dir/file-12-name.py-8-context`` → (``dir/file-12-name.py``, 8, ``context``)."""
if not line or line == "--":
return None
match = None
for candidate in re.finditer(r'-(\d+)-', line):
match = candidate
if match is None:
if match is None or match.start() == 0:
return None
path = line[:match.start()]
if not path:
return None
return path, int(match.group(1)), line[match.end():]
return line[:match.start()], int(match.group(1)), line[match.end():]
_REGEX_NEWLINE_ESCAPE_RE = re.compile(r"(?<!\\)(?:\\\\)*\\n")
@@ -127,9 +112,7 @@ def _pattern_has_regex_newline(pattern: str) -> bool:
def _is_line_oriented_newline_error(error: Optional[str]) -> bool:
"""Return True for rg's hard error when multiline mode is required."""
if not error:
return False
return "literal \"\\n\" is not allowed" in error and "--multiline" in error
return bool(error) and "literal \"\\n\" is not allowed" in error and "--multiline" in error
def _maybe_warn_line_oriented_newline_pattern(result: SearchResult, pattern: str) -> SearchResult:
@@ -170,17 +153,12 @@ def _parse_search_output(result, output_mode: str, limit: int, offset: int,
if result.exit_code == 2 and not payload.strip():
error_msg = diagnostics.strip() or result.stdout.strip() or "Search error"
return SearchResult(error=f"Search failed: {error_msg}", total_count=0)
lines = [ln for ln in payload.strip().split('\n') if ln]
if output_mode == "files_only":
return SearchResult(
files=lines[offset:offset + limit],
total_count=len(lines),
truncated=bool(limit_reason),
limit_reason=limit_reason,
warning=warning,
files=lines[offset:offset + limit], total_count=len(lines),
truncated=bool(limit_reason), limit_reason=limit_reason, warning=warning,
)
if output_mode == "count":
counts = {}
for line in lines:
@@ -191,12 +169,9 @@ def _parse_search_output(result, output_mode: str, limit: int, offset: int,
except ValueError:
pass
return SearchResult(
counts=counts,
total_count=sum(counts.values()),
truncated=bool(limit_reason),
limit_reason=limit_reason,
counts=counts, total_count=sum(counts.values()),
truncated=bool(limit_reason), limit_reason=limit_reason,
)
matches = []
for line in lines:
if line == "--":
@@ -204,9 +179,7 @@ def _parse_search_output(result, output_mode: str, limit: int, offset: int,
m = _MATCH_LINE_RE.match(line)
if m:
matches.append(SearchMatch(
path=(m.group(1) or '') + m.group(2),
line_number=int(m.group(3)),
content=m.group(4)[:500],
path=(m.group(1) or '') + m.group(2), line_number=int(m.group(3)), content=m.group(4)[:500],
))
continue
# Context lines ("file-line-content") only when context was requested,
@@ -214,19 +187,18 @@ def _parse_search_output(result, output_mode: str, limit: int, offset: int,
if context > 0:
parsed = _parse_search_context_line(line)
if parsed:
matches.append(SearchMatch(
path=parsed[0], line_number=parsed[1], content=parsed[2][:500],
))
matches.append(SearchMatch(path=parsed[0], line_number=parsed[1], content=parsed[2][:500]))
total = len(matches)
return SearchResult(
matches=matches[offset:offset + limit],
total_count=total,
truncated=total > offset + limit or bool(limit_reason),
limit_reason=limit_reason,
warning=warning,
matches=matches[offset:offset + limit], total_count=total,
truncated=total > offset + limit or bool(limit_reason), limit_reason=limit_reason, warning=warning,
)
def _has_hidden_part(parts) -> bool:
return any(part not in {".", ".."} and part.startswith(".") for part in parts)
class SearchMixin:
"""File-name and content search via rg with find/grep fallbacks. Requires
``_exec``, ``_has_command``, ``_expand_path``, ``_escape_shell_arg``,
@@ -246,51 +218,49 @@ class SearchMixin:
return []
from tools import file_operations as _fo # lazy: _HOME is monkeypatched there
cwd = getattr(self.env, "cwd", None) or self.cwd
return _macos_protected_search_exclusions(
path, cwd=cwd, home=_fo._HOME, platform=sys.platform
)
return _macos_protected_search_exclusions(path, cwd=cwd, home=_fo._HOME, platform=sys.platform)
def _protected_prune_paths(self, path: str) -> List[str]:
"""Absolute-ish protected paths for find's ``-path ... -prune``."""
return [
os.path.normpath(os.path.join(path, item))
for item in self._macos_search_exclusions(path)
]
return [os.path.normpath(os.path.join(path, item)) for item in self._macos_search_exclusions(path)]
def _prune_expr(self, protected_paths: List[str]) -> str:
"""find ``\\( -path A -o -path B \\) -prune`` clause for the protected dirs."""
terms = " -o ".join(f"-path {self._escape_shell_arg(item)}" for item in protected_paths)
return f"\\( {terms} \\) -prune"
def _rg_exclusion_globs(self, path: str) -> List[str]:
"""``--glob '!<dir>/**'`` pairs excluding protected dirs from an rg run."""
out: List[str] = []
for item in self._macos_search_exclusions(path):
out.extend(["--glob", self._escape_shell_arg(f"!{item}/**")])
return out
def _path_exists_probe(self, path: str) -> str:
"""Stdout of the existence probe: contains "exists" or "not_found"."""
return self._exec(
f"test -e {self._escape_shell_arg(path)} && echo exists || echo not_found"
).stdout
return self._exec(f"test -e {self._escape_shell_arg(path)} && echo exists || echo not_found").stdout
def _dispatch_search(self, pattern: str, path: str, target: str,
file_glob: Optional[str], limit: int, offset: int,
output_mode: str, context: int) -> SearchResult:
if target == "files":
return self._search_files(pattern, path, limit, offset)
return self._search_content(pattern, path, file_glob, limit, offset,
output_mode, context)
return self._search_content(pattern, path, file_glob, limit, offset, output_mode, context)
def _path_not_found_result(self, path: str) -> SearchResult:
"""Error result for a missing search root, with nearby-entry suggestions."""
parent = os.path.dirname(path) or "."
basename_query = os.path.basename(path)
hint_parts = [f"Path not found: {path}"]
parent_check = self._exec(
f"test -d {self._escape_shell_arg(parent)} && echo yes || echo no"
)
parent_check = self._exec(f"test -d {self._escape_shell_arg(parent)} && echo yes || echo no")
if "yes" in parent_check.stdout and basename_query:
ls_result = self._exec(
f"ls -1 {self._escape_shell_arg(parent)} 2>/dev/null | head -20"
)
ls_result = self._exec(f"ls -1 {self._escape_shell_arg(parent)} 2>/dev/null | head -20")
if ls_result.exit_code == 0 and ls_result.stdout.strip():
lower_q = basename_query.lower()
candidates = []
for entry in ls_result.stdout.strip().split('\n'):
if not entry:
continue
le = entry.lower()
if lower_q in le or le in lower_q or le.startswith(lower_q[:3]):
if entry and (lower_q in le or le in lower_q or le.startswith(lower_q[:3])):
candidates.append(os.path.join(parent, entry))
if candidates:
hint_parts.append("Similar paths: " + ", ".join(candidates[:5]))
@@ -311,11 +281,9 @@ class SearchMixin:
(existing if "exists" in self._path_exists_probe(expanded) else missing).append(expanded)
if not existing:
return None
merged = SearchResult()
for p in existing:
sub = self._dispatch_search(pattern, p, target, file_glob, limit, offset,
output_mode, context)
sub = self._dispatch_search(pattern, p, target, file_glob, limit, offset, output_mode, context)
if sub.error:
continue
merged.matches.extend(sub.matches)
@@ -346,8 +314,7 @@ class SearchMixin:
"(or pass a simpler substring)."),
)
def _zero_match_probe(self, pattern: str, path: str,
file_glob: Optional[str]) -> Optional[str]:
def _zero_match_probe(self, pattern: str, path: str, file_glob: Optional[str]) -> Optional[str]:
"""Steering hint for a 0-match content search, or None.
A bare zero gives the model nothing to act on, so run cheap count-only rg
@@ -383,13 +350,6 @@ class SearchMixin:
"""Search for files by name (glob-like): rg --files, else find."""
search_pattern = pattern if (not pattern.startswith('**/') and '/' not in pattern) \
else pattern.split('/')[-1]
search_root = Path(path)
has_hidden_path_ancestor = any(
part not in {".", ".."} and part.startswith(".")
for part in search_root.parts
)
# rg respects .gitignore, skips hidden dirs, and walks in parallel (~200x find).
if self._has_command('rg'):
return self._search_files_rg(search_pattern, path, limit, offset)
@@ -399,21 +359,15 @@ class SearchMixin:
"Install ripgrep for best results: "
"https://github.com/BurntSushi/ripgrep#installation"
)
# Hidden roots: find's path filter would exclude everything under the root,
# so gather full output and filter descendants in Python (pagination too).
search_root = Path(path)
has_hidden_path_ancestor = _has_hidden_part(search_root.parts)
hidden_filter_expr = "" if has_hidden_path_ancestor else " -not -path '*/.*'"
pagination_expr = "" if has_hidden_path_ancestor else f" | tail -n +{offset + 1} | head -n {limit}"
# Prune protected dirs BEFORE traversal so macOS never sees an access attempt.
protected_paths = self._protected_prune_paths(path)
prune_expr = ""
if protected_paths:
prune_terms = " -o ".join(
f"-path {self._escape_shell_arg(item)}" for item in protected_paths
)
prune_expr = f" \\( {prune_terms} \\) -prune -o"
prune_expr = f" {self._prune_expr(protected_paths)} -o" if protected_paths else ""
base = (f"find {self._escape_shell_arg(path)}{prune_expr}{hidden_filter_expr} "
f"-type f -name {self._escape_shell_arg(search_pattern)} ")
result = self._exec(f"{base}-printf '%T@ %p\\n' 2>/dev/null | sort -rn{pagination_expr}", timeout=60)
@@ -422,14 +376,12 @@ class SearchMixin:
# BSD find (macOS) has no -printf.
result = self._exec(f"{base}2>/dev/null | sort -rn{pagination_expr}", timeout=60)
stdout, limit_reason = _search_stdout_and_limit(result)
files = []
for line in stdout.strip().split('\n'):
if not line:
continue
parts = line.split(' ', 1)
files.append(parts[1] if len(parts) == 2 and parts[0].replace('.', '').isdigit() else line)
if has_hidden_path_ancestor:
normalized_root = search_root.resolve()
filtered_files = []
@@ -438,65 +390,46 @@ class SearchMixin:
rel_parts = Path(file_path).resolve().relative_to(normalized_root).parts
except ValueError:
rel_parts = Path(file_path).parts
if any(part not in {".", ".."} and part.startswith(".") for part in rel_parts):
continue
filtered_files.append(file_path)
if not _has_hidden_part(rel_parts):
filtered_files.append(file_path)
files = filtered_files[offset:offset + limit]
return SearchResult(
files=files,
total_count=len(files),
truncated=bool(limit_reason),
limit_reason=limit_reason,
)
return SearchResult(files=files, total_count=len(files), truncated=bool(limit_reason), limit_reason=limit_reason)
def _search_files_rg(self, pattern: str, path: str, limit: int, offset: int) -> SearchResult:
"""File-name search via ``rg --files``, mtime-sorted when rg >= 13 supports --sortr."""
# Wrap bare names so -g matches at any depth (equivalent to find -name).
glob_pattern = f"*{pattern}" if ('/' not in pattern and not pattern.startswith('*')) else pattern
fetch_limit = limit + offset
exclusion_globs = " ".join(
f"--glob {self._escape_shell_arg(f'!{item}/**')}"
for item in self._macos_search_exclusions(path)
)
exclusion_globs = " ".join(self._rg_exclusion_globs(path))
exclusion_args = f" {exclusion_globs}" if exclusion_globs else ""
tail = (f"-g {self._escape_shell_arg(glob_pattern)}{exclusion_args} "
f"{self._escape_native_tool_arg(path)} 2>/dev/null | head -n {fetch_limit}")
result = self._exec(f"rg --files --sortr=modified {tail}", timeout=60)
stdout, limit_reason = _search_stdout_and_limit(result)
all_files = [f for f in stdout.strip().split('\n') if f]
if not all_files and not limit_reason:
# --sortr may have failed on older rg; retry without it.
result = self._exec(f"rg --files {tail}", timeout=60)
stdout, limit_reason = _search_stdout_and_limit(result)
all_files = [f for f in stdout.strip().split('\n') if f]
return SearchResult(
files=all_files[offset:offset + limit],
total_count=len(all_files),
truncated=len(all_files) >= fetch_limit or bool(limit_reason),
limit_reason=limit_reason,
files=all_files[offset:offset + limit], total_count=len(all_files),
truncated=len(all_files) >= fetch_limit or bool(limit_reason), limit_reason=limit_reason,
)
def _search_content(self, pattern: str, path: str, file_glob: Optional[str],
limit: int, offset: int, output_mode: str, context: int) -> SearchResult:
"""Content search: rg, else grep; attaches zero-match steering hints."""
used_rg = False
if self._has_command('rg'):
used_rg = True
result = self._search_with_rg(pattern, path, file_glob, limit, offset,
output_mode, context)
used_rg = self._has_command('rg')
if used_rg:
result = self._search_with_rg(pattern, path, file_glob, limit, offset, output_mode, context)
elif self._has_command('grep'):
result = self._search_with_grep(pattern, path, file_glob, limit, offset,
output_mode, context)
result = self._search_with_grep(pattern, path, file_glob, limit, offset, output_mode, context)
else:
return SearchResult(
error="Content search requires ripgrep (rg) or grep. "
"Install ripgrep: https://github.com/BurntSushi/ripgrep#installation"
)
if (not result.error and result.total_count == 0
and not result.matches and not result.files and not result.counts):
try:
@@ -505,18 +438,30 @@ class SearchMixin:
hint = None
if hint:
result.warning = hint if not result.warning else f"{result.warning} {hint}"
# rg auto-enables --multiline for \n patterns, so the line-oriented
# explanation only applies to the grep fallback.
if used_rg:
return result
return _maybe_warn_line_oriented_newline_pattern(result, pattern)
def _run_search_pipeline(self, cmd_parts: List[str], output_mode: str, limit: int,
offset: int, context: int, warning: Optional[str] = None) -> SearchResult:
"""Run ``cmd_parts | head -n <fetch_limit>`` under pipefail and parse.
Extra rows are fetched to report the true total; context mode also emits
"--" separators, so grab 200 more and filter in Python. pipefail keeps the
engine's exit 2 alive across ``| head`` (a truncating head makes rg exit 0
/ grep exit 141 on SIGPIPE, neither of which the strict ==2 guard flags).
"""
fetch_limit = limit + offset + (200 if context > 0 else 0)
cmd = "set -o pipefail; " + " ".join(cmd_parts + ["|", "head", "-n", str(fetch_limit)])
result = self._exec(cmd, timeout=60)
return _parse_search_output(result, output_mode, limit, offset, context, warning=warning)
def _search_with_rg(self, pattern: str, path: str, file_glob: Optional[str],
limit: int, offset: int, output_mode: str, context: int) -> SearchResult:
"""Search using ripgrep."""
cmd_parts = ["rg", "--line-number", "--no-heading", "--with-filename"]
# A regex \n can't match in line-oriented mode (rg hard-errors); enable -U
# up front when the pattern clearly wants to cross lines, and say so.
multiline = _pattern_has_regex_newline(pattern)
@@ -524,8 +469,7 @@ class SearchMixin:
cmd_parts.append("--multiline")
if context > 0:
cmd_parts.extend(["-C", str(context)])
for item in self._macos_search_exclusions(path):
cmd_parts.extend(["--glob", self._escape_shell_arg(f"!{item}/**")])
cmd_parts.extend(self._rg_exclusion_globs(path))
if file_glob:
cmd_parts.extend(["--glob", self._escape_shell_arg(file_glob)])
if output_mode in _OUTPUT_MODE_FLAGS:
@@ -533,39 +477,26 @@ class SearchMixin:
cmd_parts.append(self._escape_shell_arg(pattern))
# rg is a native Windows binary (winget/cargo/choco): needs C:/... not MSYS /c/...
cmd_parts.append(self._escape_native_tool_arg(path))
# Fetch extra rows to report the true total; context mode also emits "--"
# separators, so grab generously and filter in Python.
fetch_limit = limit + offset + 200 if context > 0 else limit + offset
cmd_parts.extend(["|", "head", "-n", str(fetch_limit)])
# pipefail so rg's exit 2 survives `| head` (else head's 0 masks it). rg
# exits 0 on SIGPIPE from a truncating head, so no false errors.
cmd = "set -o pipefail; " + " ".join(cmd_parts)
result = self._exec(cmd, timeout=60)
ml_note = (
"Pattern contains \\n — multiline mode (-U) was enabled automatically "
"so the regex can match across line boundaries."
) if multiline else None
return _parse_search_output(result, output_mode, limit, offset, context, warning=ml_note)
return self._run_search_pipeline(cmd_parts, output_mode, limit, offset, context, warning=ml_note)
def _search_with_grep(self, pattern: str, path: str, file_glob: Optional[str],
limit: int, offset: int, output_mode: str, context: int) -> SearchResult:
"""Fallback search using grep."""
# -H forces filenames; -E matches rg regex behavior; --exclude-dir='.*'
# mirrors rg's hidden-dir default (.git/, .hub/index-cache/, ...).
cmd_parts = ["grep", "-rnHE", "--exclude-dir='.*'"]
# grep's --exclude-dir matches BASENAMES anywhere in the tree, so it can't
# express "only the home-level Downloads"; route protected-dir pruning
# through find's path-scoped -prune instead.
protected_paths = self._protected_prune_paths(path)
if protected_paths:
return self._search_with_grep_pruned(
pattern, path, file_glob, limit, offset, output_mode, context,
protected_paths,
pattern, path, file_glob, limit, offset, output_mode, context, protected_paths,
)
# -H forces filenames; -E matches rg regex behavior; --exclude-dir='.*'
# mirrors rg's hidden-dir default (.git/, .hub/index-cache/, ...).
cmd_parts = ["grep", "-rnHE", "--exclude-dir='.*'"]
if context > 0:
cmd_parts.extend(["-C", str(context)])
if file_glob:
@@ -573,13 +504,10 @@ class SearchMixin:
if output_mode in _OUTPUT_MODE_FLAGS:
cmd_parts.append(_OUTPUT_MODE_FLAGS[output_mode])
cmd_parts.append(self._escape_shell_arg(pattern))
# grep applies --exclude-dir to the search root too, so a relative root
# "." would be excluded by '.*'. Anchor relative paths at the shell's
# live $PWD (quoted separately so user paths stay escaped).
is_absolute = path.startswith(("/", "\\\\")) or bool(
re.match(r"^[A-Za-z]:[\\/]", path)
)
is_absolute = path.startswith(("/", "\\\\")) or bool(re.match(r"^[A-Za-z]:[\\/]", path))
if is_absolute:
search_root = self._escape_shell_arg(path)
else:
@@ -588,15 +516,7 @@ class SearchMixin:
if relative_path not in {"", "."}:
search_root += f"/{self._escape_shell_arg(relative_path)}"
cmd_parts.append(search_root)
fetch_limit = limit + offset + (200 if context > 0 else 0)
cmd_parts.extend(["|", "head", "-n", str(fetch_limit)])
# pipefail so grep's exit 2 survives `| head`; a truncating head makes
# grep exit 141 (SIGPIPE), which the strict ==2 guard ignores.
cmd = "set -o pipefail; " + " ".join(cmd_parts)
result = self._exec(cmd, timeout=60)
return _parse_search_output(result, output_mode, limit, offset, context)
return self._run_search_pipeline(cmd_parts, output_mode, limit, offset, context)
def _search_with_grep_pruned(self, pattern: str, path: str, file_glob: Optional[str],
limit: int, offset: int, output_mode: str, context: int,
@@ -616,23 +536,13 @@ class SearchMixin:
if output_mode in _OUTPUT_MODE_FLAGS:
grep_parts.append(_OUTPUT_MODE_FLAGS[output_mode])
grep_parts.append(self._escape_shell_arg(pattern))
prune_terms = " -o ".join(
f"-path {self._escape_shell_arg(item)}" for item in protected_paths
)
find_parts = [
"find", self._escape_shell_arg(path or "."),
f"\\( {prune_terms} \\) -prune", "-o",
self._prune_expr(protected_paths), "-o",
"\\( -type d -name '.*' \\) -prune", "-o",
"-type f",
]
if file_glob:
find_parts.extend(["-name", self._escape_shell_arg(file_glob)])
find_parts.extend(["-exec", *grep_parts, "{}", "+"])
fetch_limit = limit + offset + (200 if context > 0 else 0)
cmd = (
"set -o pipefail; " + " ".join(find_parts)
+ f" 2>/dev/null | head -n {fetch_limit}"
)
result = self._exec(cmd, timeout=60)
return _parse_search_output(result, output_mode, limit, offset, context)
find_parts.extend(["-exec", *grep_parts, "{}", "+", "2>/dev/null"])
return self._run_search_pipeline(find_parts, output_mode, limit, offset, context)

View File

@@ -1,25 +1,12 @@
"""Spill oversized hook-injected context to disk with a preview placeholder.
Shell hooks and plugin ``pre_llm_call`` hooks can return ``{"context": ...}``
that is concatenated into the user message on EVERY subsequent API call, so a
large blob inflates every turn and breaks the prompt-cache prefix. Above a
configured budget the full text is written to a per-session directory and the
in-prompt payload becomes a head/tail preview plus the saved path.
Config (``config.yaml``)::
hooks:
output_spill:
enabled: true # default: true; set false to disable spilling
max_chars: 10000 # default; context above this is spilled
preview_head: 500 # chars shown at the start of the preview
preview_tail: 500 # chars shown at the end of the preview
directory: null # default: <HERMES_HOME>/hook_outputs
Invariants: unchanged input when disabled or under the cap; never raises —
an I/O failure still returns a bounded preview with an in-prompt notice. Spill
files are grouped per session so ``/new`` sessions don't pile into one directory.
Ported from openai/codex PR #21069.
Hook ``{"context": ...}`` output is concatenated into the user message on EVERY
subsequent API call, so a large blob inflates every turn and breaks the
prompt-cache prefix. Above ``hooks.output_spill.max_chars`` (default 10000) the
full text is written under ``hooks.output_spill.directory`` (default
``<HERMES_HOME>/hook_outputs/<session>``) and the in-prompt payload becomes a
``preview_head``/``preview_tail`` excerpt plus the saved path. ``enabled: false``
disables. Never raises: an I/O failure still returns a bounded preview.
"""
from __future__ import annotations
@@ -48,27 +35,19 @@ def get_spill_config() -> Dict[str, Any]:
from hermes_cli.config import load_config
cfg = load_config() or {}
hooks = cfg.get("hooks") if isinstance(cfg, dict) else None
if isinstance(hooks, dict):
sub = hooks.get("output_spill")
if isinstance(sub, dict):
section = sub
if isinstance(hooks, dict) and isinstance(hooks.get("output_spill"), dict):
section = hooks["output_spill"]
except Exception:
section = {}
enabled_raw = section.get("enabled", DEFAULT_ENABLED)
enabled = bool(enabled_raw) if enabled_raw is not None else DEFAULT_ENABLED
directory = section.get("directory")
if directory is not None and not isinstance(directory, str):
directory = None
return {
"enabled": enabled,
"enabled": bool(enabled_raw) if enabled_raw is not None else DEFAULT_ENABLED,
"max_chars": _coerce_positive_int(section.get("max_chars"), DEFAULT_MAX_CHARS),
# head/tail allow zero (empty tail), max_chars must be positive.
"preview_head": _coerce_int(section.get("preview_head"), DEFAULT_PREVIEW_HEAD, 0),
"preview_tail": _coerce_int(section.get("preview_tail"), DEFAULT_PREVIEW_TAIL, 0),
"directory": directory,
"directory": directory if isinstance(directory, str) else None,
}
@@ -78,46 +57,13 @@ def _resolve_spill_dir(directory_override: Optional[str], session_id: Optional[s
base = Path(os.path.expanduser(directory_override))
else:
from hermes_constants import get_hermes_home
base = Path(get_hermes_home()) / "hook_outputs"
session_segment = session_id or "no-session"
session_segment = session_segment.replace("/", "_").replace("\\", "_").replace("..", "_")
session_segment = (session_id or "no-session").replace("/", "_").replace("\\", "_").replace("..", "_")
return base / session_segment
def _build_preview(
text: str,
head: int,
tail: int,
saved_path: Optional[str],
*,
source: str,
) -> str:
"""Assemble the in-prompt preview with head/tail and saved-path footer."""
total = len(text)
head_chunk = text[:head] if head > 0 else ""
tail_chunk = text[-tail:] if tail > 0 and total > head else ""
parts = [
f"[{source} output truncated — {total:,} chars; full content "
+ (f"saved to {saved_path}]" if saved_path else "unavailable — spill write failed]"),
]
if head_chunk:
parts.append("--- head ---")
parts.append(head_chunk)
if tail_chunk:
parts.append("--- tail ---")
parts.append(tail_chunk)
return "\n".join(parts)
def spill_if_oversized(
text: str,
*,
session_id: Optional[str] = None,
source: str = "hook",
config: Optional[Dict[str, Any]] = None,
text: str, *, session_id: Optional[str] = None, source: str = "hook", config: Optional[Dict[str, Any]] = None,
) -> str:
"""Spill ``text`` to disk if it exceeds the configured cap.
@@ -132,42 +78,40 @@ def spill_if_oversized(
text = str(text)
except Exception:
return ""
cfg = config if config is not None else get_spill_config()
if not cfg.get("enabled", True):
return text
max_chars = int(cfg.get("max_chars") or DEFAULT_MAX_CHARS)
if len(text) <= max_chars:
if len(text) <= int(cfg.get("max_chars") or DEFAULT_MAX_CHARS):
return text
head = int(cfg.get("preview_head") or 0)
tail = int(cfg.get("preview_tail") or 0)
directory_override = cfg.get("directory")
# A disk failure must never blow up the turn — fall through to a preview
# without a saved path.
saved_path: Optional[str] = None
try:
spill_dir = _resolve_spill_dir(directory_override, session_id)
spill_dir = _resolve_spill_dir(cfg.get("directory"), session_id)
from tools.spill_safety import ensure_spill_dir, write_text_exclusive
# Hook context may embed raw secrets: private perms + exclusive,
# symlink-refusing create (the per-session dir is predictable).
ensure_spill_dir(spill_dir, private=True)
spill_path = spill_dir / f"{uuid.uuid4().hex}.txt"
# Trailing newline so tail readers don't report "missing newline".
write_text_exclusive(
spill_path,
text if text.endswith("\n") else text + "\n",
private=True,
)
write_text_exclusive(spill_path, text if text.endswith("\n") else text + "\n", private=True)
saved_path = str(spill_path)
except Exception as exc:
logger.warning("hook output spill failed: %s", exc)
saved_path = None
return _build_preview(text, head, tail, saved_path, source=source)
total = len(text)
parts = [
f"[{source} output truncated — {total:,} chars; full content "
+ (f"saved to {saved_path}]" if saved_path else "unavailable — spill write failed]"),
]
if head > 0 and text[:head]:
parts.extend(["--- head ---", text[:head]])
if tail > 0 and total > head:
parts.extend(["--- tail ---", text[-tail:]])
return "\n".join(parts)
__all__ = [