refactor(tools): AST-neutral layout compaction of file_tools group
This commit is contained in:
@@ -139,16 +139,14 @@ class FileStateRegistry:
|
||||
f"{resolved} was modified by sibling subagent "
|
||||
f"{writer_tid!r} but this agent never read it. "
|
||||
"Read the file before writing to avoid overwriting "
|
||||
"the sibling's changes."
|
||||
)
|
||||
"the sibling's changes.")
|
||||
read_ts = stamp[1]
|
||||
if writer_ts > read_ts:
|
||||
return (
|
||||
f"{resolved} was modified by sibling subagent "
|
||||
f"{writer_tid!r} at {_fmt_ts(writer_ts)} — after "
|
||||
f"this agent's last read at {_fmt_ts(read_ts)}. "
|
||||
"Re-read the file before writing."
|
||||
)
|
||||
"Re-read the file before writing.")
|
||||
|
||||
if stamp is not None:
|
||||
read_mtime, _read_ts, partial = stamp
|
||||
@@ -156,20 +154,17 @@ class FileStateRegistry:
|
||||
return (
|
||||
f"{resolved} was modified since you last read it "
|
||||
"on disk (external edit or unrecorded writer). "
|
||||
"Re-read the file before writing."
|
||||
)
|
||||
"Re-read the file before writing.")
|
||||
if partial:
|
||||
return (
|
||||
f"{resolved} was last read with offset/limit pagination "
|
||||
"(partial view). Re-read the whole file before "
|
||||
"overwriting it."
|
||||
)
|
||||
"overwriting it.")
|
||||
return None
|
||||
|
||||
return (
|
||||
f"{resolved} was not read by this agent. "
|
||||
"Read the file first so you can write an informed edit."
|
||||
)
|
||||
"Read the file first so you can write an informed edit.")
|
||||
|
||||
def writes_since(self, exclude_task_id: str, since_ts: float,
|
||||
paths: Iterable[str]) -> Dict[str, List[str]]:
|
||||
@@ -242,5 +237,4 @@ __all__ = [
|
||||
"check_stale",
|
||||
"lock_path",
|
||||
"writes_since",
|
||||
"known_reads",
|
||||
]
|
||||
"known_reads"]
|
||||
|
||||
@@ -24,10 +24,7 @@ from pathlib import Path
|
||||
from agent.file_safety import get_read_block_error
|
||||
from tools.binary_extensions import has_binary_extension
|
||||
from tools.file_operations import (
|
||||
ShellFileOperations,
|
||||
normalize_read_pagination,
|
||||
normalize_search_pagination,
|
||||
)
|
||||
ShellFileOperations, normalize_read_pagination, normalize_search_pagination)
|
||||
from tools import file_state
|
||||
from agent.redact import redact_sensitive_text
|
||||
from tools.file_tools_paths import ( # noqa: F401 (re-exported)
|
||||
@@ -44,8 +41,7 @@ from tools.file_tools_paths import ( # noqa: F401 (re-exported)
|
||||
_resolve_path_for_task,
|
||||
_sentinel_free_abs_cwd,
|
||||
_terminal_env_type_for_task,
|
||||
_uses_container_paths,
|
||||
)
|
||||
_uses_container_paths)
|
||||
from tools.file_tools_write_guards import ( # noqa: F401 (re-exported)
|
||||
_PROTECTED_INSTRUCTION_BASENAMES,
|
||||
_READ_DEDUP_STATUS_MESSAGE,
|
||||
@@ -64,8 +60,7 @@ from tools.file_tools_write_guards import ( # noqa: F401 (re-exported)
|
||||
_looks_like_read_file_line_numbered_content,
|
||||
_protected_instruction_config,
|
||||
_protected_instruction_reason,
|
||||
_request_protected_instruction_approval,
|
||||
)
|
||||
_request_protected_instruction_approval)
|
||||
from tools.file_tools_read_tracking import ( # noqa: F401 (re-exported)
|
||||
_DEDUP_CAP,
|
||||
_NOT_FOUND_CAP,
|
||||
@@ -88,8 +83,7 @@ from tools.file_tools_read_tracking import ( # noqa: F401 (re-exported)
|
||||
_task_data,
|
||||
_update_read_timestamp,
|
||||
notify_other_tool_call,
|
||||
reset_file_dedup,
|
||||
)
|
||||
reset_file_dedup)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -155,14 +149,12 @@ def _apply_char_budget(result_dict: dict, content: str, offset: int, total_lines
|
||||
result_dict["hint"] = (
|
||||
f"Output truncated at the {max_chars:,}-char read budget after "
|
||||
f"{lines_kept} line(s) (showing lines {offset}-{next_offset - 1} of "
|
||||
f"{total_lines}). Use offset={next_offset} to continue."
|
||||
)
|
||||
f"{total_lines}). Use offset={next_offset} to continue.")
|
||||
if len(trimmed.split("\n", 1)[0]) >= max_chars:
|
||||
result_dict["hint"] += (
|
||||
" Note: the first line alone exceeded the budget and was "
|
||||
"clamped mid-line; its remainder is not retrievable via "
|
||||
"offset."
|
||||
)
|
||||
"offset.")
|
||||
return trimmed
|
||||
|
||||
|
||||
@@ -182,8 +174,7 @@ _BLOCKED_DEVICE_PATHS = frozenset({
|
||||
_BLOCKED_PROC_SUFFIXES = (
|
||||
"/fd/0", "/fd/1", "/fd/2", # stdio aliases
|
||||
"/environ", "/cmdline", "/maps", "/smaps", "/smaps_rollup", "/numa_maps",
|
||||
"/mem", "/auxv", "/pagemap",
|
||||
)
|
||||
"/mem", "/auxv", "/pagemap")
|
||||
|
||||
|
||||
def _file_ops_uses_host_paths(file_ops) -> bool:
|
||||
@@ -330,8 +321,7 @@ def _create_terminal_env_for_file_ops(raw_task_id: str, task_id: str):
|
||||
_resolve_task_host_cwd,
|
||||
_select_image,
|
||||
get_session_cwd,
|
||||
resolve_task_overrides,
|
||||
)
|
||||
resolve_task_overrides)
|
||||
|
||||
config = _get_env_config()
|
||||
env_type = config["env_type"]
|
||||
@@ -351,8 +341,7 @@ def _create_terminal_env_for_file_ops(raw_task_id: str, task_id: str):
|
||||
logger.info(
|
||||
"Ignoring host/relative cwd override %r for %s backend "
|
||||
"(won't exist in sandbox). Using %r instead.",
|
||||
cwd, env_type, config["cwd"],
|
||||
)
|
||||
cwd, env_type, config["cwd"])
|
||||
cwd = config["cwd"]
|
||||
logger.info("Creating new %s environment for task %s...", env_type, task_id[:8])
|
||||
terminal_env = _create_configured_env(
|
||||
@@ -376,8 +365,7 @@ def _get_file_ops(task_id: str = "default") -> ShellFileOperations:
|
||||
from tools.terminal_tool import (
|
||||
_active_environments, _env_lock, _last_activity, _start_cleanup_thread,
|
||||
_creation_locks, _creation_locks_lock, _resolve_container_task_id,
|
||||
get_session_cwd, record_session_cwd,
|
||||
)
|
||||
get_session_cwd, record_session_cwd)
|
||||
|
||||
raw_task_id = task_id or "default"
|
||||
task_id = _resolve_container_task_id(raw_task_id)
|
||||
@@ -441,8 +429,7 @@ _SPECIAL_FILE_KINDS = (
|
||||
(stat.S_ISFIFO, "a FIFO (named pipe)"),
|
||||
(stat.S_ISSOCK, "a socket"),
|
||||
(stat.S_ISCHR, "a character device"),
|
||||
(stat.S_ISBLK, "a block device"),
|
||||
)
|
||||
(stat.S_ISBLK, "a block device"))
|
||||
|
||||
|
||||
def _special_file_kind(path) -> str | None:
|
||||
@@ -478,8 +465,7 @@ def _read_extracted_document(path: str, _resolved, offset: int, limit: int, task
|
||||
MAX_DOCUMENT_BYTES,
|
||||
ExtractionError,
|
||||
extract_document_bytes,
|
||||
is_extractable_document,
|
||||
)
|
||||
is_extractable_document)
|
||||
|
||||
if not is_extractable_document(str(_resolved)):
|
||||
return None
|
||||
@@ -504,8 +490,7 @@ def _read_extracted_document(path: str, _resolved, offset: int, limit: int, task
|
||||
return tool_error(
|
||||
f"Cannot read '{path}' ({_doc_ext}): document "
|
||||
f"extraction failed — {exc}. Use terminal utilities "
|
||||
"to inspect or convert the file."
|
||||
)
|
||||
"to inspect or convert the file.")
|
||||
return None
|
||||
|
||||
lines = extracted_text.splitlines()
|
||||
@@ -517,13 +502,11 @@ def _read_extracted_document(path: str, _resolved, offset: int, limit: int, task
|
||||
"total_lines": total_lines,
|
||||
"file_size": binary.file_size,
|
||||
"truncated": total_lines > end_line,
|
||||
"extracted_document": True,
|
||||
}
|
||||
"extracted_document": True}
|
||||
if result_dict["truncated"]:
|
||||
result_dict["hint"] = (
|
||||
f"Use offset={end_line + 1} to continue reading "
|
||||
f"(showing {offset}-{min(end_line, total_lines)} of {total_lines} lines)"
|
||||
)
|
||||
f"(showing {offset}-{min(end_line, total_lines)} of {total_lines} lines)")
|
||||
max_chars = _get_max_read_chars()
|
||||
if len(result_dict["content"]) > max_chars:
|
||||
_apply_char_budget(result_dict, result_dict["content"], offset, total_lines, max_chars)
|
||||
@@ -550,8 +533,7 @@ def _dedup_stub_or_block(task_data: dict, dedup_key: tuple, path: str) -> str:
|
||||
"still current. Proceed with your task using "
|
||||
"the information you already have.",
|
||||
path=path,
|
||||
already_read=hits + 1,
|
||||
)
|
||||
already_read=hits + 1)
|
||||
|
||||
return json.dumps({
|
||||
"status": "unchanged",
|
||||
@@ -614,8 +596,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str =
|
||||
if _is_blocked_device(path, base_dir=device_base):
|
||||
return tool_error(
|
||||
f"Cannot read '{path}': this is a device file that would "
|
||||
"block or produce infinite output."
|
||||
)
|
||||
"block or produce infinite output.")
|
||||
|
||||
_resolved = _resolve_path_for_task(path, task_id)
|
||||
|
||||
@@ -629,9 +610,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str =
|
||||
f"'{path}' is {kind}, not a regular file — reading "
|
||||
"it would block indefinitely, so no read was "
|
||||
"attempted. Use terminal utilities if you need to "
|
||||
"interact with it."
|
||||
),
|
||||
})
|
||||
"interact with it.")})
|
||||
|
||||
extracted = _read_extracted_document(path, _resolved, offset, limit, task_id)
|
||||
if extracted is not None:
|
||||
@@ -642,8 +621,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str =
|
||||
if has_binary_extension(str(_resolved)):
|
||||
return tool_error(
|
||||
f"Cannot read binary file '{path}' ({_resolved.suffix.lower()}). "
|
||||
"Use vision_analyze for images, or terminal to inspect binary files."
|
||||
)
|
||||
"Use vision_analyze for images, or terminal to inspect binary files.")
|
||||
|
||||
# Hermes internal denylist: blocks prompt injection via catalog/hub
|
||||
# metadata and credential stores under HERMES_HOME. Pass the resolved
|
||||
@@ -692,8 +670,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str =
|
||||
if len(result.content or "") > max_chars:
|
||||
result.content = _apply_char_budget(
|
||||
result_dict, result.content or "", offset,
|
||||
result_dict.get("total_lines", "unknown"), max_chars,
|
||||
)
|
||||
result_dict.get("total_lines", "unknown"), max_chars)
|
||||
if result.content:
|
||||
result.content = redact_sensitive_text(result.content, file_read=True)
|
||||
result_dict["content"] = result.content
|
||||
@@ -703,8 +680,7 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str =
|
||||
result_dict.setdefault("_hint", (
|
||||
f"This file is large ({file_size:,} bytes). "
|
||||
"Consider reading only the section you need with offset and limit "
|
||||
"to keep context usage efficient."
|
||||
))
|
||||
"to keep context usage efficient."))
|
||||
|
||||
count = _record_successful_read(task_data, task_id, path, resolved_str, offset, limit,
|
||||
dedup_key, partial=(offset > 1) or bool(result_dict.get("truncated")))
|
||||
@@ -714,14 +690,12 @@ def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str =
|
||||
"The content has NOT changed. You already have this information. "
|
||||
"STOP re-reading and proceed with your task.",
|
||||
path=path,
|
||||
already_read=count,
|
||||
)
|
||||
already_read=count)
|
||||
if count >= 3:
|
||||
result_dict["_warning"] = (
|
||||
f"You have read this exact file region {count} times consecutively. "
|
||||
"The content has not changed since your last read. Use the information you already have. "
|
||||
"If you are stuck in a loop, stop reading and proceed with writing or responding."
|
||||
)
|
||||
"If you are stuck in a loop, stop reading and proceed with writing or responding.")
|
||||
return json.dumps(result_dict, ensure_ascii=False)
|
||||
except Exception as e:
|
||||
return tool_error(str(e))
|
||||
@@ -764,8 +738,7 @@ def _write_precheck_error(paths: list[str], content_paths: list[str], task_id: s
|
||||
return _first_error(
|
||||
*(lambda p=p: _check_binary_document_write(p, task_id) for p in content_paths),
|
||||
lambda: _check_protected_instruction_write(paths, task_id),
|
||||
lambda: _check_approval_required_write(paths, task_id),
|
||||
)
|
||||
lambda: _check_approval_required_write(paths, task_id))
|
||||
|
||||
|
||||
def _edit_warnings(paths: list[str], path_to_resolved: dict, task_id: str) -> list[str]:
|
||||
@@ -813,8 +786,7 @@ def write_file_tool(path: str, content: str, task_id: str = "default",
|
||||
"Refusing to write internal read_file display text as file content. "
|
||||
"Strip read_file line-number prefixes or reconstruct the intended "
|
||||
"file contents before writing."
|
||||
) if _is_internal_file_tool_content(content) else None,
|
||||
)
|
||||
) if _is_internal_file_tool_content(content) else None)
|
||||
if err:
|
||||
return tool_error(err)
|
||||
try:
|
||||
@@ -873,8 +845,7 @@ def _collect_v4a_header_paths(patch: str) -> tuple[list[str], list[str]] | str:
|
||||
f"V4A patch header contains '..' traversal: {v4a_path!r}. "
|
||||
"Use the agent's cwd-relative path (no '..') or an absolute "
|
||||
"path in '*** Update File:' / '*** Add File:' / "
|
||||
"'*** Delete File:' / '*** Move File:' headers."
|
||||
)
|
||||
"'*** Delete File:' / '*** Move File:' headers.")
|
||||
paths.append(v4a_path)
|
||||
if writes_text:
|
||||
content_paths.append(v4a_path)
|
||||
@@ -958,13 +929,11 @@ def patch_tool(mode: str = "replace", path: str = None, old_string: str = None,
|
||||
"content, (2) use a longer / more unique old_string with "
|
||||
"surrounding context lines, or (3) use write_file to "
|
||||
"replace the entire file if the targeted region is hard "
|
||||
"to anchor."
|
||||
)
|
||||
"to anchor.")
|
||||
elif "Did you mean one of these sections?" not in str(result_dict["error"]):
|
||||
result_dict["_hint"] = (
|
||||
"old_string not found. Use read_file to verify the current "
|
||||
"content, or search_files to locate the text."
|
||||
)
|
||||
"content, or search_files to locate the text.")
|
||||
return json.dumps(result_dict, ensure_ascii=False)
|
||||
except Exception as e:
|
||||
return tool_error(str(e))
|
||||
@@ -983,8 +952,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".",
|
||||
search_key = ("search", pattern, target, str(path), file_glob or "", limit, offset)
|
||||
with _read_tracker_lock:
|
||||
task_data = _read_tracker.setdefault(task_id, {
|
||||
"last_key": None, "consecutive": 0, "read_history": set(),
|
||||
})
|
||||
"last_key": None, "consecutive": 0, "read_history": set()})
|
||||
count = _bump_consecutive(task_data, search_key)
|
||||
|
||||
if count >= 4:
|
||||
@@ -993,8 +961,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".",
|
||||
"The results have NOT changed. You already have this information. "
|
||||
"STOP re-searching and proceed with your task.",
|
||||
pattern=pattern,
|
||||
already_searched=count,
|
||||
)
|
||||
already_searched=count)
|
||||
|
||||
try:
|
||||
resolved_search_path = str(_resolve_path_for_task(path, task_id))
|
||||
@@ -1016,8 +983,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".",
|
||||
|
||||
result = _get_file_ops(task_id).search(
|
||||
pattern=pattern, path=path, target=target, file_glob=file_glob,
|
||||
limit=limit, offset=offset, output_mode=output_mode, context=context
|
||||
)
|
||||
limit=limit, offset=offset, output_mode=output_mode, context=context)
|
||||
omitted = _filter_read_blocked_search_results(result, task_id)
|
||||
for m in getattr(result, "matches", None) or ():
|
||||
if getattr(m, "content", None):
|
||||
@@ -1027,8 +993,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".",
|
||||
if omitted:
|
||||
result_dict["_omitted"] = (
|
||||
f"{omitted} result(s) omitted because they target credential, "
|
||||
"token, cache, or secret-bearing environment files."
|
||||
)
|
||||
"token, cache, or secret-bearing environment files.")
|
||||
|
||||
# No early return on a cached miss — same rationale as the read path.
|
||||
_search_err = result_dict.get("error") or ""
|
||||
@@ -1038,8 +1003,7 @@ def search_tool(pattern: str, target: str = "content", path: str = ".",
|
||||
if count >= 3:
|
||||
result_dict["_warning"] = (
|
||||
f"You have run this exact search {count} times consecutively. "
|
||||
"The results have not changed. Use the information you already have."
|
||||
)
|
||||
"The results have not changed. Use the information you already have.")
|
||||
|
||||
result_json = json.dumps(result_dict, ensure_ascii=False)
|
||||
if result_dict.get("truncated"):
|
||||
|
||||
@@ -47,8 +47,7 @@ def _terminal_env_type_for_task(task_id: str = "default") -> str:
|
||||
"""Best-effort terminal backend type for path-resolution decisions."""
|
||||
try:
|
||||
from tools.terminal_tool import (
|
||||
_active_environments, _env_lock, _get_env_config, _resolve_container_task_id,
|
||||
)
|
||||
_active_environments, _env_lock, _get_env_config, _resolve_container_task_id)
|
||||
|
||||
try:
|
||||
container_key = _resolve_container_task_id(task_id)
|
||||
@@ -147,10 +146,7 @@ def _authoritative_workspace_root(task_id: str = "default") -> str | None:
|
||||
|
||||
|
||||
def _resolve_base_dir(
|
||||
task_id: str = "default",
|
||||
*,
|
||||
container_paths: bool | None = None,
|
||||
) -> Path | PurePosixPath:
|
||||
task_id: str = "default", *, container_paths: bool | None = None) -> Path | PurePosixPath:
|
||||
"""Return the ABSOLUTE base directory for resolving relative paths.
|
||||
|
||||
Uses ``_authoritative_workspace_root`` (live cwd → registered override →
|
||||
@@ -249,7 +245,6 @@ def _path_resolution_warning(filepath: str, resolved: Path, task_id: str = "defa
|
||||
f"OUTSIDE the active workspace ({str(root)!r}). The edit will land in "
|
||||
f"a different directory than the terminal's cwd. If this is not "
|
||||
f"intended (e.g. a git-worktree session writing into the main "
|
||||
f"checkout), pass an absolute path under the workspace instead."
|
||||
)
|
||||
f"checkout), pass an absolute path under the workspace instead.")
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
@@ -44,8 +44,7 @@ def _task_data(task_id: str) -> dict:
|
||||
code paths (or injected by tests) may lack the newer containers.
|
||||
"""
|
||||
task_data = _read_tracker.setdefault(task_id, {
|
||||
"last_key": None, "consecutive": 0, "read_history": set(),
|
||||
})
|
||||
"last_key": None, "consecutive": 0, "read_history": set()})
|
||||
for key in ("dedup", "dedup_hits", "read_timestamps"):
|
||||
task_data.setdefault(key, {})
|
||||
return task_data
|
||||
@@ -80,8 +79,7 @@ def _cap_read_tracker_data(task_data: dict) -> None:
|
||||
("dedup", _DEDUP_CAP),
|
||||
("dedup_hits", _DEDUP_CAP),
|
||||
("read_timestamps", _READ_TIMESTAMPS_CAP),
|
||||
("not_found", _NOT_FOUND_CAP),
|
||||
):
|
||||
("not_found", _NOT_FOUND_CAP)):
|
||||
container = task_data.get(key)
|
||||
if container is not None and len(container) > cap:
|
||||
_evict_oldest(container, cap)
|
||||
@@ -237,8 +235,7 @@ def _check_file_staleness(filepath: str, task_id: str) -> str | None:
|
||||
return (
|
||||
f"Warning: {filepath} was modified since you last read it "
|
||||
"(external edit or concurrent agent). The content you read may be "
|
||||
"stale. Consider re-reading the file to verify before writing."
|
||||
)
|
||||
"stale. Consider re-reading the file to verify before writing.")
|
||||
return None
|
||||
|
||||
|
||||
|
||||
@@ -21,8 +21,7 @@ from tools.file_tools_paths import _expand_tilde, _resolve_path_for_task
|
||||
_SENSITIVE_PATH_PREFIXES = (
|
||||
"/etc/", "/boot/", "/usr/lib/systemd/",
|
||||
"/private/etc/",
|
||||
"/private/var/db/", "/private/var/root/",
|
||||
)
|
||||
"/private/var/db/", "/private/var/root/")
|
||||
_SENSITIVE_EXACT_PATHS = {"/var/run/docker.sock", "/run/docker.sock"}
|
||||
|
||||
_hermes_config_resolved: str | None = None
|
||||
@@ -77,8 +76,7 @@ def _check_sensitive_path(filepath: str, task_id: str = "default") -> str | None
|
||||
if any(c.startswith(_SENSITIVE_PATH_PREFIXES) or c in _SENSITIVE_EXACT_PATHS for c in candidates):
|
||||
return (
|
||||
f"Refusing to write to sensitive system path: {filepath}\n"
|
||||
"Use the terminal tool with sudo if you need to modify system files."
|
||||
)
|
||||
"Use the terminal tool with sudo if you need to modify system files.")
|
||||
# approvals.mode and other security settings live in config.yaml; a
|
||||
# prompt-injected agent could silently disable exec approval by editing it.
|
||||
hermes_config = _get_hermes_config_resolved()
|
||||
@@ -86,8 +84,7 @@ def _check_sensitive_path(filepath: str, task_id: str = "default") -> str | None
|
||||
return (
|
||||
f"Refusing to write to Hermes config file: {filepath}\n"
|
||||
"Agent cannot modify security-sensitive configuration. "
|
||||
"Edit ~/.hermes/config.yaml directly or use 'hermes config' instead."
|
||||
)
|
||||
"Edit ~/.hermes/config.yaml directly or use 'hermes config' instead.")
|
||||
return None
|
||||
|
||||
|
||||
@@ -100,8 +97,7 @@ def _check_sensitive_path(filepath: str, task_id: str = "default") -> str | None
|
||||
# human approval — even under --yolo — and fail closed without a human channel.
|
||||
# Basenames match in ANY directory and case-insensitively (loaders probe variants).
|
||||
_PROTECTED_INSTRUCTION_BASENAMES = frozenset({
|
||||
"agents.md", "claude.md", "soul.md", ".cursorrules",
|
||||
})
|
||||
"agents.md", "claude.md", "soul.md", ".cursorrules"})
|
||||
|
||||
|
||||
def _protected_instruction_config() -> tuple[bool, list[str]]:
|
||||
@@ -182,15 +178,13 @@ def _request_protected_instruction_approval(reasons: list[str], task_id: str = "
|
||||
description = (
|
||||
f"Write to protected agent-instruction file(s): {targets}. "
|
||||
"These files steer future agent behavior; approval is always "
|
||||
"required (not bypassed by auto-approve)."
|
||||
)
|
||||
"required (not bypassed by auto-approve).")
|
||||
display = f"<write to {targets}>"
|
||||
blocked = (
|
||||
f"BLOCKED: write to protected agent-instruction file(s) ({targets}) "
|
||||
"{why} The user has NOT consented to this write. Do NOT retry it or "
|
||||
"attempt the same edit via another path (terminal, execute_code, "
|
||||
"etc.)."
|
||||
)
|
||||
"etc.).")
|
||||
timed_out = blocked.format(why="approval prompt timed out without a user response. Silence is not consent.")
|
||||
denied = blocked.format(why="was denied by the user.")
|
||||
|
||||
@@ -215,8 +209,7 @@ def _request_protected_instruction_approval(reasons: list[str], task_id: str = "
|
||||
"pattern_keys": ["protected_instruction_file"],
|
||||
"description": description,
|
||||
"allow_permanent": False,
|
||||
"allow_session": False,
|
||||
}
|
||||
"allow_session": False}
|
||||
decision = _approval._await_gateway_decision(session_key, notify_cb, approval_data, surface="gateway")
|
||||
if decision.get("notify_failed"):
|
||||
return blocked.format(why="requires approval but the approval request could not be delivered.")
|
||||
@@ -282,13 +275,11 @@ def _check_approval_required_write(paths: list[str], task_id: str = "default") -
|
||||
description = (
|
||||
f"Write to SSH client config file(s): {display_targets}. "
|
||||
"The SSH config can carry ProxyCommand / Match exec directives that "
|
||||
"run commands, so writes require your approval."
|
||||
)
|
||||
"run commands, so writes require your approval.")
|
||||
blocked = (
|
||||
f"BLOCKED: write to SSH config file(s) ({display_targets}) "
|
||||
"{why} Do NOT retry it via another path (terminal, execute_code) "
|
||||
"without the user's explicit consent."
|
||||
)
|
||||
"without the user's explicit consent.")
|
||||
|
||||
try:
|
||||
import tools.approval as _approval
|
||||
@@ -307,8 +298,7 @@ def _check_approval_required_write(paths: list[str], task_id: str = "default") -
|
||||
"approve in config.yaml."),
|
||||
autoapprove_log_prefix="ssh_config_write",
|
||||
fail_closed_when_no_human=True,
|
||||
no_human_block_message=blocked.format(why=_NO_HUMAN),
|
||||
)
|
||||
no_human_block_message=blocked.format(why=_NO_HUMAN))
|
||||
if result.get("approved"):
|
||||
return None
|
||||
return result.get("message") or blocked.format(why="was denied.")
|
||||
@@ -318,8 +308,7 @@ def _get_container_mirror_prefix_for_task(task_id: str = "default") -> str | Non
|
||||
"""Return the container-side Hermes mirror prefix for persistent Docker file tools."""
|
||||
try:
|
||||
from tools.terminal_tool import (
|
||||
_active_environments, _env_lock, _get_env_config, _resolve_container_task_id,
|
||||
)
|
||||
_active_environments, _env_lock, _get_env_config, _resolve_container_task_id)
|
||||
container_key = _resolve_container_task_id(task_id)
|
||||
with _env_lock:
|
||||
env = _active_environments.get(container_key) or _active_environments.get(task_id)
|
||||
@@ -372,8 +361,7 @@ def _check_binary_document_write(filepath: str, task_id: str = "default") -> str
|
||||
"corrupt the file (read_file showed you EXTRACTED text, not the real "
|
||||
"bytes). Use the docx/xlsx/powerpoint skills or a library like "
|
||||
"python-docx/openpyxl/python-pptx via the terminal to create or edit "
|
||||
"this document."
|
||||
)
|
||||
"this document.")
|
||||
if is_pdf_path(filepath):
|
||||
try:
|
||||
resolved = Path(_resolve_path_for_task(filepath, task_id))
|
||||
@@ -386,8 +374,7 @@ def _check_binary_document_write(filepath: str, task_id: str = "default") -> str
|
||||
"read_file showed you EXTRACTED text, not the real bytes — writing "
|
||||
"text back would destroy the document. Use the pdf skill or a PDF "
|
||||
"library via the terminal to modify it. (Creating a NEW .pdf file "
|
||||
"is allowed.)"
|
||||
)
|
||||
"is allowed.)")
|
||||
except OSError:
|
||||
pass
|
||||
return None
|
||||
@@ -399,8 +386,7 @@ def _check_binary_document_write(filepath: str, task_id: str = "default") -> str
|
||||
_READ_DEDUP_STATUS_MESSAGE = (
|
||||
"File unchanged since last read. The content from "
|
||||
"the earlier read_file result in this conversation is "
|
||||
"still current — refer to that instead of re-reading."
|
||||
)
|
||||
"still current — refer to that instead of re-reading.")
|
||||
|
||||
|
||||
def _is_internal_file_status_text(content: str) -> bool:
|
||||
|
||||
@@ -18,8 +18,7 @@ Span = tuple[int, int]
|
||||
IDENTICAL_STRINGS_ERROR = (
|
||||
"No edit was applied because old_string and new_string are identical. "
|
||||
"Provide the existing text to replace in old_string and the changed "
|
||||
"replacement text in new_string."
|
||||
)
|
||||
"replacement text in new_string.")
|
||||
|
||||
UNICODE_MAP = {
|
||||
"\u201c": '"', "\u201d": '"', # smart double quotes
|
||||
@@ -32,8 +31,7 @@ UNICODE_MAP = {
|
||||
"\u2000": " ", "\u2001": " ", "\u2002": " ", "\u2003": " ",
|
||||
"\u2004": " ", "\u2005": " ", "\u2006": " ", "\u2007": " ",
|
||||
"\u2008": " ", "\u2009": " ", "\u200a": " ", "\u202f": " ",
|
||||
"\u205f": " ", "\u3000": " ",
|
||||
}
|
||||
"\u205f": " ", "\u3000": " "}
|
||||
|
||||
|
||||
def _unicode_normalize(text: str) -> str:
|
||||
@@ -58,8 +56,7 @@ def _window_spans(content: str, content_lines: list[str], n: int,
|
||||
"""Spans of every ``n``-line window starting at ``i`` for which ``accept(i)``."""
|
||||
return [
|
||||
_calculate_line_positions(content_lines, i, i + n, len(content))
|
||||
for i in range(len(content_lines) - n + 1) if accept(i)
|
||||
]
|
||||
for i in range(len(content_lines) - n + 1) if accept(i)]
|
||||
|
||||
|
||||
def _match_transformed_lines(content: str, pattern: str,
|
||||
@@ -257,8 +254,7 @@ def _strategy_block_anchor(content: str, pattern: str) -> list[Span]:
|
||||
potential_matches = {
|
||||
i for i in range(len(norm_content_lines) - n + 1)
|
||||
if norm_content_lines[i].strip() == first_line
|
||||
and norm_content_lines[i + n - 1].strip() == last_line
|
||||
}
|
||||
and norm_content_lines[i + n - 1].strip() == last_line}
|
||||
# Looser thresholds (0.10/0.30) matched unrelated blocks; these are the safe floor.
|
||||
threshold = 0.50 if len(potential_matches) == 1 else 0.70
|
||||
pattern_middle = '\n'.join(pattern_lines[1:-1])
|
||||
@@ -300,8 +296,7 @@ def _strategy_context_aware(content: str, pattern: str) -> list[Span]:
|
||||
return False
|
||||
return all(
|
||||
not p_line.strip() or _sim(p_line.strip(), c_line.strip()) >= 0.80
|
||||
for p_line, c_line in zip(pattern_lines, block_lines)
|
||||
)
|
||||
for p_line, c_line in zip(pattern_lines, block_lines))
|
||||
|
||||
return _window_spans(content, content_lines, n, accept)
|
||||
|
||||
@@ -316,8 +311,7 @@ STRATEGIES: list[tuple[str, Callable[[str, str], list[Span]]]] = [
|
||||
("trimmed_boundary", _strategy_trimmed_boundary),
|
||||
("unicode_normalized", _strategy_unicode_normalized),
|
||||
("block_anchor", _strategy_block_anchor),
|
||||
("context_aware", _strategy_context_aware),
|
||||
]
|
||||
("context_aware", _strategy_context_aware)]
|
||||
|
||||
# Matches from these only *approximately* resemble old_string — fine for one
|
||||
# unique replacement, never safe under replace_all.
|
||||
@@ -386,15 +380,13 @@ def fuzzy_find_and_replace(content: str, old_string: str, new_string: str,
|
||||
return content, 0, None, (
|
||||
f"Found {len(matches)} matches for old_string. "
|
||||
f"Provide more context to make it unique, or use replace_all=True. "
|
||||
f"Matches:\n{locations}"
|
||||
)
|
||||
f"Matches:\n{locations}")
|
||||
if replace_all and len(matches) > 1 and strategy_name in SIMILARITY_STRATEGIES:
|
||||
return content, 0, None, (
|
||||
f"Found {len(matches)} approximate matches via the "
|
||||
f"'{strategy_name}' strategy; replace_all only applies to exact "
|
||||
f"matches. Provide the precise text (whitespace included) so an "
|
||||
f"exact/line-trimmed match can be made."
|
||||
)
|
||||
f"exact/line-trimmed match can be made.")
|
||||
|
||||
# Non-exact matches came through some normalization, so new_string may
|
||||
# carry serialization drift the file doesn't have.
|
||||
@@ -408,8 +400,7 @@ def fuzzy_find_and_replace(content: str, old_string: str, new_string: str,
|
||||
effective_new = _preserve_unicode_in_replacement(content, matches, old_string, effective_new)
|
||||
new_content = _apply_replacements(
|
||||
content, matches, effective_new,
|
||||
old_string=old_string if strategy_name != "exact" else None,
|
||||
)
|
||||
old_string=old_string if strategy_name != "exact" else None)
|
||||
return new_content, len(matches), strategy_name, None
|
||||
|
||||
return content, 0, None, "Could not find a match for old_string in the file"
|
||||
@@ -441,8 +432,7 @@ def _detect_escape_drift(content: str, matches: list[Span],
|
||||
f"serialization artifact where an apostrophe or quote got "
|
||||
f"prefixed with a spurious backslash. Re-read the file with "
|
||||
f"read_file and pass old_string/new_string without "
|
||||
f"backslash-escaping {plain!r} characters."
|
||||
)
|
||||
f"backslash-escaping {plain!r} characters.")
|
||||
return _detect_backslash_doubling(matched_regions, old_string, new_string)
|
||||
|
||||
|
||||
@@ -476,8 +466,7 @@ def _detect_backslash_doubling(matched_regions: str, old_string: str,
|
||||
"were JSON-escaped one extra time; applying new_string verbatim would "
|
||||
"double every backslash in the file. Re-read the file with read_file "
|
||||
"and resend old_string/new_string with the backslash counts exactly "
|
||||
"as they appear in the file."
|
||||
)
|
||||
"as they appear in the file.")
|
||||
|
||||
|
||||
def _maybe_unescape_new_string(new_string: str, content: str, matches: list[Span]) -> str:
|
||||
@@ -626,8 +615,7 @@ def find_closest_lines(old_string: str, content: str, context_lines: int = 2, ma
|
||||
continue
|
||||
seen_ranges.add((start, end))
|
||||
parts.append("\n".join(
|
||||
f"{start + j + 1:4d}| {content_lines[start + j]}" for j in range(end - start)
|
||||
))
|
||||
f"{start + j + 1:4d}| {content_lines[start + j]}" for j in range(end - start)))
|
||||
if not parts:
|
||||
return ""
|
||||
result = "\n---\n".join(parts)
|
||||
@@ -640,8 +628,7 @@ def find_closest_lines(old_string: str, content: str, context_lines: int = 2, ma
|
||||
"\n\nWhitespace difference detected (→ = tab, · = space):\n"
|
||||
f" file has: {_visualize_whitespace(best_line)}\n"
|
||||
f" you sent: {_visualize_whitespace(old_lines[0])}\n"
|
||||
"Use the exact whitespace shown in 'file has'."
|
||||
)
|
||||
"Use the exact whitespace shown in 'file has'.")
|
||||
return result
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user