Files
hermes-agent/tools/binary_extensions.py
Teknium 6d51c831eb fix(file_tools): refuse plain-text writes that corrupt binary documents
Port from nearai/ironclaw#7109: read_file auto-extracts .docx/.xlsx/.pptx
(and PDF via anydoc) to readable text, so a model plausibly believes it
holds the file's contents and writes the edited text back with
write_file/patch — silently destroying the document container. Proven
live on main: write_file over a valid .docx left a non-zip corpse, and a
text write over an existing .pdf clobbered the %PDF header.

- tools/binary_extensions.py: OPAQUE_DOCUMENT_EXTENSIONS +
  has_opaque_document_extension() + is_pdf_path() (pure string checks)
- tools/file_tools.py: _check_binary_document_write() — opaque container
  formats (doc/docx/xls/xlsx/ppt/pptx/odt/ods/odp) always rejected; .pdf
  rejected only when overwriting an existing regular file (new-PDF
  creation stays allowed, matching the upstream split guard). Wired into
  write_file_tool and patch_tool (replace + V4A Update/Add headers;
  Delete/Move skip the guard since they write no text).
- tests/tools/test_binary_document_write_guard.py: guard unit tests +
  end-to-end write_file/patch coverage incl. bytes-untouched assertions.
2026-08-15 02:51:59 +05:30

72 lines
2.6 KiB
Python

"""Binary file extensions to skip for text-based operations.
These files can't be meaningfully compared as text and are often large.
Ported from free-code src/constants/files.ts.
"""
BINARY_EXTENSIONS = frozenset({
# Images
".png", ".jpg", ".jpeg", ".gif", ".bmp", ".ico", ".webp", ".tiff", ".tif",
# Videos
".mp4", ".mov", ".avi", ".mkv", ".webm", ".wmv", ".flv", ".m4v", ".mpeg", ".mpg",
# Audio
".mp3", ".wav", ".ogg", ".flac", ".aac", ".m4a", ".wma", ".aiff", ".opus",
# Archives
".zip", ".tar", ".gz", ".bz2", ".7z", ".rar", ".xz", ".z", ".tgz", ".iso",
# Executables/binaries
".exe", ".dll", ".so", ".dylib", ".bin", ".o", ".a", ".obj", ".lib",
".app", ".msi", ".deb", ".rpm",
# Documents (exclude .pdf — text-based, agents may want to inspect)
".doc", ".docx", ".xls", ".xlsx", ".ppt", ".pptx",
".odt", ".ods", ".odp",
# Fonts
".ttf", ".otf", ".woff", ".woff2", ".eot",
# Bytecode / VM artifacts
".pyc", ".pyo", ".class", ".jar", ".war", ".ear", ".node", ".wasm", ".rlib",
# Database files
".sqlite", ".sqlite3", ".db", ".mdb", ".idx",
# Design / 3D
".psd", ".ai", ".eps", ".sketch", ".fig", ".xd", ".blend", ".3ds", ".max",
# Flash
".swf", ".fla",
# Lock/profiling data
".lockb", ".dat", ".data",
})
def has_binary_extension(path: str) -> bool:
"""Check if a file path has a binary extension. Pure string check, no I/O."""
dot = path.rfind(".")
if dot == -1:
return False
return path[dot:].lower() in BINARY_EXTENSIONS
# Container document formats (OOXML zip / OLE compound / ODF zip) that a
# plain-text write can NEVER produce validly. read_file auto-extracts these
# to readable text, so a model that "read" report.docx and then writes the
# edited text back via write_file/patch silently destroys the document.
# PDF is intentionally NOT here: raw PDF syntax is text-authorable, so
# new-file creation is legitimate — only overwrites are dangerous (handled
# separately by the write guard).
OPAQUE_DOCUMENT_EXTENSIONS = frozenset({
".doc", ".docx", ".xls", ".xlsx", ".ppt", ".pptx",
".odt", ".ods", ".odp",
})
def has_opaque_document_extension(path: str) -> bool:
"""True when the path names an opaque container document (.docx etc.).
Pure string check, no I/O.
"""
dot = path.rfind(".")
if dot == -1:
return False
return path[dot:].lower() in OPAQUE_DOCUMENT_EXTENSIONS
def is_pdf_path(path: str) -> bool:
"""True when the path has a .pdf extension. Pure string check, no I/O."""
return path.lower().endswith(".pdf")