Port from nearai/ironclaw#7109: read_file auto-extracts .docx/.xlsx/.pptx (and PDF via anydoc) to readable text, so a model plausibly believes it holds the file's contents and writes the edited text back with write_file/patch — silently destroying the document container. Proven live on main: write_file over a valid .docx left a non-zip corpse, and a text write over an existing .pdf clobbered the %PDF header. - tools/binary_extensions.py: OPAQUE_DOCUMENT_EXTENSIONS + has_opaque_document_extension() + is_pdf_path() (pure string checks) - tools/file_tools.py: _check_binary_document_write() — opaque container formats (doc/docx/xls/xlsx/ppt/pptx/odt/ods/odp) always rejected; .pdf rejected only when overwriting an existing regular file (new-PDF creation stays allowed, matching the upstream split guard). Wired into write_file_tool and patch_tool (replace + V4A Update/Add headers; Delete/Move skip the guard since they write no text). - tests/tools/test_binary_document_write_guard.py: guard unit tests + end-to-end write_file/patch coverage incl. bytes-untouched assertions.
72 lines
2.6 KiB
Python
72 lines
2.6 KiB
Python
"""Binary file extensions to skip for text-based operations.
|
|
|
|
These files can't be meaningfully compared as text and are often large.
|
|
Ported from free-code src/constants/files.ts.
|
|
"""
|
|
|
|
BINARY_EXTENSIONS = frozenset({
|
|
# Images
|
|
".png", ".jpg", ".jpeg", ".gif", ".bmp", ".ico", ".webp", ".tiff", ".tif",
|
|
# Videos
|
|
".mp4", ".mov", ".avi", ".mkv", ".webm", ".wmv", ".flv", ".m4v", ".mpeg", ".mpg",
|
|
# Audio
|
|
".mp3", ".wav", ".ogg", ".flac", ".aac", ".m4a", ".wma", ".aiff", ".opus",
|
|
# Archives
|
|
".zip", ".tar", ".gz", ".bz2", ".7z", ".rar", ".xz", ".z", ".tgz", ".iso",
|
|
# Executables/binaries
|
|
".exe", ".dll", ".so", ".dylib", ".bin", ".o", ".a", ".obj", ".lib",
|
|
".app", ".msi", ".deb", ".rpm",
|
|
# Documents (exclude .pdf — text-based, agents may want to inspect)
|
|
".doc", ".docx", ".xls", ".xlsx", ".ppt", ".pptx",
|
|
".odt", ".ods", ".odp",
|
|
# Fonts
|
|
".ttf", ".otf", ".woff", ".woff2", ".eot",
|
|
# Bytecode / VM artifacts
|
|
".pyc", ".pyo", ".class", ".jar", ".war", ".ear", ".node", ".wasm", ".rlib",
|
|
# Database files
|
|
".sqlite", ".sqlite3", ".db", ".mdb", ".idx",
|
|
# Design / 3D
|
|
".psd", ".ai", ".eps", ".sketch", ".fig", ".xd", ".blend", ".3ds", ".max",
|
|
# Flash
|
|
".swf", ".fla",
|
|
# Lock/profiling data
|
|
".lockb", ".dat", ".data",
|
|
})
|
|
|
|
|
|
def has_binary_extension(path: str) -> bool:
|
|
"""Check if a file path has a binary extension. Pure string check, no I/O."""
|
|
dot = path.rfind(".")
|
|
if dot == -1:
|
|
return False
|
|
return path[dot:].lower() in BINARY_EXTENSIONS
|
|
|
|
|
|
# Container document formats (OOXML zip / OLE compound / ODF zip) that a
|
|
# plain-text write can NEVER produce validly. read_file auto-extracts these
|
|
# to readable text, so a model that "read" report.docx and then writes the
|
|
# edited text back via write_file/patch silently destroys the document.
|
|
# PDF is intentionally NOT here: raw PDF syntax is text-authorable, so
|
|
# new-file creation is legitimate — only overwrites are dangerous (handled
|
|
# separately by the write guard).
|
|
OPAQUE_DOCUMENT_EXTENSIONS = frozenset({
|
|
".doc", ".docx", ".xls", ".xlsx", ".ppt", ".pptx",
|
|
".odt", ".ods", ".odp",
|
|
})
|
|
|
|
|
|
def has_opaque_document_extension(path: str) -> bool:
|
|
"""True when the path names an opaque container document (.docx etc.).
|
|
|
|
Pure string check, no I/O.
|
|
"""
|
|
dot = path.rfind(".")
|
|
if dot == -1:
|
|
return False
|
|
return path[dot:].lower() in OPAQUE_DOCUMENT_EXTENSIONS
|
|
|
|
|
|
def is_pdf_path(path: str) -> bool:
|
|
"""True when the path has a .pdf extension. Pure string check, no I/O."""
|
|
return path.lower().endswith(".pdf")
|