Files
hermes-agent/tools/read_extract.py

763 lines
28 KiB
Python

"""Stdlib document-to-text extraction for ``read_file``.
Supports Jupyter notebooks, DOCX, and XLSX without hard dependencies. When the
optional ``firecrawl-anydoc`` package is installed (imports as ``anydoc``),
coverage widens to legacy Office (.doc/.ppt/.xls), OpenDocument, RTF, EPUB, and
PDF via its Rust core. The stdlib extractors stay authoritative for their three
formats so behavior is identical whether or not anydoc is present. Malformed
documents raise :class:`ExtractionError`; callers then fall back to normal
text/binary handling.
"""
from __future__ import annotations
import contextlib
import importlib
import json
import os
import posixpath
import re
import shutil
import subprocess
import tempfile
import threading
import time
import zipfile
from pathlib import Path
from typing import Any, Callable, Iterator, Optional
from xml.etree import ElementTree as ET
__all__ = [
"EXTRACTABLE_EXTENSIONS",
"ExtractionError",
"extract_document_bytes",
"extract_document_text",
"is_extractable_document",
]
EXTRACTABLE_EXTENSIONS = frozenset({".ipynb", ".docx", ".xlsx"})
# Formats handled only when the optional anydoc converter is installed.
ANYDOC_EXTENSIONS = frozenset({
".doc", ".docm",
".ppt", ".pps", ".pot", ".pptx", ".pptm", ".ppsx", ".ppsm",
".xls", ".xlsm", ".xlsb",
".odt", ".ods", ".odp",
".rtf", ".epub", ".pdf",
})
# anydoc loads the whole file through its Rust core with no streaming, and the
# read_file char budget only applies after conversion — cap the input size.
MAX_ANYDOC_BYTES = 50 * 1024 * 1024
MAX_DOCUMENT_BYTES = 50 * 1024 * 1024
_MAX_XLSX_ROWS_PER_SHEET = 5000
_MAX_XLSX_COLS = 256
_NS_W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
_NS_S = "http://schemas.openxmlformats.org/spreadsheetml/2006/main"
_NS_REL = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
_NS_PKG_REL = "http://schemas.openxmlformats.org/package/2006/relationships"
class ExtractionError(Exception):
"""Raised when a supported-looking document cannot be rendered as text."""
def _extension(path: str) -> str:
ext = Path(path).suffix.lower()
if ext in EXTRACTABLE_EXTENSIONS or (ext in ANYDOC_EXTENSIONS and _anydoc() is not None):
return ext
return ""
_ANYDOC_UNSET = object()
_anydoc_module: Any = _ANYDOC_UNSET
_anydoc_lock = threading.Lock()
# After a failed load, wait this long before retrying: the attempt can shell out
# to pip, so retrying every call would hammer the network where install can't succeed.
ANYDOC_RETRY_SECONDS = 300.0
_anydoc_failed_at: Optional[float] = None
def _anydoc() -> Optional[Any]:
"""Lazily import the optional anydoc converter; None when unavailable.
A failed load is retried after :data:`ANYDOC_RETRY_SECONDS` rather than
disabling extraction for the rest of the process, so one transient failure
(network blip, pip race) does not stick in long-lived workers.
"""
global _anydoc_module, _anydoc_failed_at
if _anydoc_module is not _ANYDOC_UNSET:
return _anydoc_module
with _anydoc_lock:
if _anydoc_module is not _ANYDOC_UNSET:
return _anydoc_module
if (
_anydoc_failed_at is not None
and time.monotonic() - _anydoc_failed_at < ANYDOC_RETRY_SECONDS
):
return None
try:
from tools.lazy_deps import ensure as _lazy_ensure
# prompt=False: read_file must never block on an install prompt.
_lazy_ensure("tool.doc_extract", prompt=False)
except Exception:
_anydoc_failed_at = time.monotonic()
return None
try:
_anydoc_module = importlib.import_module("anydoc")
except Exception: # ImportError or a broken native binding
_anydoc_failed_at = time.monotonic()
return None
_anydoc_failed_at = None
return _anydoc_module # type: ignore[return-value]
def is_extractable_document(path: str) -> bool:
return bool(_extension(path))
def _unsupported(path: str) -> ExtractionError:
return ExtractionError(f"Unsupported document type: {path!r}")
def _check_size(size: int, limit: int) -> None:
if size > limit:
raise ExtractionError(f"Document too large to convert ({size:,} bytes, limit is {limit:,})")
@contextlib.contextmanager
def _temp_copy(data: bytes, suffix: str) -> Iterator[str]:
"""Materialize backend bytes in a private host temp file; removed even when parsing fails."""
temp_path = ""
try:
with tempfile.NamedTemporaryFile(suffix=suffix, delete=False) as fh:
fh.write(data)
temp_path = fh.name
yield temp_path
finally:
if temp_path:
try:
os.unlink(temp_path)
except OSError:
pass
def extract_document_text(path: str) -> str:
ext = _extension(path)
extractor = _STDLIB_EXTRACTORS.get(ext)
if extractor is not None:
return extractor(path)
if ext in ANYDOC_EXTENSIONS:
return _extract_anydoc(path)
raise _unsupported(path)
def extract_document_bytes(data: bytes, path: str) -> str:
"""Extract a document already fetched across a file backend boundary."""
_check_size(len(data), MAX_DOCUMENT_BYTES)
ext = _extension(path)
if ext in ANYDOC_EXTENSIONS:
return _extract_anydoc_bytes(data, path)
if ext not in EXTRACTABLE_EXTENSIONS:
raise _unsupported(path)
# The stdlib extractors are path-oriented.
with _temp_copy(data, ext) as temp_path:
return extract_document_text(temp_path)
def _anydoc_missing_error(path: str) -> str:
"""Teaching error for anydoc-gated formats when the converter is absent.
The schema deliberately omits these formats and this caveat; the explanation
(and the fix) is paid for only by sessions that actually hit one.
"""
return (
f"Cannot convert {path!r}: this format needs the optional anydoc "
"converter, which is not installed (install blocked or first "
"attempt failed; retried every 5 minutes). Fix: `pip install "
"firecrawl-anydoc` in Hermes's environment, or convert the file "
"yourself via terminal (e.g. libreoffice --headless --convert-to "
"txt)."
)
def _hosted_ocr_config() -> tuple:
"""Resolve hosted-OCR settings: (enabled, api_key, api_url). Never raises.
Maintainer decision: the ONLY route is a direct ``FIRECRAWL_API_KEY``
(anydoc defaults api_url to https://api.firecrawl.dev); the Nous managed
gateway is NOT used — its Parse proxy live-probed broken while scrape/search
worked (revisit when it grows Parse support). ``file_tools.hosted_ocr:
false`` disables even with a key; true/unset → enabled iff the key is
present. Env probe only, no network at schema-build time.
"""
api_key = os.environ.get("FIRECRAWL_API_KEY") or None
enabled = api_key is not None
try:
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
section = cfg.get("file_tools") if isinstance(cfg, dict) else None
if isinstance(section, dict) and section.get("hosted_ocr") is False:
enabled = False
except Exception: # noqa: BLE001
pass
return enabled, api_key, None
def hosted_ocr_available() -> bool:
"""Public probe for read_file's schema line: is hosted OCR unlocked?
Same single gate as :func:`_hosted_ocr_config`. A key that fails at
conversion time lands in the NEEDS-OCR warning instead.
"""
return _hosted_ocr_config()[0]
def _needs_ocr_warning(path: str, pages, hosted_error: str = "") -> str:
"""Result text when anydoc raises NeedsOcrError and hosted OCR is off/failed.
Hints at CHECKING for an OCR skill (never names one — none is guaranteed to
exist) and never advertises the hosted_ocr config knob.
"""
page_list = ", ".join(str(p) for p in pages) if pages else "unknown"
msg = (
f"[NEEDS OCR: pages {page_list} of this PDF are scanned images "
"with no text layer — their content is MISSING below. "
)
if hosted_error:
msg += f"Hosted OCR was attempted and failed ({hosted_error}). "
msg += (
"If the missing pages matter: render just those pages with "
f"`pdftoppm -jpeg -r 150 -f <first> -l <last> '{path}' /tmp/page` "
"and inspect via vision_analyze, or check whether an OCR skill is "
"available (skills_list)."
)
return msg + "]\n"
def _finalize_anydoc_text(text: Any, path: str, pdf_note: Callable[[], str]) -> str:
"""Normalize converter output and, for PDFs, PREPEND the coverage note.
Prepended because read_file paginates the extraction: a footer on a long
document would sit on a page the model may never fetch. The note covers
PARTIAL gaps (text layer plus some scanned pages) that convert without
raising NeedsOcrError.
"""
if not isinstance(text, str) or not text.strip():
raise ExtractionError("Document contains no extractable text")
text = text.rstrip("\n") + "\n"
if Path(path).suffix.lower() == ".pdf":
note = pdf_note()
if note:
text = note + text
return text
def _ocr_scanned_pdf(mod: Any, path: str, exc: BaseException) -> str:
"""Typed scanned-pages signal (anydoc >= 0.2): try hosted OCR when a Firecrawl route exists, else teach recovery."""
pages = list(getattr(exc, "pages", []) or [])
enabled, api_key, api_url = _hosted_ocr_config()
hosted_error = ""
if enabled:
try:
kwargs = {"ocr": "hosted"}
if api_key:
kwargs["api_key"] = api_key
if api_url:
kwargs["api_url"] = api_url
return mod.to_markdown(path, **kwargs).rstrip("\n") + "\n"
except Exception as hosted_exc: # noqa: BLE001
hosted_error = f"{type(hosted_exc).__name__}: {hosted_exc}"
# No route / disabled / hosted failed: whole doc is scans — nothing to
# extract, so the warning IS the result.
return _needs_ocr_warning(path, pages, hosted_error)
def _require_anydoc(path: str) -> Any:
mod = _anydoc()
if mod is None:
raise ExtractionError(_anydoc_missing_error(path))
return mod
def _extract_anydoc(path: str) -> str:
mod = _require_anydoc(path)
try:
size = os.path.getsize(path)
except OSError as exc:
raise ExtractionError(str(exc)) from exc
_check_size(size, MAX_ANYDOC_BYTES)
try:
text = mod.to_markdown(path)
except OSError as exc:
raise ExtractionError(str(exc)) from exc
except Exception as exc:
needs_ocr = getattr(mod, "NeedsOcrError", None)
if needs_ocr is not None and isinstance(exc, needs_ocr):
return _ocr_scanned_pdf(mod, path, exc)
# anydoc raises one ConvertError subclass per failure mode (Unsupported,
# Malformed, Encrypted, ResourceLimit, MissingPart); all mean "no
# meaningful text", so read_file falls back to path/binary handling.
raise ExtractionError(f"{type(exc).__name__}: {exc}") from exc
return _finalize_anydoc_text(text, path, lambda: _pdf_coverage_note(path))
def _extract_anydoc_bytes(data: bytes, path: str) -> str:
mod = _require_anydoc(path)
_check_size(len(data), MAX_ANYDOC_BYTES)
try:
text = mod.to_markdown_bytes(data)
except Exception as exc:
raise ExtractionError(f"{type(exc).__name__}: {exc}") from exc
return _finalize_anydoc_text(text, path, lambda: _pdf_coverage_note_from_bytes(data, path))
# ── Scanned-PDF coverage detection ──────────────────────────────────
# Text-layer extractors return nothing for scanned pages and emit no
# placeholders, so a mostly-scanned PDF converts "successfully" into headers
# with empty bodies — silent data loss the model cannot detect. Count per-page
# text via pdftotext (form-feed separators) and warn when many pages are empty.
# A page with fewer extracted characters than this is considered empty.
PDF_EMPTY_PAGE_CHARS = 20
# Warn when empty pages reach both MIN_EMPTY and MIN_RATIO, or ABSOLUTE_EMPTY alone.
PDF_COVERAGE_MIN_EMPTY = 2
PDF_COVERAGE_MIN_RATIO = 0.2
PDF_COVERAGE_ABSOLUTE_EMPTY = 10
PDF_PAGE_SCAN_TIMEOUT = 20.0
# Cap the per-gap breakdown so alternating text/scan pages can't balloon the warning.
PDF_GAP_MAP_MAX_ENTRIES = 20
_GAP_CONTEXT_CHARS = 60
def _pdf_page_texts(path: str) -> Optional[list[str]]:
"""Per-page extracted text, or None when undeterminable."""
if shutil.which("pdftotext") is None:
return None
try:
proc = subprocess.run(
["pdftotext", path, "-"],
capture_output=True,
timeout=PDF_PAGE_SCAN_TIMEOUT,
)
except (OSError, subprocess.SubprocessError):
return None
if proc.returncode != 0:
return None
pages = proc.stdout.decode("utf-8", errors="replace").split("\f")
if pages and not pages[-1].strip():
pages.pop() # trailing form-feed artifact
return pages or None
def _group_ranges(pages: list[int]) -> list[list[int]]:
"""Group sorted 1-based page numbers into [start, end] runs."""
ranges: list[list[int]] = []
for p in pages:
if ranges and p == ranges[-1][1] + 1:
ranges[-1][1] = p
else:
ranges.append([p, p])
return ranges
def _gap_map(counts: list[int], texts: list[str], empty: list[int]) -> str:
"""Per-gap breakdown, each empty range labeled with the last text seen before
it (usually a section header), so the agent can pick WHICH gaps to OCR."""
ranges = _group_ranges(empty)
lines: list[str] = []
for a, b in ranges[:PDF_GAP_MAP_MAX_ENTRIES]:
label = ""
for prev in range(a - 2, -1, -1): # nearest preceding page with text
if counts[prev] >= PDF_EMPTY_PAGE_CHARS:
snippet = " ".join(texts[prev].split())[:_GAP_CONTEXT_CHARS]
label = f' — after "{snippet}" (p{prev + 1})'
break
span = f"page {a}" if a == b else f"pages {a}-{b}"
n = b - a + 1
lines.append(f" {span} ({n} page{'s' if n != 1 else ''}){label}")
if len(ranges) > PDF_GAP_MAP_MAX_ENTRIES:
rest = ranges[PDF_GAP_MAP_MAX_ENTRIES:]
rest_pages = sum(b - a + 1 for a, b in rest)
lines.append(f" … {len(rest)} more gaps ({rest_pages} pages)")
return "\n".join(lines)
def _pdf_coverage_note(path: str, display_path: Optional[str] = None) -> str:
"""Warning header when many PDF pages produced no text, else ''.
``path`` is scanned with pdftotext (may be a host temp file); ``display_path``
is what the recovery command shows — the path the agent's terminal can see.
"""
texts = _pdf_page_texts(path)
if not texts or len(texts) < 2:
return ""
counts = [len(page.strip()) for page in texts]
empty = [i + 1 for i, n in enumerate(counts) if n < PDF_EMPTY_PAGE_CHARS]
total = len(counts)
if len(empty) < PDF_COVERAGE_MIN_EMPTY:
return ""
if (
len(empty) / total < PDF_COVERAGE_MIN_RATIO
and len(empty) < PDF_COVERAGE_ABSOLUTE_EMPTY
):
return ""
shown = display_path or path
return (
"[EXTRACTION COVERAGE WARNING: "
f"{len(empty)} of {total} pages in this PDF yielded no text. "
"Those pages are likely scanned images (or blank) — their content "
"is MISSING from the extracted text below, even where section "
"headers appear with empty bodies. Unreadable gaps, each labeled "
"with the last text extracted before it:\n"
f"{_gap_map(counts, texts, empty)}\n"
"Decide which gaps you actually need — do NOT OCR or render "
"everything. For the gaps that matter, render just that range with "
f"`pdftoppm -jpeg -r 150 -f <first> -l <last> '{shown}' /tmp/page` "
"and inspect each image with the vision_analyze tool, or use the "
"ocr-and-documents skill (marker-pdf) for bulk OCR of large "
"ranges.]\n"
)
def _pdf_coverage_note_from_bytes(data: bytes, display_path: str) -> str:
"""Coverage note for backend-transferred PDF bytes.
pdftotext is path-oriented, so scan a host temp copy; the recovery command
still names ``display_path`` — the path the agent's terminal backend can see.
"""
try:
with _temp_copy(data, ".pdf") as temp_path:
return _pdf_coverage_note(temp_path, display_path=display_path)
except OSError:
return ""
def _source_text(source) -> str:
if isinstance(source, str):
return source
if isinstance(source, list):
return "".join(item for item in source if isinstance(item, str))
return ""
def _human_size(n_bytes: int) -> str:
return f"{round(n_bytes / 1024)} KB" if n_bytes >= 1024 else f"{n_bytes} B"
def _base64_bytes(payload: str) -> int:
"""Approximate decoded size of a base64 payload (whitespace ignored)."""
clean = re.sub(r"[^0-9+/=A-Za-z]", "", payload)
padding = min(2, len(clean) - len(clean.rstrip("=")))
return max(0, (len(clean) * 3) // 4 - padding)
def _clean_stream_text(text: str) -> str:
"""Strip ANSI escapes and collapse ``\\r`` progress-bar rewrites.
Jupyter renders only the final frame of a ``\\r``-redrawn line (tqdm), so
keep the text after the last ``\\r`` of each line.
"""
from tools.ansi_strip import strip_ansi
cleaned = strip_ansi(text).replace("\r\n", "\n")
lines = []
for line in cleaned.split("\n"):
frames = [frame for frame in line.split("\r") if frame]
lines.append(frames[-1] if frames else "")
return "\n".join(lines)
# Notebook outputs longer than this are tail-truncated per output block so a
# single runaway training log cannot flood the extracted text.
_MAX_OUTPUT_CHARS = 20_000
# nbformat v3 stores mime data flat on the output dict under these keys.
_V3_MIME_KEYS = (("png", "image/png"), ("jpeg", "image/jpeg"), ("svg", "image/svg+xml"), ("html", "text/html"))
def _notebook_output_text(output: Any) -> str:
"""Render one notebook output as compact text.
Keeps stream text, tracebacks, and textual results; replaces token-heavy
payloads (base64 images, HTML, widget state) with short sized placeholders.
Handles nbformat v4 and legacy v3 (``pyout``/``pyerr``) shapes.
"""
if not isinstance(output, dict):
return ""
otype = output.get("output_type")
if otype == "stream":
body = _clean_stream_text(_source_text(output.get("text", "")))
return body if body.strip() else ""
if otype in {"error", "pyerr"}:
traceback = output.get("traceback")
tb_text = ""
if isinstance(traceback, list):
tb_text = _clean_stream_text(
"\n".join(line for line in traceback if isinstance(line, str))
)
header = f"Error: {output.get('ename', '')}: {output.get('evalue', '')}".rstrip(": ")
return f"{header}\n{tb_text}".rstrip()
if otype in {"execute_result", "display_data", "pyout"}:
data = output.get("data")
if not isinstance(data, dict):
data = {}
if isinstance(output.get("text"), (str, list)):
data["text/plain"] = output["text"]
for v3_key, mime in _V3_MIME_KEYS:
if v3_key in output:
data[mime] = output[v3_key]
if "application/vnd.jupyter.widget-view+json" in data:
return "[interactive widget — omitted]"
# Prefer readable text: models consume text/plain far better than markup.
for mime in ("text/plain", "text/markdown"):
if mime in data:
body = _clean_stream_text(_source_text(data[mime]))
if body.strip():
return body
for mime, value in data.items():
if isinstance(mime, str) and mime.startswith("image/"):
size = _base64_bytes(_source_text(value))
return f"[{mime} output — {_human_size(size)}, omitted]"
if "text/html" in data:
html = _source_text(data["text/html"])
return f"[text/html output — {len(html):,} chars, omitted]"
mimes = ", ".join(str(m) for m in data) or "unknown"
return f"[{mimes} output — omitted]"
return ""
def _notebook_outputs(cell: dict, jq_pointer: str = "", filename: str = "") -> str:
outputs = cell.get("outputs")
if not isinstance(outputs, list):
return ""
blocks = [text for text in (_notebook_output_text(o) for o in outputs) if text]
if not blocks:
return ""
joined = "\n".join(blocks)
if len(joined) > _MAX_OUTPUT_CHARS:
omitted = len(joined) - _MAX_OUTPUT_CHARS
hint = f" — full output: jq -r '{jq_pointer}' {filename}" if jq_pointer and filename else ""
joined = joined[:_MAX_OUTPUT_CHARS] + f"\n… [{omitted:,} output chars truncated{hint}]"
return joined
_CELL_LABELS = {"markdown": "Markdown", "code": "Code", "raw": "Raw"}
def _extract_notebook(path: str) -> str:
try:
with open(path, encoding="utf-8", errors="replace") as fh:
nb = json.load(fh)
except (OSError, ValueError, json.JSONDecodeError) as exc:
raise ExtractionError(f"Not a valid notebook: {exc}") from exc
if not isinstance(nb, dict):
raise ExtractionError("Notebook root is not an object")
raw_cells = nb.get("cells")
if isinstance(raw_cells, list):
cells = [(f".cells[{i}].outputs", cell) for i, cell in enumerate(raw_cells)]
else:
cells = [
(f".worksheets[{wi}].cells[{ci}].outputs", cell)
for wi, ws in enumerate(nb.get("worksheets", []))
if isinstance(ws, dict)
for ci, cell in enumerate(ws.get("cells", []))
]
if not cells:
raise ExtractionError("Notebook contains no cells")
nb_name = os.path.basename(path)
counts = dict.fromkeys(_CELL_LABELS, 0)
out: list[str] = []
for jq_pointer, cell in cells:
if not isinstance(cell, dict):
continue
typ = cell.get("cell_type")
if typ not in _CELL_LABELS:
continue
counts[typ] += 1
suffix = f" {counts[typ]}" if typ != "raw" else ""
out.extend((f"# ── {_CELL_LABELS[typ]} cell{suffix} ──", _source_text(cell.get("source", "")).rstrip("\n"), ""))
if typ == "code":
rendered = _notebook_outputs(cell, jq_pointer, nb_name)
if rendered:
out.extend((f"# ── Output (cell {counts[typ]}) ──", rendered.rstrip("\n"), ""))
if not out:
raise ExtractionError("Notebook contains no readable cells")
return "\n".join(out).rstrip("\n") + "\n"
def _zip_xml(zf: zipfile.ZipFile, name: str) -> ET.Element:
try:
return ET.fromstring(zf.read(name))
except KeyError as exc:
raise ExtractionError(f"Missing {name}") from exc
except ET.ParseError as exc:
raise ExtractionError(f"Malformed XML in {name}: {exc}") from exc
def _optional_zip_xml(zf: zipfile.ZipFile, names: set[str], name: str) -> Optional[ET.Element]:
"""Parse an optional package part; None when absent or malformed."""
if name not in names:
return None
try:
return ET.fromstring(zf.read(name))
except ET.ParseError:
return None
def _extract_docx(path: str) -> str:
try:
with zipfile.ZipFile(path) as zf:
root = _zip_xml(zf, "word/document.xml")
except zipfile.BadZipFile as exc:
raise ExtractionError(f"Not a valid DOCX: {exc}") from exc
except OSError as exc:
raise ExtractionError(str(exc)) from exc
w = f"{{{_NS_W}}}"
lines: list[str] = []
for para in root.iter(f"{w}p"):
buf: list[str] = []
for node in para.iter():
if node.tag == f"{w}t":
buf.append(node.text or "")
elif node.tag == f"{w}tab":
buf.append("\t")
elif node.tag in {f"{w}br", f"{w}cr"}:
buf.append("\n")
lines.extend("".join(buf).split("\n"))
if not any(line.strip() for line in lines):
raise ExtractionError("DOCX contains no extractable text")
return "\n".join(lines).rstrip("\n") + "\n"
def _extract_xlsx(path: str) -> str:
try:
with zipfile.ZipFile(path) as zf:
names = set(zf.namelist())
shared = _shared_strings(zf, names)
sheets = _workbook_sheets(zf)
rels = _workbook_rels(zf, names)
out: list[str] = []
for name, state, rid in sheets:
if state in {"hidden", "veryHidden"}:
continue
part = _sheet_part(rels.get(rid, ""))
if part not in names:
continue
try:
rows = _sheet_rows(zf.read(part), shared)
except ET.ParseError:
continue
out.append(f"# ── Sheet: {name} ──")
out.extend("\t".join(row) for row in rows)
if not rows:
out.append("(empty)")
out.append("")
except zipfile.BadZipFile as exc:
raise ExtractionError(f"Not a valid XLSX: {exc}") from exc
except OSError as exc:
raise ExtractionError(str(exc)) from exc
if not out:
raise ExtractionError("XLSX has no visible sheets with content")
return "\n".join(out).rstrip("\n") + "\n"
def _shared_strings(zf: zipfile.ZipFile, names: set[str]) -> list[str]:
root = _optional_zip_xml(zf, names, "xl/sharedStrings.xml")
if root is None:
return []
s = f"{{{_NS_S}}}"
return ["".join(t.text or "" for t in item.iter(f"{s}t")) for item in root.iter(f"{s}si")]
def _workbook_sheets(zf: zipfile.ZipFile) -> list[tuple[str, str, str]]:
root = _zip_xml(zf, "xl/workbook.xml")
s, r = f"{{{_NS_S}}}", f"{{{_NS_REL}}}"
return [
(sheet.get("name", "Sheet"), sheet.get("state", "visible"), sheet.get(f"{r}id", ""))
for sheet in root.iter(f"{s}sheet")
]
def _workbook_rels(zf: zipfile.ZipFile, names: set[str]) -> dict[str, str]:
root = _optional_zip_xml(zf, names, "xl/_rels/workbook.xml.rels")
if root is None:
return {}
rel_tag = f"{{{_NS_PKG_REL}}}Relationship"
return {rel.get("Id", ""): rel.get("Target", "") for rel in root.iter(rel_tag) if rel.get("Id")}
def _sheet_part(target: str) -> str:
target = target.lstrip("/")
return posixpath.normpath(target if target.startswith("xl/") else f"xl/{target}")
def _col_index(ref: str) -> int:
idx = 0
for ch in ref:
if not ch.isalpha():
break
idx = idx * 26 + ord(ch.upper()) - ord("A") + 1
return max(idx - 1, 0)
def _sheet_rows(xml_bytes: bytes, shared: list[str]) -> list[list[str]]:
root = ET.fromstring(xml_bytes)
s = f"{{{_NS_S}}}"
rows: list[list[str]] = []
for row in root.iter(f"{s}row"):
if len(rows) >= _MAX_XLSX_ROWS_PER_SHEET:
break
cells: dict[int, str] = {}
max_col = -1
for cell in row.iter(f"{s}c"):
col = _col_index(cell.get("r", "")) if cell.get("r") else max_col + 1
if col >= _MAX_XLSX_COLS:
continue
cells[col] = _cell_value(cell, shared, s)
max_col = max(max_col, col)
rows.append([cells.get(i, "") for i in range(max_col + 1)] if max_col >= 0 else [])
while rows and not any(value.strip() for value in rows[-1]):
rows.pop()
return rows
def _cell_value(cell: ET.Element, shared: list[str], s: str) -> str:
value = cell.findtext(f"{s}v") or ""
typ = cell.get("t", "")
if typ == "s":
try:
return shared[int(value)]
except (ValueError, IndexError):
return ""
if typ == "inlineStr":
inline = cell.find(f"{s}is")
return "" if inline is None else "".join(t.text or "" for t in inline.iter(f"{s}t"))
if typ == "b":
return "TRUE" if value.strip() in {"1", "true", "TRUE"} else "FALSE"
if typ == "e":
return value or "#ERROR"
return value
# Extension -> stdlib extractor; anydoc formats fall through in extract_document_text.
_STDLIB_EXTRACTORS: dict[str, Callable[[str], str]] = {
".ipynb": _extract_notebook,
".docx": _extract_docx,
".xlsx": _extract_xlsx,
}