diff --git a/tools/neutts_synth.py b/tools/neutts_synth.py index 9d51153199..da5d9473eb 100644 --- a/tools/neutts_synth.py +++ b/tools/neutts_synth.py @@ -16,7 +16,6 @@ from pathlib import Path def _write_wav(path: str, samples, sample_rate: int = 24000) -> None: """Write a WAV file from float32 samples (no soundfile dependency).""" import numpy as np - if not isinstance(samples, np.ndarray): samples = np.array(samples, dtype=np.float32) pcm = (np.clip(samples.flatten(), -1.0, 1.0) * 32767).astype(np.int16) @@ -33,7 +32,8 @@ def main(): parser.add_argument("--out", required=True, help="Output WAV path") parser.add_argument("--ref-audio", required=True, help="Reference voice audio path") parser.add_argument("--ref-text", required=True, help="Reference voice transcript path") - parser.add_argument("--model", default="neuphonic/neutts-air-q4-gguf", help="HuggingFace backbone model repo") + parser.add_argument("--model", default="neuphonic/neutts-air-q4-gguf", + help="HuggingFace backbone model repo") parser.add_argument("--device", default="cpu", help="Device (cpu/cuda/mps)") args = parser.parse_args() @@ -60,8 +60,7 @@ def main(): backbone_repo=args.model, backbone_device="gpu" if args.device == "cuda" else args.device, codec_repo="neuphonic/neucodec", - codec_device=args.device, - ) + codec_device=args.device) wav = tts.infer(args.text, tts.encode_reference(str(ref_audio)), ref_text) out_path = Path(args.out) diff --git a/tools/open_preview_tool.py b/tools/open_preview_tool.py index f800653f97..009c194313 100644 --- a/tools/open_preview_tool.py +++ b/tools/open_preview_tool.py @@ -34,13 +34,11 @@ def open_preview_tool(url: str, label: str = "") -> str: if not target: return tool_error( "url is required — a web URL (https://…), a localhost dev server, or a " - "file path to show in the preview pane." - ) + "file path to show in the preview pane.") label = (label or "").strip() return desktop_ui.emit_or_error( "preview.open", {"url": target, "label": label}, "Failed to open the preview pane: ", "The preview pane is only available in the Hermes desktop app.", - {"success": True, "url": target, "label": label}, - ) + {"success": True, "url": target, "label": label}) diff --git a/tools/openrouter_client.py b/tools/openrouter_client.py index 83d8cbbc55..262b4d73c7 100644 --- a/tools/openrouter_client.py +++ b/tools/openrouter_client.py @@ -11,7 +11,6 @@ def check_api_key() -> bool: """ try: from agent.secret_scope import UnscopedSecretError, get_secret - try: return bool(get_secret("OPENROUTER_API_KEY")) except UnscopedSecretError: diff --git a/tools/osv_check.py b/tools/osv_check.py index 7947b182a5..1bd99e0d2a 100644 --- a/tools/osv_check.py +++ b/tools/osv_check.py @@ -36,13 +36,10 @@ def _cache_get(key) -> Tuple[bool, Optional[str]]: """Return (hit, result) for a fresh cache entry.""" with _cache_lock: entry = _cache.get(key) - if entry is None: - return False, None - expiry, result = entry - if time.monotonic() >= expiry: - del _cache[key] - return False, None - return True, result + if entry is not None and time.monotonic() < entry[0]: + return True, entry[1] + _cache.pop(key, None) # absent or expired + return False, None def _cache_put(key, result: Optional[str]) -> None: @@ -86,15 +83,14 @@ def check_package_for_malware(command: str, args: list) -> Optional[str]: if malware: ids = ", ".join(m["id"] for m in malware[:3]) summaries = "; ".join(m.get("summary", m["id"])[:100] for m in malware[:3]) - result = f"BLOCKED: Package '{package}' ({ecosystem}) has known malware advisories: {ids}. Details: {summaries}" + result = (f"BLOCKED: Package '{package}' ({ecosystem}) has known malware " + f"advisories: {ids}. Details: {summaries}") _cache_put(cache_key, result) return result _ECOSYSTEM_BY_COMMAND = { - "npx": "npm", "npx.cmd": "npm", - "uvx": "PyPI", "uvx.cmd": "PyPI", "pipx": "PyPI", -} + "npx": "npm", "npx.cmd": "npm", "uvx": "PyPI", "uvx.cmd": "PyPI", "pipx": "PyPI"} def _infer_ecosystem(command: str) -> Optional[str]: @@ -160,8 +156,7 @@ def _query_osv(package: str, ecosystem: str, version: Optional[str] = None) -> l _OSV_ENDPOINT, data=json.dumps(payload).encode("utf-8"), headers={"Content-Type": "application/json", "User-Agent": "hermes-agent-osv-check/1.0"}, - method="POST", - ) + method="POST") with urllib.request.urlopen(req, timeout=_TIMEOUT) as resp: result = json.loads(resp.read()) return [v for v in result.get("vulns", []) if v.get("id", "").startswith("MAL-")] diff --git a/tools/patch_parser.py b/tools/patch_parser.py index 66d9feb379..96ebfbab20 100644 --- a/tools/patch_parser.py +++ b/tools/patch_parser.py @@ -61,17 +61,13 @@ _OP_MARKERS: List[Tuple[OperationType, re.Pattern]] = [ (OperationType.UPDATE, re.compile(r'\*\*\*\s*Update\s+File:\s*(.+)')), (OperationType.ADD, re.compile(r'\*\*\*\s*Add\s+File:\s*(.+)')), (OperationType.DELETE, re.compile(r'\*\*\*\s*Delete\s+File:\s*(.+)')), - (OperationType.MOVE, re.compile(r'\*\*\*\s*Move\s+File:\s*(.+?)\s*->\s*(.+)')), -] + (OperationType.MOVE, re.compile(r'\*\*\*\s*Move\s+File:\s*(.+?)\s*->\s*(.+)'))] _HINT_RE = re.compile(r'@@\s*(.+?)\s*@@') def parse_v4a_patch(patch_content: str) -> Tuple[List[PatchOperation], Optional[str]]: - """Parse a V4A patch into operations. - - Returns ``(operations, None)`` — ``[]`` for an empty patch is not an - error — or ``([], "Parse error: ...")`` for malformed operations. - """ + """Parse a V4A patch -> ``(operations, None)`` (``[]`` for an empty patch is not an + error) or ``([], "Parse error: ...")`` for malformed operations.""" # Tolerate CRLF bodies: a stray ``\r`` would otherwise end up in every # HunkLine.content and defeat the anchored Begin/End markers. lines = [ln[:-1] if ln.endswith('\r') else ln for ln in patch_content.split('\n')] @@ -99,15 +95,15 @@ def parse_v4a_patch(patch_content: str) -> Tuple[List[PatchOperation], Optional[ operations.append(current_op) for line in lines[start_idx + 1:end_idx]: - op_match = next(((kind, m) for kind, rx in _OP_MARKERS if (m := rx.match(line))), None) + op_match = next( + ((kind, m) for kind, rx in _OP_MARKERS if (m := rx.match(line))), None) if op_match: kind, m = op_match _flush() current_op = PatchOperation( operation=kind, file_path=m.group(1).strip(), - new_path=m.group(2).strip() if kind is OperationType.MOVE else None, - ) + new_path=m.group(2).strip() if kind is OperationType.MOVE else None) # UPDATE hunks start lazily (at '@@' or the first hunk line); ADD # collects all '+' lines into one hunk; DELETE/MOVE are complete. current_hunk = Hunk() if kind is OperationType.ADD else None @@ -166,16 +162,14 @@ def _hint_ambiguity(content: str, hint: str, tail: str = "") -> Tuple[int, str]: """(occurrences, error) for an addition-only hunk's context hint; error is '' when unique.""" occurrences = _count_occurrences(content, hint) if occurrences > 1: - return occurrences, f"context hint '{hint}' is ambiguous ({occurrences} occurrences){tail}" + return occurrences, (f"context hint '{hint}' is ambiguous " + f"({occurrences} occurrences){tail}") return occurrences, "" def _validate_operations(operations: List[PatchOperation], file_ops: Any) -> List[str]: - """Dry-run every operation; return error strings (empty list = safe to apply). - - UPDATE hunks are simulated in order so later hunks validate against - post-earlier-hunk content, exactly as the apply phase will see it. - """ + """Dry-run every operation; return error strings (empty list = safe to apply). UPDATE + hunks are simulated in order so later hunks see post-earlier-hunk content, as apply will.""" from tools.fuzzy_match import fuzzy_find_and_replace, is_already_applied errors: List[str] = [] @@ -213,7 +207,8 @@ def _validate_operations(operations: List[PatchOperation], file_ops: Any) -> Lis if hunk.context_hint: occurrences, ambiguous = _hint_ambiguity(simulated, hunk.context_hint) if occurrences == 0: - errors.append(f"{op.file_path}: addition-only hunk context hint '{hunk.context_hint}' not found") + errors.append(f"{op.file_path}: addition-only hunk context hint " + f"'{hunk.context_hint}' not found") elif ambiguous: errors.append(f"{op.file_path}: addition-only hunk {ambiguous}") continue @@ -224,8 +219,7 @@ def _validate_operations(operations: List[PatchOperation], file_ops: Any) -> Lis # validation must not reject it with the identical-strings error. continue new_simulated, count, _strategy, match_error = fuzzy_find_and_replace( - simulated, search_pattern, replacement, replace_all=False - ) + simulated, search_pattern, replacement, replace_all=False) if count: simulated = new_simulated elif not is_already_applied(simulated or "", search_pattern, replacement): @@ -235,8 +229,7 @@ def _validate_operations(operations: List[PatchOperation], file_ops: Any) -> Lis errors.append( f"{op.file_path}: hunk {hunk_index} {label} not found" + (f" — {match_error}" if match_error else "") - + _no_match_hint(match_error, search_pattern, simulated) - ) + + _no_match_hint(match_error, search_pattern, simulated)) pending_content[op.file_path] = simulated for op in operations: @@ -279,11 +272,8 @@ ApplyResult = Tuple[bool, str, Optional[str], Optional[dict]] def apply_v4a_operations(operations: List[PatchOperation], file_ops: Any) -> 'PatchResult': """Validate all operations, then apply them (two-phase, atomic on validation failure). - - A phase-2 failure (e.g. a race between validation and apply) is reported - with a note to run ``git diff`` since state may be inconsistent. - ``file_ops`` needs ``read_file_raw``, ``write_file``, ``delete_file``, ``move_file``. - """ + A phase-2 failure (validate/apply race) is reported with a ``git diff`` note since state + may be inconsistent. ``file_ops`` needs read_file_raw/write_file/delete_file/move_file.""" from tools.file_operations import PatchResult # avoid circular import validation_errors = _validate_operations(operations, file_ops) @@ -291,8 +281,7 @@ def apply_v4a_operations(operations: List[PatchOperation], file_ops: Any) -> 'Pa return PatchResult( success=False, error="Patch validation failed (no files were modified):\n" - + "\n".join(f" • {e}" for e in validation_errors), - ) + + "\n".join(f" • {e}" for e in validation_errors)) files_modified: List[str] = [] files_created: List[str] = [] @@ -308,8 +297,7 @@ def apply_v4a_operations(operations: List[PatchOperation], file_ops: Any) -> 'Pa OperationType.ADD: (_apply_add, files_created, "add"), OperationType.DELETE: (_apply_delete, files_deleted, "delete"), OperationType.MOVE: (_apply_move, files_modified, "move"), - OperationType.UPDATE: (_apply_update, files_modified, "update"), - } + OperationType.UPDATE: (_apply_update, files_modified, "update")} for op in operations: try: @@ -318,7 +306,10 @@ def apply_v4a_operations(operations: List[PatchOperation], file_ops: Any) -> 'Pa if not ok: errors.append(f"Failed to {verb} {op.file_path}: {payload}") continue - bucket.append(f"{op.file_path} -> {op.new_path}" if op.operation is OperationType.MOVE else op.file_path) + label = op.file_path + if op.operation is OperationType.MOVE: + label = f"{op.file_path} -> {op.new_path}" + bucket.append(label) all_diffs.append(payload) if lsp: lsp_blocks.append(lsp) @@ -335,30 +326,26 @@ def apply_v4a_operations(operations: List[PatchOperation], file_ops: Any) -> 'Pa files_created=files_created, files_deleted=files_deleted, lint=lint_results if lint_results else None, - lsp_diagnostics="\n\n".join(lsp_blocks) if lsp_blocks else None, - ) + lsp_diagnostics="\n\n".join(lsp_blocks) if lsp_blocks else None) if errors: return PatchResult( success=False, error="Apply phase failed (state may be inconsistent — run `git diff` to assess):\n" + "\n".join(f" • {e}" for e in errors), - **result_kwargs, - ) + **result_kwargs) return PatchResult(success=True, **result_kwargs) def _write_file_accepts_pre_content(file_ops: Any) -> bool: - """True when ``file_ops.write_file`` accepts a ``pre_content`` kwarg. - - Decided from the signature rather than catching TypeError around the call, so a - TypeError raised *inside* a capable write_file propagates instead of triggering a - second, duplicate write. Unintrospectable callables get the two-argument form. - """ + """True when ``file_ops.write_file`` accepts ``pre_content``. Decided from the signature, + not by catching TypeError around the call, so a TypeError raised *inside* a capable + write_file propagates instead of triggering a duplicate write.""" try: params = inspect.signature(file_ops.write_file).parameters except (TypeError, ValueError): return False - return "pre_content" in params or any(p.kind is inspect.Parameter.VAR_KEYWORD for p in params.values()) + return "pre_content" in params or any( + p.kind is inspect.Parameter.VAR_KEYWORD for p in params.values()) def _apply_add(op: PatchOperation, file_ops: Any) -> ApplyResult: @@ -382,8 +369,7 @@ def _apply_delete(op: PatchOperation, file_ops: Any) -> ApplyResult: return False, result.error, None, None diff = ''.join(difflib.unified_diff( read_result.content.splitlines(keepends=True), [], - fromfile=f"a/{op.file_path}", tofile="/dev/null", - )) + fromfile=f"a/{op.file_path}", tofile="/dev/null")) return True, diff or f"# Deleted: {op.file_path}", None, None @@ -397,7 +383,8 @@ def _apply_move(op: PatchOperation, file_ops: Any) -> ApplyResult: def _insert_addition_only(new_content: str, hunk: Hunk, insert_text: str) -> Tuple[Optional[str], Optional[str]]: """Place an addition-only hunk after its context hint (or at EOF). Returns (content, error).""" if hunk.context_hint: - occurrences, ambiguous = _hint_ambiguity(new_content, hunk.context_hint, " — provide a more unique hint") + occurrences, ambiguous = _hint_ambiguity( + new_content, hunk.context_hint, " — provide a more unique hint") if ambiguous: return None, f"Addition-only hunk: {ambiguous}" if occurrences == 1: @@ -433,8 +420,7 @@ def _apply_update(op: PatchOperation, file_ops: Any) -> ApplyResult: search_pattern = '\n'.join(search_lines) replacement = '\n'.join(replace_lines) new_content, count, _strategy, error = fuzzy_find_and_replace( - new_content, search_pattern, replacement, replace_all=False - ) + new_content, search_pattern, replacement, replace_all=False) if not (error and count == 0): continue @@ -466,6 +452,5 @@ def _apply_update(op: PatchOperation, file_ops: Any) -> ApplyResult: diff = ''.join(difflib.unified_diff( current_content.splitlines(keepends=True), new_content.splitlines(keepends=True), - fromfile=f"a/{op.file_path}", tofile=f"b/{op.file_path}", - )) + fromfile=f"a/{op.file_path}", tofile=f"b/{op.file_path}")) return True, diff, getattr(write_result, "lsp_diagnostics", None), getattr(write_result, "lint", None) diff --git a/tools/preview_tool.py b/tools/preview_tool.py index 6856118f17..cba89ff8a5 100644 --- a/tools/preview_tool.py +++ b/tools/preview_tool.py @@ -27,7 +27,8 @@ _ACTIONS = { "open": lambda args: open_preview_tool(url=args.get("url", ""), label=args.get("label", "")), "close": lambda args: preview_close(url=args.get("url", "")), # read needs the GUI callback and is dispatched at the agent level. - "read": lambda args: tool_error("preview read must run inside a desktop session (no GUI callback here)."), + "read": lambda args: tool_error( + "preview read must run inside a desktop session (no GUI callback here)."), } diff --git a/tools/read_extract.py b/tools/read_extract.py index aeed3c6ae3..dba4c9020e 100644 --- a/tools/read_extract.py +++ b/tools/read_extract.py @@ -30,15 +30,13 @@ __all__ = [ "ExtractionError", "extract_document_bytes", "extract_document_text", - "is_extractable_document", -] + "is_extractable_document"] EXTRACTABLE_EXTENSIONS = frozenset({".ipynb", ".docx", ".xlsx"}) # Formats handled only when the optional anydoc converter is installed. ANYDOC_EXTENSIONS = frozenset({ ".doc", ".docm", ".ppt", ".pps", ".pot", ".pptx", ".pptm", ".ppsx", ".ppsm", - ".xls", ".xlsm", ".xlsb", ".odt", ".ods", ".odp", ".rtf", ".epub", ".pdf", -}) + ".xls", ".xlsm", ".xlsb", ".odt", ".ods", ".odp", ".rtf", ".epub", ".pdf"}) # anydoc loads the whole file through its Rust core with no streaming, and the # read_file char budget only applies after conversion — cap the input size. MAX_ANYDOC_BYTES = 50 * 1024 * 1024 @@ -73,18 +71,16 @@ _anydoc_failed_at: Optional[float] = None def _anydoc() -> Optional[Any]: - """Lazily import the optional anydoc converter; None when unavailable. - - A failed load is retried after :data:`ANYDOC_RETRY_SECONDS` rather than disabling - extraction for the rest of the process (one transient pip/network blip must not stick). - """ + """Lazily import the optional anydoc converter; None when unavailable. A failed load is + retried after ANYDOC_RETRY_SECONDS so one transient pip/network blip does not stick.""" global _anydoc_module, _anydoc_failed_at if _anydoc_module is not _ANYDOC_UNSET: return _anydoc_module with _anydoc_lock: if _anydoc_module is not _ANYDOC_UNSET: return _anydoc_module - if _anydoc_failed_at is not None and time.monotonic() - _anydoc_failed_at < ANYDOC_RETRY_SECONDS: + if (_anydoc_failed_at is not None + and time.monotonic() - _anydoc_failed_at < ANYDOC_RETRY_SECONDS): return None try: from tools.lazy_deps import ensure as _lazy_ensure @@ -155,8 +151,7 @@ def _anydoc_missing_error(path: str) -> str: "attempt failed; retried every 5 minutes). Fix: `pip install " "firecrawl-anydoc` in Hermes's environment, or convert the file " "yourself via terminal (e.g. libreoffice --headless --convert-to " - "txt)." - ) + "txt).") def _hosted_ocr_config() -> tuple: @@ -172,7 +167,6 @@ def _hosted_ocr_config() -> tuple: enabled = api_key is not None with contextlib.suppress(Exception): from hermes_cli.config import load_config_readonly - cfg = load_config_readonly() section = cfg.get("file_tools") if isinstance(cfg, dict) else None if isinstance(section, dict) and section.get("hosted_ocr") is False: @@ -187,33 +181,25 @@ def hosted_ocr_available() -> bool: def _needs_ocr_warning(path: str, pages, hosted_error: str = "") -> str: - """Result text when anydoc raises NeedsOcrError and hosted OCR is off/failed. - - Hints at CHECKING for an OCR skill (never names one — none is guaranteed to - exist) and never advertises the hosted_ocr config knob. - """ + """Result text when anydoc raises NeedsOcrError and hosted OCR is off/failed. Hints at + CHECKING for an OCR skill (never names one) and never advertises the hosted_ocr knob.""" page_list = ", ".join(str(p) for p in pages) if pages else "unknown" msg = ( f"[NEEDS OCR: pages {page_list} of this PDF are scanned images " - "with no text layer — their content is MISSING below. " - ) + "with no text layer — their content is MISSING below. ") if hosted_error: msg += f"Hosted OCR was attempted and failed ({hosted_error}). " msg += ( "If the missing pages matter: render just those pages with " f"`pdftoppm -jpeg -r 150 -f -l '{path}' /tmp/page` " "and inspect via vision_analyze, or check whether an OCR skill is " - "available (skills_list)." - ) + "available (skills_list).") return msg + "]\n" def _finalize_anydoc_text(text: Any, path: str, pdf_note: Callable[[], str]) -> str: - """Normalize converter output and, for PDFs, PREPEND the coverage note. - - Prepended because read_file paginates: a footer on a long document would sit on a page - the model may never fetch. Covers PARTIAL gaps that convert without NeedsOcrError. - """ + """Normalize converter output and, for PDFs, PREPEND the coverage note (read_file + paginates: a footer may never be fetched). Covers PARTIAL gaps without NeedsOcrError.""" if not isinstance(text, str) or not text.strip(): raise ExtractionError("Document contains no extractable text") text = text.rstrip("\n") + "\n" @@ -231,8 +217,8 @@ def _ocr_scanned_pdf(mod: Any, path: str, exc: BaseException) -> str: hosted_error = "" if enabled: try: - kwargs = {"ocr": "hosted", **{k: v for k, v in (("api_key", api_key), ("api_url", api_url)) if v}} - return mod.to_markdown(path, **kwargs).rstrip("\n") + "\n" + extra = {k: v for k, v in (("api_key", api_key), ("api_url", api_url)) if v} + return mod.to_markdown(path, ocr="hosted", **extra).rstrip("\n") + "\n" except Exception as hosted_exc: # noqa: BLE001 hosted_error = f"{type(hosted_exc).__name__}: {hosted_exc}" # No route / disabled / hosted failed: whole doc is scans — the warning IS the result. @@ -277,10 +263,9 @@ def _extract_anydoc_bytes(data: bytes, path: str) -> str: return _finalize_anydoc_text(text, path, lambda: _pdf_coverage_note_from_bytes(data, path)) -# ── Scanned-PDF coverage detection ────────────────────────────────── -# Text-layer extractors return nothing for scanned pages, so a mostly-scanned PDF -# converts "successfully" into headers with empty bodies — silent data loss the model -# cannot detect. Count per-page text via pdftotext (form-feed separated) and warn. +# ── Scanned-PDF coverage detection: text-layer extractors return nothing for scanned +# pages, so a mostly-scanned PDF converts "successfully" into headers with empty bodies — +# silent data loss. Count per-page text via pdftotext (form-feed separated) and warn. PDF_EMPTY_PAGE_CHARS = 20 # fewer extracted chars than this = empty page # Warn when empty pages reach both MIN_EMPTY and MIN_RATIO, or ABSOLUTE_EMPTY alone. PDF_COVERAGE_MIN_EMPTY = 2 @@ -297,7 +282,8 @@ def _pdf_page_texts(path: str) -> Optional[list[str]]: if shutil.which("pdftotext") is None: return None try: - proc = subprocess.run(["pdftotext", path, "-"], capture_output=True, timeout=PDF_PAGE_SCAN_TIMEOUT) + proc = subprocess.run( + ["pdftotext", path, "-"], capture_output=True, timeout=PDF_PAGE_SCAN_TIMEOUT) except (OSError, subprocess.SubprocessError): return None if proc.returncode != 0: @@ -308,21 +294,15 @@ def _pdf_page_texts(path: str) -> Optional[list[str]]: return pages or None -def _group_ranges(pages: list[int]) -> list[list[int]]: - """Group sorted 1-based page numbers into [start, end] runs.""" - ranges: list[list[int]] = [] - for p in pages: +def _gap_map(counts: list[int], texts: list[str], empty: list[int]) -> str: + """Per-gap breakdown, each empty range labeled with the last text seen before + it (usually a section header), so the agent can pick WHICH gaps to OCR.""" + ranges: list[list[int]] = [] # sorted 1-based page numbers -> [start, end] runs + for p in empty: if ranges and p == ranges[-1][1] + 1: ranges[-1][1] = p else: ranges.append([p, p]) - return ranges - - -def _gap_map(counts: list[int], texts: list[str], empty: list[int]) -> str: - """Per-gap breakdown, each empty range labeled with the last text seen before - it (usually a section header), so the agent can pick WHICH gaps to OCR.""" - ranges = _group_ranges(empty) lines: list[str] = [] for a, b in ranges[:PDF_GAP_MAP_MAX_ENTRIES]: label = "" @@ -341,20 +321,17 @@ def _gap_map(counts: list[int], texts: list[str], empty: list[int]) -> str: def _pdf_coverage_note(path: str, display_path: Optional[str] = None) -> str: - """Warning header when many PDF pages produced no text, else ''. - - ``path`` is scanned with pdftotext (may be a host temp file); ``display_path`` - is what the recovery command shows — the path the agent's terminal can see. - """ + """Warning header when many PDF pages produced no text, else ''. ``path`` is scanned + (may be a host temp file); ``display_path`` is what the recovery command shows.""" texts = _pdf_page_texts(path) if not texts or len(texts) < 2: return "" counts = [len(page.strip()) for page in texts] empty = [i + 1 for i, n in enumerate(counts) if n < PDF_EMPTY_PAGE_CHARS] total = len(counts) - if len(empty) < PDF_COVERAGE_MIN_EMPTY: - return "" - if len(empty) / total < PDF_COVERAGE_MIN_RATIO and len(empty) < PDF_COVERAGE_ABSOLUTE_EMPTY: + if len(empty) < PDF_COVERAGE_MIN_EMPTY or ( + len(empty) / total < PDF_COVERAGE_MIN_RATIO and len(empty) < PDF_COVERAGE_ABSOLUTE_EMPTY + ): return "" shown = display_path or path return ( @@ -370,8 +347,7 @@ def _pdf_coverage_note(path: str, display_path: Optional[str] = None) -> str: f"`pdftoppm -jpeg -r 150 -f -l '{shown}' /tmp/page` " "and inspect each image with the vision_analyze tool, or use the " "ocr-and-documents skill (marker-pdf) for bulk OCR of large " - "ranges.]\n" - ) + "ranges.]\n") def _pdf_coverage_note_from_bytes(data: bytes, display_path: str) -> str: @@ -407,7 +383,6 @@ def _clean_stream_text(text: str) -> str: """Strip ANSI escapes and collapse ``\\r`` progress-bar rewrites: Jupyter renders only the final frame of a ``\\r``-redrawn line (tqdm), so keep the text after the last ``\\r``.""" from tools.ansi_strip import strip_ansi - lines = [] for line in strip_ansi(text).replace("\r\n", "\n").split("\n"): frames = [frame for frame in line.split("\r") if frame] @@ -424,12 +399,9 @@ _V3_MIME_KEYS = (("png", "image/png"), ("jpeg", "image/jpeg"), ("svg", "image/sv def _notebook_output_text(output: Any) -> str: - """Render one notebook output as compact text. - - Keeps stream text, tracebacks, and textual results; replaces token-heavy payloads - (base64 images, HTML, widget state) with short sized placeholders. Handles nbformat - v4 and legacy v3 (``pyout``/``pyerr``) shapes. - """ + """Render one notebook output as compact text: stream text, tracebacks and textual + results kept; token-heavy payloads (images, HTML, widgets) become sized placeholders. + Handles nbformat v4 and legacy v3 (``pyout``/``pyerr``) shapes.""" if not isinstance(output, dict): return "" otype = output.get("output_type") @@ -440,7 +412,8 @@ def _notebook_output_text(output: Any) -> str: traceback = output.get("traceback") tb_text = "" if isinstance(traceback, list): - tb_text = _clean_stream_text("\n".join(line for line in traceback if isinstance(line, str))) + tb_text = _clean_stream_text( + "\n".join(line for line in traceback if isinstance(line, str))) header = f"Error: {output.get('ename', '')}: {output.get('evalue', '')}".rstrip(": ") return f"{header}\n{tb_text}".rstrip() if otype not in {"execute_result", "display_data", "pyout"}: @@ -464,9 +437,11 @@ def _notebook_output_text(output: Any) -> str: return body for mime, value in data.items(): if isinstance(mime, str) and mime.startswith("image/"): - return f"[{mime} output — {_human_size(_base64_bytes(_source_text(value)))}, omitted]" + size = _base64_bytes(_source_text(value)) + return f"[{mime} output — {_human_size(size)}, omitted]" if "text/html" in data: - return f"[text/html output — {len(_source_text(data['text/html'])):,} chars, omitted]" + html = _source_text(data["text/html"]) + return f"[text/html output — {len(html):,} chars, omitted]" mimes = ", ".join(str(m) for m in data) or "unknown" return f"[{mimes} output — omitted]" @@ -505,8 +480,7 @@ def _extract_notebook(path: str) -> str: cells = [ (f".worksheets[{wi}].cells[{ci}].outputs", cell) for wi, ws in enumerate(nb.get("worksheets", [])) if isinstance(ws, dict) - for ci, cell in enumerate(ws.get("cells", [])) - ] + for ci, cell in enumerate(ws.get("cells", []))] if not cells: raise ExtractionError("Notebook contains no cells") @@ -533,7 +507,7 @@ def _extract_notebook(path: str) -> str: @contextlib.contextmanager def _open_zip(path: str, kind: str) -> Iterator[zipfile.ZipFile]: - """Open an OOXML package, mapping bad-zip/OS failures (also from the body) to ExtractionError.""" + """Open an OOXML package; bad-zip/OS failures (also from the body) become ExtractionError.""" try: with zipfile.ZipFile(path) as zf: yield zf @@ -560,18 +534,12 @@ def _zip_xml(zf: zipfile.ZipFile, name: str, optional: bool = False) -> Any: def _extract_docx(path: str) -> str: with _open_zip(path, "DOCX") as zf: root = _zip_xml(zf, "word/document.xml") - w = f"{{{_NS_W}}}" + breaks = {f"{w}tab": "\t", f"{w}br": "\n", f"{w}cr": "\n"} lines: list[str] = [] for para in root.iter(f"{w}p"): - buf: list[str] = [] - for node in para.iter(): - if node.tag == f"{w}t": - buf.append(node.text or "") - elif node.tag == f"{w}tab": - buf.append("\t") - elif node.tag in {f"{w}br", f"{w}cr"}: - buf.append("\n") + buf = [(node.text or "") if node.tag == f"{w}t" else breaks.get(node.tag, "") + for node in para.iter()] lines.extend("".join(buf).split("\n")) if not any(line.strip() for line in lines): raise ExtractionError("DOCX contains no extractable text") @@ -579,15 +547,21 @@ def _extract_docx(path: str) -> str: def _extract_xlsx(path: str) -> str: + s, r, pr = f"{{{_NS_S}}}", f"{{{_NS_REL}}}", f"{{{_NS_PKG_REL}}}" with _open_zip(path, "XLSX") as zf: names = set(zf.namelist()) - shared = _shared_strings(zf) - rels = _workbook_rels(zf) + sst = _zip_xml(zf, "xl/sharedStrings.xml", optional=True) + shared = [] if sst is None else [ + "".join(t.text or "" for t in item.iter(f"{s}t")) for item in sst.iter(f"{s}si")] + rels_root = _zip_xml(zf, "xl/_rels/workbook.xml.rels", optional=True) + rels = {} if rels_root is None else { + rel.get("Id", ""): rel.get("Target", "") + for rel in rels_root.iter(f"{pr}Relationship") if rel.get("Id")} out: list[str] = [] - for name, state, rid in _workbook_sheets(zf): - if state in {"hidden", "veryHidden"}: + for sheet in _zip_xml(zf, "xl/workbook.xml").iter(f"{s}sheet"): + if sheet.get("state", "visible") in {"hidden", "veryHidden"}: continue - target = rels.get(rid, "").lstrip("/") + target = rels.get(sheet.get(f"{r}id", ""), "").lstrip("/") part = posixpath.normpath(target if target.startswith("xl/") else f"xl/{target}") if part not in names: continue @@ -595,7 +569,7 @@ def _extract_xlsx(path: str) -> str: rows = _sheet_rows(zf.read(part), shared) except ET.ParseError: continue - out.append(f"# ── Sheet: {name} ──") + out.append(f"# ── Sheet: {sheet.get('name', 'Sheet')} ──") out.extend("\t".join(row) for row in rows) if not rows: out.append("(empty)") @@ -606,31 +580,6 @@ def _extract_xlsx(path: str) -> str: return "\n".join(out).rstrip("\n") + "\n" -def _shared_strings(zf: zipfile.ZipFile) -> list[str]: - root = _zip_xml(zf, "xl/sharedStrings.xml", optional=True) - if root is None: - return [] - s = f"{{{_NS_S}}}" - return ["".join(t.text or "" for t in item.iter(f"{s}t")) for item in root.iter(f"{s}si")] - - -def _workbook_sheets(zf: zipfile.ZipFile) -> list[tuple[str, str, str]]: - root = _zip_xml(zf, "xl/workbook.xml") - s, r = f"{{{_NS_S}}}", f"{{{_NS_REL}}}" - return [ - (sheet.get("name", "Sheet"), sheet.get("state", "visible"), sheet.get(f"{r}id", "")) - for sheet in root.iter(f"{s}sheet") - ] - - -def _workbook_rels(zf: zipfile.ZipFile) -> dict[str, str]: - root = _zip_xml(zf, "xl/_rels/workbook.xml.rels", optional=True) - if root is None: - return {} - rel_tag = f"{{{_NS_PKG_REL}}}Relationship" - return {rel.get("Id", ""): rel.get("Target", "") for rel in root.iter(rel_tag) if rel.get("Id")} - - def _col_index(ref: str) -> int: idx = 0 for ch in ref: @@ -674,14 +623,9 @@ def _cell_value(cell: ET.Element, shared: list[str], s: str) -> str: return "" if inline is None else "".join(t.text or "" for t in inline.iter(f"{s}t")) if typ == "b": return "TRUE" if value.strip() in {"1", "true", "TRUE"} else "FALSE" - if typ == "e": - return value or "#ERROR" - return value + return (value or "#ERROR") if typ == "e" else value # Extension -> stdlib extractor; anydoc formats fall through in extract_document_text. _STDLIB_EXTRACTORS: dict[str, Callable[[str], str]] = { - ".ipynb": _extract_notebook, - ".docx": _extract_docx, - ".xlsx": _extract_xlsx, -} + ".ipynb": _extract_notebook, ".docx": _extract_docx, ".xlsx": _extract_xlsx} diff --git a/tools/read_preview_tool.py b/tools/read_preview_tool.py index 57b52db34a..68e3f63b93 100644 --- a/tools/read_preview_tool.py +++ b/tools/read_preview_tool.py @@ -13,14 +13,11 @@ from tools.read_terminal_tool import read_pane def read_preview_tool( - start: Optional[int] = None, - count: Optional[int] = None, - callback: Optional[Callable] = None, + start: Optional[int] = None, count: Optional[int] = None, callback: Optional[Callable] = None ) -> str: """Return the active preview tab's contents (+ metadata) as a JSON string.""" return read_pane(callback, (("start", start, 0), ("count", count, 1)), ( "read_preview is only available in the Hermes desktop app.", "start and count must be integers.", "Failed to read the preview pane: ", - "No preview tab is open, or the read timed out.", - )) + "No preview tab is open, or the read timed out.")) diff --git a/tools/read_terminal_tool.py b/tools/read_terminal_tool.py index 945bedf474..5f34e370f3 100644 --- a/tools/read_terminal_tool.py +++ b/tools/read_terminal_tool.py @@ -77,9 +77,7 @@ registry.register( toolset="desktop_ui", schema=READ_TERMINAL_SCHEMA, handler=lambda args, **kw: read_terminal_tool( - start_line=args.get("start_line"), - count=args.get("count"), - callback=kw.get("callback"), + start_line=args.get("start_line"), count=args.get("count"), callback=kw.get("callback") ), emoji="🖥️", ) diff --git a/tools/read_window_tool.py b/tools/read_window_tool.py index ba0c5cf45c..b41ab6db06 100644 --- a/tools/read_window_tool.py +++ b/tools/read_window_tool.py @@ -34,8 +34,7 @@ READ_WINDOW_BELOW_SCHEMA = { "retrying. Metadata only; never captures pixels." ), "parameters": { - "type": "object", - "properties": {}, + "type": "object", "properties": {} }, } diff --git a/tools/shell_heredoc.py b/tools/shell_heredoc.py index c8fa8bb0c9..df822daff2 100644 --- a/tools/shell_heredoc.py +++ b/tools/shell_heredoc.py @@ -27,8 +27,7 @@ _INERT_HEREDOC_CONSUMER_RE = re.compile( r"(?:env\s+)?" r"(?:[A-Za-z0-9_./-]+/)?" r"(?:python(?:3(?:\.\d+)*)?|osascript|cat)(?=\s|$)", - re.IGNORECASE, -) + re.IGNORECASE) def _span_end(command: str, cursor: int, closer: str) -> int: @@ -196,7 +195,8 @@ def _scan_heredoc_command_unit(command: str, start: int): return len(command), specs, unknown_operator, has_list_operator -def _find_heredoc_close(command: str, body_start: int, delimiter: str, strip_tabs: bool) -> int | None: +def _find_heredoc_close( + command: str, body_start: int, delimiter: str, strip_tabs: bool) -> int | None: """Return the position after an exact shell heredoc terminator line.""" cursor = body_start while True: @@ -222,7 +222,8 @@ def strip_inert_heredoc_bodies(command: str) -> str: command_start = 0 while command_start <= last_opener_index: - command_end, specs, unknown_operator, has_list_operator = _scan_heredoc_command_unit(command, command_start) + command_end, specs, unknown_operator, has_list_operator = ( + _scan_heredoc_command_unit(command, command_start)) if unknown_operator: return command if not specs: