diff --git a/pyproject.toml b/pyproject.toml index 8d13244794..a3f22c9638 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -48,6 +48,15 @@ dependencies = [ "ruamel.yaml==0.18.17", "requests==2.33.0", # CVE-2026-25645 "jinja2==3.1.6", + # Document-to-Markdown extraction for read_file (PDF, legacy Office, + # OpenDocument, RTF, EPUB) + typed NeedsOcrError for scanned pages. + # Bundled in core by maintainer decision (read_file is a core tool and + # PDF reads are a common first-session action; the previous lazy-only + # arrangement dated to the package's uv exclude-newer quarantine, which + # has long expired). tools/lazy_deps.py `tool.doc_extract` remains the + # self-heal path for lean/broken installs — keep the pin below and the + # lazy pin in lockstep. + "firecrawl-anydoc==0.2.4", # Bumped from 2.12.5 to 2.13.4 to pull in pydantic-core 2.46.4. # pydantic-core 2.41.5 (pulled by 2.12.5) segfaults when the OpenAI SDK's # Responses API resource is exercised from a non-main thread, which is the @@ -420,7 +429,11 @@ exclude-newer = "14 days" # cutoff can still brick installs. Exempting exact pins is pure brick-risk # removal at no supply-chain cost. Guarded by # tests/test_packaging_metadata.py::test_build_system_requires_exempt_from_exclude_newer. -exclude-newer-package = { vercel = false, nemo-relay = false, huggingface_hub = false, h2 = false, aiohttp = false, cryptography = false, defusedxml = false, python-olm = false, unpaddedbase64 = false, setuptools = false, pillow = false, mcp = false } +# +# firecrawl-anydoc: same exact-pin shape (==0.2.4, hosted-OCR wiring PR). +# The pin bump WAS the review; exclude-newer adds zero float protection to +# an exact pin and would only delay the reviewed version 14 days. +exclude-newer-package = { vercel = false, nemo-relay = false, huggingface_hub = false, h2 = false, aiohttp = false, cryptography = false, defusedxml = false, python-olm = false, unpaddedbase64 = false, setuptools = false, pillow = false, mcp = false, firecrawl-anydoc = false } [tool.setuptools] # Top-level single-file modules (not packages). Without this, uv2nix's diff --git a/tests/tools/test_read_file_schema_gating.py b/tests/tools/test_read_file_schema_gating.py new file mode 100644 index 0000000000..de05bdb82f --- /dev/null +++ b/tests/tools/test_read_file_schema_gating.py @@ -0,0 +1,220 @@ +"""read_file schema diet (#95681): static unconditional format list +(anydoc bundled in core) + PDF-coverage teaching moved to the +response-time warning. + +Maintainer-directed: the schema advertised anydoc-gated formats +unconditionally ("convert too when the optional anydoc converter is +available") and pre-taught the EXTRACTION COVERAGE WARNING's own +instructions. Now the format list renders only when anydoc is importable, +and the warning (read_extract.py) is the single teacher — it fires exactly +when pages are missing, with the page map and recovery commands. +""" +import os +import sys +import unittest +from unittest.mock import patch + +sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "..")) + + + +class TestReadFileSchemaStatic(unittest.TestCase): + """Gate DROPPED by maintainer decision: anydoc is a core dependency + (bundled), so format support is stated unconditionally — a missing + converter is a broken install handled by read_extract's teaching + error, not a schema variant.""" + + def test_formats_stated_unconditionally(self): + from tools.file_tools import READ_FILE_SCHEMA + + desc = READ_FILE_SCHEMA["description"] + for token in (".ipynb", ".docx", ".pptx", ".doc/.ppt/.xls", + "PDF (text layer)", "OpenDocument", "RTF", "EPUB"): + self.assertIn(token, desc, token) + # No availability hedging, no install mechanics, no gate. + self.assertNotIn("when the optional", desc) + self.assertNotIn("auto-installed", desc) + self.assertNotIn("anydoc", desc) + + def test_pdf_wording_upgrades_with_hosted_ocr_route(self): + """The ONE dynamic word: text-layer → scanned-or-text, keyed on + hosted_ocr_available(). Nous gateway deliberately does not + upgrade (Parse proxy live-probed broken 2026-08-28).""" + import tools.file_tools as ft + + with patch("tools.read_extract.hosted_ocr_available", + return_value=True): + d = ft._read_file_schema_overrides()["description"] + self.assertIn("PDF (scanned or text)", d) + self.assertNotIn("PDF (text layer)", d) + with patch("tools.read_extract.hosted_ocr_available", + return_value=False): + o = ft._read_file_schema_overrides() + self.assertEqual(o, {}) # base wording stands + + def test_hosted_ocr_available_gate_states(self): + """Maintainer decision: ONLY a direct FIRECRAWL_API_KEY unlocks — + not config true, not the Nous gateway.""" + import tools.read_extract as rx + + # direct key → True + with patch.dict(rx.os.environ, {"FIRECRAWL_API_KEY": "fc-x"}): + with patch("hermes_cli.config.load_config_readonly", + return_value={}): + self.assertTrue(rx.hosted_ocr_available()) + # config false beats key + with patch.dict(rx.os.environ, {"FIRECRAWL_API_KEY": "fc-x"}): + with patch("hermes_cli.config.load_config_readonly", + return_value={"file_tools": {"hosted_ocr": False}}): + self.assertFalse(rx.hosted_ocr_available()) + # config true WITHOUT key → False (key is the one gate) + with patch.dict(rx.os.environ, {}, clear=False): + rx.os.environ.pop("FIRECRAWL_API_KEY", None) + with patch("hermes_cli.config.load_config_readonly", + return_value={"file_tools": {"hosted_ocr": True}}): + self.assertFalse(rx.hosted_ocr_available()) + # nothing → False (Nous gateway alone must NOT unlock) + with patch("hermes_cli.config.load_config_readonly", + return_value={}): + rx.os.environ.pop("FIRECRAWL_API_KEY", None) + self.assertFalse(rx.hosted_ocr_available()) + + def test_runtime_route_is_direct_key_only(self): + """_hosted_ocr_config never resolves the Nous gateway: api_url is + always None (anydoc defaults to api.firecrawl.dev) and enabled + tracks the key.""" + import tools.read_extract as rx + + with patch.dict(rx.os.environ, {"FIRECRAWL_API_KEY": "fc-x"}): + with patch("hermes_cli.config.load_config_readonly", + return_value={}): + enabled, key, url = rx._hosted_ocr_config() + self.assertTrue(enabled) + self.assertEqual(key, "fc-x") + self.assertIsNone(url) + with patch("hermes_cli.config.load_config_readonly", + return_value={}): + rx.os.environ.pop("FIRECRAWL_API_KEY", None) + enabled, key, url = rx._hosted_ocr_config() + self.assertFalse(enabled) + self.assertIsNone(key) + self.assertIsNone(url) + + def test_coverage_warning_teaching_left_to_the_warning(self): + """The response-time warning owns the recovery curriculum.""" + from tools.file_tools import READ_FILE_SCHEMA + + desc = READ_FILE_SCHEMA["description"] + self.assertNotIn("EXTRACTION COVERAGE WARNING", desc) + self.assertNotIn("NEEDS OCR", desc) + self.assertNotIn("pdftoppm", desc) + import inspect + from tools import read_extract + + src = inspect.getsource(read_extract) + self.assertIn("NEEDS OCR", src) + self.assertIn("pdftoppm", src) + self.assertIn("vision_analyze", src) + + def test_binary_note_stays_last(self): + from tools.file_tools import READ_FILE_SCHEMA + + desc = READ_FILE_SCHEMA["description"] + self.assertLess(desc.find("EPUB"), desc.find("Cannot read images/binary")) + + def test_missing_anydoc_error_teaches_install(self): + from tools.read_extract import _anydoc_missing_error + + err = _anydoc_missing_error("x.epub") + self.assertIn("firecrawl-anydoc", err) + self.assertNotEqual(err, "Unsupported document type: 'x.epub'") + + +class TestNeedsOcrPath(unittest.TestCase): + """anydoc>=0.2 NeedsOcrError wiring: hosted OCR attempt + typed warning + (maintainer caveats: #1 nous-gateway Parse was live-probed HTTP 500 → + attempt-and-fall-through; #2 warning recommends LOCAL OCR skills).""" + + def _fake_mod(self, hosted_result=None, hosted_exc=None): + class NeedsOcrError(Exception): + def __init__(self, pages): + super().__init__("needs ocr") + self.pages = pages + + calls = [] + + class Mod: + pass + + mod = Mod() + mod.NeedsOcrError = NeedsOcrError + + def to_markdown(path, **kw): + calls.append(kw) + if not kw: + raise NeedsOcrError([2, 3]) + if hosted_exc is not None: + raise hosted_exc + return hosted_result + + mod.to_markdown = to_markdown + return mod, calls + + def test_hosted_success_returns_ocr_text(self): + from tools import read_extract as rx + + mod, calls = self._fake_mod(hosted_result="OCR TEXT") + with patch.object(rx, "_anydoc", return_value=mod), patch.object(rx, "_hosted_ocr_config", + return_value=(True, "key", None)), patch.object(rx.os.path, "getsize", return_value=10): + out = rx._extract_anydoc("scan.pdf") + self.assertEqual(out, "OCR TEXT\n") + self.assertEqual(calls[1].get("ocr"), "hosted") + + def test_hosted_failure_warns_and_prefers_local_skills(self): + from tools import read_extract as rx + + mod, _ = self._fake_mod(hosted_exc=RuntimeError("HTTP 500")) + with patch.object(rx, "_anydoc", return_value=mod), patch.object(rx, "_hosted_ocr_config", + return_value=(True, "key", "https://gw")), patch.object(rx.os.path, "getsize", return_value=10): + out = rx._extract_anydoc("scan.pdf") + self.assertIn("[NEEDS OCR", out) + self.assertIn("pages 2, 3", out) + self.assertIn("attempted and failed", out) + # Maintainer-directed: HINT at checking for an OCR skill; never + # name one (none is guaranteed to exist), never sell config knobs. + self.assertIn("check whether an OCR skill is available", out) + self.assertIn("skills_list", out) + self.assertNotIn("ocr-and-documents", out) + self.assertNotIn("marker-pdf", out) + self.assertNotIn("hosted_ocr", out) + + def test_disabled_warns_without_attempt(self): + from tools import read_extract as rx + + mod, calls = self._fake_mod() + with patch.object(rx, "_anydoc", return_value=mod), patch.object(rx, "_hosted_ocr_config", + return_value=(False, None, None)), patch.object(rx.os.path, "getsize", return_value=10): + out = rx._extract_anydoc("scan.pdf") + self.assertIn("[NEEDS OCR", out) + self.assertEqual(len(calls), 1) # no hosted attempt + # Same shape when disabled: skill hint, no knob advertising. + self.assertIn("check whether an OCR skill is available", out) + self.assertNotIn("hosted_ocr", out) + self.assertNotIn("ocr-and-documents", out) + + def test_pin_lockstep(self): + """pyproject core pin and lazy_deps self-heal pin must match.""" + import re + from pathlib import Path + + py = Path("pyproject.toml").read_text(encoding="utf-8") + lz = Path("tools/lazy_deps.py").read_text(encoding="utf-8") + m1 = re.search(r'"firecrawl-anydoc==([\d.]+)"', py) + m2 = re.search(r'"firecrawl-anydoc==([\d.]+)"', lz) + self.assertIsNotNone(m1) + self.assertIsNotNone(m2) + self.assertEqual(m1.group(1), m2.group(1)) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/file_tools.py b/tools/file_tools.py index da3f1bf5b3..a83d73e13f 100644 --- a/tools/file_tools.py +++ b/tools/file_tools.py @@ -2648,7 +2648,15 @@ def _check_file_reqs(): READ_FILE_SCHEMA = { "name": "read_file", - "description": "Read a text file with line numbers and pagination. Use this instead of cat/head/tail in terminal. Output format: 'LINE_NUM|CONTENT'. Suggests similar filenames if not found. Use offset and limit for large files. Reads exceeding ~100K characters are truncated on a line boundary and return a next_offset; continue with offset to read the rest. Jupyter notebooks (.ipynb), Word documents (.docx), and Excel workbooks (.xlsx) are auto-extracted to readable text; PDF, legacy Office (.doc/.ppt/.xls), OpenDocument, RTF, and EPUB convert too when the optional anydoc converter is available (auto-installed on first use where installs are permitted). PDF conversion reads the text layer only: scanned/image pages yield no text, and when many pages come back empty the output ends with an EXTRACTION COVERAGE WARNING listing the affected pages — follow its instructions (render pages with pdftoppm and inspect via vision_analyze, or OCR) instead of treating the extraction as complete. NOTE: Cannot read images or other binary files — use vision_analyze for images.", + # Document formats are stated unconditionally: firecrawl-anydoc is a + # core dependency (bundled), so its absence is a broken install, not a + # configuration — the teaching error in read_extract handles that rare + # case with the pip-install fix. The ONE dynamic word: "PDF (text + # layer)" upgrades to "PDF (scanned or text)" when hosted OCR has a + # route we trust (_read_file_schema_overrides). Scanned-page coverage + # teaching lives in the response-time NEEDS-OCR warning + # (read_extract.py); the schema doesn't pre-teach it. + "description": "Read a text file with line numbers and pagination. Use this instead of cat/head/tail in terminal. Output format: 'LINE_NUM|CONTENT'. Suggests similar filenames if not found. Use offset and limit for large files. Reads exceeding ~100K characters are truncated on a line boundary and return a next_offset; continue with offset to read the rest. Documents auto-extract to readable text: .ipynb, Office (.docx/.xlsx/.pptx and legacy .doc/.ppt/.xls), PDF (text layer), OpenDocument, RTF, EPUB. Cannot read images/binary — use vision_analyze for images.", "parameters": { "type": "object", "properties": { @@ -2800,7 +2808,28 @@ def _handle_search_files(args, **kw): output_mode=args.get("output_mode", "content"), context=args.get("context", 0), task_id=tid) -registry.register(name="read_file", toolset="file", schema=READ_FILE_SCHEMA, handler=_handle_read_file, check_fn=_check_file_reqs, emoji="📖", max_result_size_chars=100_000) +def _read_file_schema_overrides(): + """One-word capability upgrade: "PDF (text layer)" → "PDF (scanned or + text)" when hosted OCR has a trusted route (see + read_extract.hosted_ocr_available). Config/env probe only — no + network at schema-build time. Compaction's tool refresh (#97073) + picks up a key added mid-session. + """ + try: + from tools.read_extract import hosted_ocr_available + + if hosted_ocr_available(): + return { + "description": READ_FILE_SCHEMA["description"].replace( + "PDF (text layer)", "PDF (scanned or text)" + ) + } + except Exception: # noqa: BLE001 + pass + return {} + + +registry.register(name="read_file", toolset="file", schema=READ_FILE_SCHEMA, handler=_handle_read_file, check_fn=_check_file_reqs, emoji="📖", max_result_size_chars=100_000, dynamic_schema_overrides=_read_file_schema_overrides) registry.register(name="write_file", toolset="file", schema=WRITE_FILE_SCHEMA, handler=_handle_write_file, check_fn=_check_file_reqs, emoji="✍️", max_result_size_chars=100_000) registry.register(name="patch", toolset="file", schema=PATCH_SCHEMA, handler=_handle_patch, check_fn=_check_file_reqs, emoji="🔧", max_result_size_chars=100_000) registry.register(name="search_files", toolset="file", schema=SEARCH_FILES_SCHEMA, handler=_handle_search_files, check_fn=_check_file_reqs, emoji="🔎", max_result_size_chars=100_000) diff --git a/tools/lazy_deps.py b/tools/lazy_deps.py index 05ff08d72e..64d4f2ec31 100644 --- a/tools/lazy_deps.py +++ b/tools/lazy_deps.py @@ -292,10 +292,10 @@ LAZY_DEPS: dict[str, tuple[str, ...]] = { # the stdlib .ipynb/.docx/.xlsx to PDF, legacy Office (.doc/.ppt/.xls), # OpenDocument, RTF, and EPUB. Installed on first read of such a file; # the call site uses prompt=False so read_file never blocks on a prompt. - # NOTE: lazy-only for now — no pyproject `doc-extract` extra until the - # package clears the uv exclude-newer 14-day quarantine (first release - # 2026-08-04); add the mirrored extra then. - "tool.doc_extract": ("firecrawl-anydoc==0.1.6",), + # NOTE: bundled in core pyproject dependencies since the hosted-OCR + # wiring (keep this lazy pin in lockstep with pyproject) — this entry + # survives as the self-heal path for lean/partial installs. + "tool.doc_extract": ("firecrawl-anydoc==0.2.4",), # lockstep with pyproject # Computer Use (cua-driver) — the MCP client SDK used to spawn and talk # to the cua-driver process over stdio. Matches the `mcp` / `computer-use` # extras in pyproject.toml. The one-liner installer pulls this in via diff --git a/tools/read_extract.py b/tools/read_extract.py index 5b50c2728b..260aa045f4 100644 --- a/tools/read_extract.py +++ b/tools/read_extract.py @@ -162,10 +162,105 @@ def extract_document_bytes(data: bytes, path: str) -> str: pass +def _anydoc_missing_error(path: str) -> str: + """Teaching error for anydoc-gated formats when the converter is absent. + + Response-time hint (#95681 pattern): the schema no longer lists the + anydoc-gated formats or the availability caveat — a session that never + touches a .doc/.odt/.epub never pays for the explanation, and one that + does gets the full story here, with the fix. + """ + return ( + f"Cannot convert {path!r}: this format needs the optional anydoc " + "converter, which is not installed (install blocked or first " + "attempt failed; retried every 5 minutes). Fix: `pip install " + "firecrawl-anydoc` in Hermes's environment, or convert the file " + "yourself via terminal (e.g. libreoffice --headless --convert-to " + "txt)." + ) + + +def _hosted_ocr_config() -> tuple: + """Resolve hosted-OCR settings: (enabled, api_key, api_url). + + Maintainer decision: the ONLY route is a direct ``FIRECRAWL_API_KEY`` + (anydoc defaults api_url to https://api.firecrawl.dev). The Nous + managed gateway is NOT used — its Parse proxy was live-probed broken + (uniform HTTP 500, 2026-08-28) while scrape/search worked; revisit + when the gateway grows Parse support. ``file_tools.hosted_ocr``: + false disables even with a key; true/unset → enabled iff key + present. Never raises. + """ + api_key = os.environ.get("FIRECRAWL_API_KEY") or None + enabled = api_key is not None + try: + from hermes_cli.config import load_config_readonly + + cfg = load_config_readonly() + section = cfg.get("file_tools") if isinstance(cfg, dict) else None + if isinstance(section, dict) and section.get("hosted_ocr") is False: + enabled = False + except Exception: # noqa: BLE001 + pass + return enabled, api_key, None + + +def hosted_ocr_available() -> bool: + """Public probe for read_file's schema line: is hosted OCR unlocked? + + Maintainer decision: ONE gate — a direct ``FIRECRAWL_API_KEY`` in the + environment. Nothing else unlocks the "PDF (scanned or text)" wording + (not the Nous gateway — Parse proxy live-probed broken 2026-08-28 — + and not config assertions). ``file_tools.hosted_ocr: false`` still + disables. Env probe only — no network at schema-build time; a key + that fails at conversion time lands in the NEEDS-OCR warning. + """ + try: + if not os.environ.get("FIRECRAWL_API_KEY"): + return False + try: + from hermes_cli.config import load_config_readonly + + cfg = load_config_readonly() + section = cfg.get("file_tools") if isinstance(cfg, dict) else None + if isinstance(section, dict) and section.get("hosted_ocr") is False: + return False + except Exception: # noqa: BLE001 + pass + return True + except Exception: # noqa: BLE001 + return False + + +def _needs_ocr_warning(path: str, pages, hosted_error: str = "") -> str: + """Typed replacement for the heuristic coverage note on full-OCR PDFs. + + Fired when anydoc raises NeedsOcrError and hosted OCR is disabled, + unavailable, or failed. Maintainer-directed shape: hint at CHECKING + for an OCR skill (never name one — none is guaranteed to exist), and + never advertise the hosted_ocr config knob — when hosted fails or is + absent, a skill or ignoring the gap are the paths that exist. + """ + page_list = ", ".join(str(p) for p in pages) if pages else "unknown" + msg = ( + f"[NEEDS OCR: pages {page_list} of this PDF are scanned images " + "with no text layer — their content is MISSING below. " + ) + if hosted_error: + msg += f"Hosted OCR was attempted and failed ({hosted_error}). " + msg += ( + "If the missing pages matter: render just those pages with " + f"`pdftoppm -jpeg -r 150 -f -l '{path}' /tmp/page` " + "and inspect via vision_analyze, or check whether an OCR skill is " + "available (skills_list)." + ) + return msg + "]\n" + + def _extract_anydoc(path: str) -> str: mod = _anydoc() if mod is None: - raise ExtractionError(f"Unsupported document type: {path!r}") + raise ExtractionError(_anydoc_missing_error(path)) try: size = os.path.getsize(path) except OSError as exc: @@ -174,11 +269,32 @@ def _extract_anydoc(path: str) -> str: raise ExtractionError( f"Document too large to convert ({size:,} bytes, limit is {MAX_ANYDOC_BYTES:,})" ) + needs_ocr = getattr(mod, "NeedsOcrError", None) try: text = mod.to_markdown(path) except OSError as exc: raise ExtractionError(str(exc)) from exc except Exception as exc: + if needs_ocr is not None and isinstance(exc, needs_ocr): + # Typed scanned-pages signal (anydoc >= 0.2). Try hosted OCR + # when a Firecrawl route exists; otherwise teach recovery. + pages = list(getattr(exc, "pages", []) or []) + enabled, api_key, api_url = _hosted_ocr_config() + hosted_error = "" + if enabled: + try: + kwargs = {"ocr": "hosted"} + if api_key: + kwargs["api_key"] = api_key + if api_url: + kwargs["api_url"] = api_url + text = mod.to_markdown(path, **kwargs) + return text.rstrip("\n") + "\n" + except Exception as hosted_exc: # noqa: BLE001 + hosted_error = f"{type(hosted_exc).__name__}: {hosted_exc}" + # No route / disabled / hosted failed: whole doc is scans — + # nothing to extract, so the warning IS the result. + return _needs_ocr_warning(path, pages, hosted_error) # anydoc raises one ConvertError subclass per failure mode # (Unsupported, Malformed, Encrypted, ResourceLimit, MissingPart). # Any of them means "no meaningful text": fall back to the normal @@ -192,6 +308,9 @@ def _extract_anydoc(path: str) -> str: if note: # Prepend: read_file paginates the extraction, so a footer on a # long document would sit on a page the model may never fetch. + # This heuristic note survives for PARTIAL coverage gaps — + # documents with a text layer plus some scanned pages, which + # convert without raising NeedsOcrError. text = note + text return text @@ -335,7 +454,7 @@ def _pdf_coverage_note(path: str, display_path: Optional[str] = None) -> str: def _extract_anydoc_bytes(data: bytes, path: str) -> str: mod = _anydoc() if mod is None: - raise ExtractionError(f"Unsupported document type: {path!r}") + raise ExtractionError(_anydoc_missing_error(path)) if len(data) > MAX_ANYDOC_BYTES: raise ExtractionError( f"Document too large to convert ({len(data):,} bytes, limit is {MAX_ANYDOC_BYTES:,})" diff --git a/uv.lock b/uv.lock index 3f7b7946fb..1f77eae9e5 100644 --- a/uv.lock +++ b/uv.lock @@ -13,17 +13,18 @@ exclude-newer-span = "P14D" [options.exclude-newer-package] setuptools = false -h2 = false +huggingface-hub = false vercel = false pillow = false unpaddedbase64 = false +h2 = false defusedxml = false aiohttp = false cryptography = false mcp = false python-olm = false +firecrawl-anydoc = false nemo-relay = false -huggingface-hub = false [manifest] overrides = [ @@ -1268,6 +1269,21 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/e5/4c/93d0f85318da65923e4b91c1c2ff03d8a458cbefebe3bc612a6693c7906d/fire-0.7.1-py3-none-any.whl", hash = "sha256:e43fd8a5033a9001e7e2973bab96070694b9f12f2e0ecf96d4683971b5ab1882", size = 115945, upload-time = "2025-08-16T20:20:22.87Z" }, ] +[[package]] +name = "firecrawl-anydoc" +version = "0.2.4" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/35/54/ee43b1661a954e53eaed85878a7ccc37ed73e02e6577bb26437ec5f9de94/firecrawl_anydoc-0.2.4.tar.gz", hash = "sha256:3e29460272fea81cde08fd5af11f6b0f1ff05919214ddc939867f72362c83032", size = 270706, upload-time = "2026-08-27T19:13:58.156Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/25/f1/5616b8d703b2bf5688eda86ede4b35620356d9c60346fe2d3efbec2a872d/firecrawl_anydoc-0.2.4-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:e3745a8b108d6575beafd37bcefe52e3f6c7ab44138e05cbdf4c6adf63caba84", size = 3441281, upload-time = "2026-08-27T19:13:43.408Z" }, + { url = "https://files.pythonhosted.org/packages/59/7d/4b2a466c8953af3b9d94471081b04ce8636680111402d2140105f98c02e5/firecrawl_anydoc-0.2.4-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:b2a01eecaa0b99f99ff6749a098f8c7a5250d41db9c53673eb93536ad575bfed", size = 3262365, upload-time = "2026-08-27T19:13:46.362Z" }, + { url = "https://files.pythonhosted.org/packages/77/b5/5784fc514f58ca19328faee701d2cc2e25961389c2d4908b2a51a1071ee4/firecrawl_anydoc-0.2.4-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:17720ba093820654162891d7995fd2dbae948b16fe1c174ba799bddfc0b92ad6", size = 3265363, upload-time = "2026-08-27T19:13:48.425Z" }, + { url = "https://files.pythonhosted.org/packages/fc/16/feeca9705bfdb237f1cb69ede0b373b144c0d51df4297e595a74b815557e/firecrawl_anydoc-0.2.4-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:0e5aed01bf6d4e5c588d3363888293f2ceeb9feb00449a32e0ba993797cf0bd3", size = 3506751, upload-time = "2026-08-27T19:13:50.372Z" }, + { url = "https://files.pythonhosted.org/packages/14/28/b4d0d3c5345cdd65c0b9113c6e6ff1b717fe360dccb658db2a9952315f9c/firecrawl_anydoc-0.2.4-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:dd162b5eb64478fcf02f53abe7765e26f4350ba657f36c25f055c4cc3dbade0a", size = 3443075, upload-time = "2026-08-27T19:13:52.417Z" }, + { url = "https://files.pythonhosted.org/packages/cb/29/50c52b0ea5dd645aea6d195da913c8d9e136601eec465bd9a590edd250c7/firecrawl_anydoc-0.2.4-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:e57c79093c1592929dd8dfaf6c60b49eefe631721d96fb833c3f465472b37b9f", size = 3742860, upload-time = "2026-08-27T19:13:54.474Z" }, + { url = "https://files.pythonhosted.org/packages/65/34/6d54fed906055e5507dd94efd63f4a05021f103dc684ea62a3b0876b7595/firecrawl_anydoc-0.2.4-cp310-abi3-win_amd64.whl", hash = "sha256:7c9ffc81946e6954c124560ed1d395e0103a1a08fdc0f6be382e2f35daf6dbc9", size = 3583555, upload-time = "2026-08-27T19:13:56.698Z" }, +] + [[package]] name = "firecrawl-py" version = "4.17.0" @@ -1578,6 +1594,7 @@ dependencies = [ { name = "cryptography" }, { name = "fastapi" }, { name = "fire" }, + { name = "firecrawl-anydoc" }, { name = "httpx", extra = ["socks"] }, { name = "jinja2" }, { name = "markdown" }, @@ -1844,6 +1861,7 @@ requires-dist = [ { name = "fastapi", marker = "extra == 'web'", specifier = "==0.133.1" }, { name = "faster-whisper", marker = "extra == 'voice'", specifier = "==1.2.1" }, { name = "fire", specifier = "==0.7.1" }, + { name = "firecrawl-anydoc", specifier = "==0.2.4" }, { name = "firecrawl-py", marker = "extra == 'firecrawl'", specifier = "==4.17.0" }, { name = "google-api-python-client", marker = "extra == 'google'", specifier = "==2.194.0" }, { name = "google-auth", marker = "extra == 'google'", specifier = "==2.55.1" }, @@ -4052,7 +4070,7 @@ resolution-markers = [ "python_full_version < '3.12'", ] dependencies = [ - { name = "numpy", marker = "python_full_version < '3.12'" }, + { name = "numpy" }, ] sdist = { url = "https://files.pythonhosted.org/packages/7a/97/5a3609c4f8d58b039179648e62dd220f89864f56f7357f5d4f45c29eb2cc/scipy-1.17.1.tar.gz", hash = "sha256:95d8e012d8cb8816c226aef832200b1d45109ed4464303e997c5b13122b297c0", size = 30573822, upload-time = "2026-02-23T00:26:24.851Z" } wheels = [ @@ -4107,7 +4125,7 @@ resolution-markers = [ "python_full_version == '3.12.*'", ] dependencies = [ - { name = "numpy", marker = "python_full_version >= '3.12'" }, + { name = "numpy" }, ] sdist = { url = "https://files.pythonhosted.org/packages/a7/25/c2700dfaf6442b4effaa91af24ebce5dc9d31bb4a69706313aae70d72cd0/scipy-1.18.0.tar.gz", hash = "sha256:67b2ad2ad54c72ca6d04975a9b2df8c3638c34ddd5b28738e94fc2b57929d378", size = 30774447, upload-time = "2026-06-19T15:01:43.456Z" } wheels = [ @@ -4755,11 +4773,11 @@ name = "vercel-workers" version = "0.0.25" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "anyio", marker = "python_full_version >= '3.12'" }, - { name = "httpx", marker = "python_full_version >= '3.12'" }, - { name = "pydantic", marker = "python_full_version >= '3.12'" }, - { name = "python-dotenv", marker = "python_full_version >= '3.12'" }, - { name = "vercel", marker = "python_full_version >= '3.12'" }, + { name = "anyio" }, + { name = "httpx" }, + { name = "pydantic" }, + { name = "python-dotenv" }, + { name = "vercel" }, ] sdist = { url = "https://files.pythonhosted.org/packages/30/df/04d37021ad7ca53b7599c313e411d91623c7a005c741f491d1eefb7a9f0c/vercel_workers-0.0.25.tar.gz", hash = "sha256:212ded01400b524be51d251df49f801caf115ad7d48cca7eb168cbeceda3def3", size = 64149, upload-time = "2026-06-20T19:26:27.177Z" } wheels = [