refactor(read_file): schema diet + bundle anydoc 0.2.4 + typed NeedsOcrError/hosted-OCR wiring (#97195)

* refactor(read_file): capability-gate the anydoc format list; PDF coverage teaching lives in the response-time warning (426 -> 244/291 tok/call)

* feat(read_file): bundle firecrawl-anydoc 0.2.4 in core, typed NeedsOcrError handling, config-gated hosted OCR with local-OCR-first guidance

* refine(read_file): NEEDS-OCR warning hints at checking for an OCR skill without naming one; hosted_ocr knob unadvertised (maintainer-directed)

* simplify(read_file): drop the anydoc schema gate — bundled core dep makes absence a broken install, not a variant; formats stated unconditionally (263 tok/call)

* feat(read_file): PDF wording upgrades to 'scanned or text' when a trusted hosted-OCR route exists (direct key or explicit config; nous gateway excluded until Parse proxy works)

* simplify(read_file): FIRECRAWL_API_KEY is the ONLY hosted-OCR gate — nous gateway route removed (Parse proxy broken), config true no longer unlocks; false still disables
This commit is contained in:
Teknium
2026-08-28 08:46:11 -07:00
committed by GitHub
parent 306db2776c
commit a9e72f1b58
6 changed files with 417 additions and 18 deletions

View File

@@ -48,6 +48,15 @@ dependencies = [
"ruamel.yaml==0.18.17",
"requests==2.33.0", # CVE-2026-25645
"jinja2==3.1.6",
# Document-to-Markdown extraction for read_file (PDF, legacy Office,
# OpenDocument, RTF, EPUB) + typed NeedsOcrError for scanned pages.
# Bundled in core by maintainer decision (read_file is a core tool and
# PDF reads are a common first-session action; the previous lazy-only
# arrangement dated to the package's uv exclude-newer quarantine, which
# has long expired). tools/lazy_deps.py `tool.doc_extract` remains the
# self-heal path for lean/broken installs — keep the pin below and the
# lazy pin in lockstep.
"firecrawl-anydoc==0.2.4",
# Bumped from 2.12.5 to 2.13.4 to pull in pydantic-core 2.46.4.
# pydantic-core 2.41.5 (pulled by 2.12.5) segfaults when the OpenAI SDK's
# Responses API resource is exercised from a non-main thread, which is the
@@ -420,7 +429,11 @@ exclude-newer = "14 days"
# cutoff can still brick installs. Exempting exact pins is pure brick-risk
# removal at no supply-chain cost. Guarded by
# tests/test_packaging_metadata.py::test_build_system_requires_exempt_from_exclude_newer.
exclude-newer-package = { vercel = false, nemo-relay = false, huggingface_hub = false, h2 = false, aiohttp = false, cryptography = false, defusedxml = false, python-olm = false, unpaddedbase64 = false, setuptools = false, pillow = false, mcp = false }
#
# firecrawl-anydoc: same exact-pin shape (==0.2.4, hosted-OCR wiring PR).
# The pin bump WAS the review; exclude-newer adds zero float protection to
# an exact pin and would only delay the reviewed version 14 days.
exclude-newer-package = { vercel = false, nemo-relay = false, huggingface_hub = false, h2 = false, aiohttp = false, cryptography = false, defusedxml = false, python-olm = false, unpaddedbase64 = false, setuptools = false, pillow = false, mcp = false, firecrawl-anydoc = false }
[tool.setuptools]
# Top-level single-file modules (not packages). Without this, uv2nix's

View File

@@ -0,0 +1,220 @@
"""read_file schema diet (#95681): static unconditional format list
(anydoc bundled in core) + PDF-coverage teaching moved to the
response-time warning.
Maintainer-directed: the schema advertised anydoc-gated formats
unconditionally ("convert too when the optional anydoc converter is
available") and pre-taught the EXTRACTION COVERAGE WARNING's own
instructions. Now the format list renders only when anydoc is importable,
and the warning (read_extract.py) is the single teacher — it fires exactly
when pages are missing, with the page map and recovery commands.
"""
import os
import sys
import unittest
from unittest.mock import patch
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", ".."))
class TestReadFileSchemaStatic(unittest.TestCase):
"""Gate DROPPED by maintainer decision: anydoc is a core dependency
(bundled), so format support is stated unconditionally — a missing
converter is a broken install handled by read_extract's teaching
error, not a schema variant."""
def test_formats_stated_unconditionally(self):
from tools.file_tools import READ_FILE_SCHEMA
desc = READ_FILE_SCHEMA["description"]
for token in (".ipynb", ".docx", ".pptx", ".doc/.ppt/.xls",
"PDF (text layer)", "OpenDocument", "RTF", "EPUB"):
self.assertIn(token, desc, token)
# No availability hedging, no install mechanics, no gate.
self.assertNotIn("when the optional", desc)
self.assertNotIn("auto-installed", desc)
self.assertNotIn("anydoc", desc)
def test_pdf_wording_upgrades_with_hosted_ocr_route(self):
"""The ONE dynamic word: text-layer → scanned-or-text, keyed on
hosted_ocr_available(). Nous gateway deliberately does not
upgrade (Parse proxy live-probed broken 2026-08-28)."""
import tools.file_tools as ft
with patch("tools.read_extract.hosted_ocr_available",
return_value=True):
d = ft._read_file_schema_overrides()["description"]
self.assertIn("PDF (scanned or text)", d)
self.assertNotIn("PDF (text layer)", d)
with patch("tools.read_extract.hosted_ocr_available",
return_value=False):
o = ft._read_file_schema_overrides()
self.assertEqual(o, {}) # base wording stands
def test_hosted_ocr_available_gate_states(self):
"""Maintainer decision: ONLY a direct FIRECRAWL_API_KEY unlocks —
not config true, not the Nous gateway."""
import tools.read_extract as rx
# direct key → True
with patch.dict(rx.os.environ, {"FIRECRAWL_API_KEY": "fc-x"}):
with patch("hermes_cli.config.load_config_readonly",
return_value={}):
self.assertTrue(rx.hosted_ocr_available())
# config false beats key
with patch.dict(rx.os.environ, {"FIRECRAWL_API_KEY": "fc-x"}):
with patch("hermes_cli.config.load_config_readonly",
return_value={"file_tools": {"hosted_ocr": False}}):
self.assertFalse(rx.hosted_ocr_available())
# config true WITHOUT key → False (key is the one gate)
with patch.dict(rx.os.environ, {}, clear=False):
rx.os.environ.pop("FIRECRAWL_API_KEY", None)
with patch("hermes_cli.config.load_config_readonly",
return_value={"file_tools": {"hosted_ocr": True}}):
self.assertFalse(rx.hosted_ocr_available())
# nothing → False (Nous gateway alone must NOT unlock)
with patch("hermes_cli.config.load_config_readonly",
return_value={}):
rx.os.environ.pop("FIRECRAWL_API_KEY", None)
self.assertFalse(rx.hosted_ocr_available())
def test_runtime_route_is_direct_key_only(self):
"""_hosted_ocr_config never resolves the Nous gateway: api_url is
always None (anydoc defaults to api.firecrawl.dev) and enabled
tracks the key."""
import tools.read_extract as rx
with patch.dict(rx.os.environ, {"FIRECRAWL_API_KEY": "fc-x"}):
with patch("hermes_cli.config.load_config_readonly",
return_value={}):
enabled, key, url = rx._hosted_ocr_config()
self.assertTrue(enabled)
self.assertEqual(key, "fc-x")
self.assertIsNone(url)
with patch("hermes_cli.config.load_config_readonly",
return_value={}):
rx.os.environ.pop("FIRECRAWL_API_KEY", None)
enabled, key, url = rx._hosted_ocr_config()
self.assertFalse(enabled)
self.assertIsNone(key)
self.assertIsNone(url)
def test_coverage_warning_teaching_left_to_the_warning(self):
"""The response-time warning owns the recovery curriculum."""
from tools.file_tools import READ_FILE_SCHEMA
desc = READ_FILE_SCHEMA["description"]
self.assertNotIn("EXTRACTION COVERAGE WARNING", desc)
self.assertNotIn("NEEDS OCR", desc)
self.assertNotIn("pdftoppm", desc)
import inspect
from tools import read_extract
src = inspect.getsource(read_extract)
self.assertIn("NEEDS OCR", src)
self.assertIn("pdftoppm", src)
self.assertIn("vision_analyze", src)
def test_binary_note_stays_last(self):
from tools.file_tools import READ_FILE_SCHEMA
desc = READ_FILE_SCHEMA["description"]
self.assertLess(desc.find("EPUB"), desc.find("Cannot read images/binary"))
def test_missing_anydoc_error_teaches_install(self):
from tools.read_extract import _anydoc_missing_error
err = _anydoc_missing_error("x.epub")
self.assertIn("firecrawl-anydoc", err)
self.assertNotEqual(err, "Unsupported document type: 'x.epub'")
class TestNeedsOcrPath(unittest.TestCase):
"""anydoc>=0.2 NeedsOcrError wiring: hosted OCR attempt + typed warning
(maintainer caveats: #1 nous-gateway Parse was live-probed HTTP 500 →
attempt-and-fall-through; #2 warning recommends LOCAL OCR skills)."""
def _fake_mod(self, hosted_result=None, hosted_exc=None):
class NeedsOcrError(Exception):
def __init__(self, pages):
super().__init__("needs ocr")
self.pages = pages
calls = []
class Mod:
pass
mod = Mod()
mod.NeedsOcrError = NeedsOcrError
def to_markdown(path, **kw):
calls.append(kw)
if not kw:
raise NeedsOcrError([2, 3])
if hosted_exc is not None:
raise hosted_exc
return hosted_result
mod.to_markdown = to_markdown
return mod, calls
def test_hosted_success_returns_ocr_text(self):
from tools import read_extract as rx
mod, calls = self._fake_mod(hosted_result="OCR TEXT")
with patch.object(rx, "_anydoc", return_value=mod), patch.object(rx, "_hosted_ocr_config",
return_value=(True, "key", None)), patch.object(rx.os.path, "getsize", return_value=10):
out = rx._extract_anydoc("scan.pdf")
self.assertEqual(out, "OCR TEXT\n")
self.assertEqual(calls[1].get("ocr"), "hosted")
def test_hosted_failure_warns_and_prefers_local_skills(self):
from tools import read_extract as rx
mod, _ = self._fake_mod(hosted_exc=RuntimeError("HTTP 500"))
with patch.object(rx, "_anydoc", return_value=mod), patch.object(rx, "_hosted_ocr_config",
return_value=(True, "key", "https://gw")), patch.object(rx.os.path, "getsize", return_value=10):
out = rx._extract_anydoc("scan.pdf")
self.assertIn("[NEEDS OCR", out)
self.assertIn("pages 2, 3", out)
self.assertIn("attempted and failed", out)
# Maintainer-directed: HINT at checking for an OCR skill; never
# name one (none is guaranteed to exist), never sell config knobs.
self.assertIn("check whether an OCR skill is available", out)
self.assertIn("skills_list", out)
self.assertNotIn("ocr-and-documents", out)
self.assertNotIn("marker-pdf", out)
self.assertNotIn("hosted_ocr", out)
def test_disabled_warns_without_attempt(self):
from tools import read_extract as rx
mod, calls = self._fake_mod()
with patch.object(rx, "_anydoc", return_value=mod), patch.object(rx, "_hosted_ocr_config",
return_value=(False, None, None)), patch.object(rx.os.path, "getsize", return_value=10):
out = rx._extract_anydoc("scan.pdf")
self.assertIn("[NEEDS OCR", out)
self.assertEqual(len(calls), 1) # no hosted attempt
# Same shape when disabled: skill hint, no knob advertising.
self.assertIn("check whether an OCR skill is available", out)
self.assertNotIn("hosted_ocr", out)
self.assertNotIn("ocr-and-documents", out)
def test_pin_lockstep(self):
"""pyproject core pin and lazy_deps self-heal pin must match."""
import re
from pathlib import Path
py = Path("pyproject.toml").read_text(encoding="utf-8")
lz = Path("tools/lazy_deps.py").read_text(encoding="utf-8")
m1 = re.search(r'"firecrawl-anydoc==([\d.]+)"', py)
m2 = re.search(r'"firecrawl-anydoc==([\d.]+)"', lz)
self.assertIsNotNone(m1)
self.assertIsNotNone(m2)
self.assertEqual(m1.group(1), m2.group(1))
if __name__ == "__main__":
unittest.main()

View File

@@ -2648,7 +2648,15 @@ def _check_file_reqs():
READ_FILE_SCHEMA = {
"name": "read_file",
"description": "Read a text file with line numbers and pagination. Use this instead of cat/head/tail in terminal. Output format: 'LINE_NUM|CONTENT'. Suggests similar filenames if not found. Use offset and limit for large files. Reads exceeding ~100K characters are truncated on a line boundary and return a next_offset; continue with offset to read the rest. Jupyter notebooks (.ipynb), Word documents (.docx), and Excel workbooks (.xlsx) are auto-extracted to readable text; PDF, legacy Office (.doc/.ppt/.xls), OpenDocument, RTF, and EPUB convert too when the optional anydoc converter is available (auto-installed on first use where installs are permitted). PDF conversion reads the text layer only: scanned/image pages yield no text, and when many pages come back empty the output ends with an EXTRACTION COVERAGE WARNING listing the affected pages — follow its instructions (render pages with pdftoppm and inspect via vision_analyze, or OCR) instead of treating the extraction as complete. NOTE: Cannot read images or other binary files — use vision_analyze for images.",
# Document formats are stated unconditionally: firecrawl-anydoc is a
# core dependency (bundled), so its absence is a broken install, not a
# configuration — the teaching error in read_extract handles that rare
# case with the pip-install fix. The ONE dynamic word: "PDF (text
# layer)" upgrades to "PDF (scanned or text)" when hosted OCR has a
# route we trust (_read_file_schema_overrides). Scanned-page coverage
# teaching lives in the response-time NEEDS-OCR warning
# (read_extract.py); the schema doesn't pre-teach it.
"description": "Read a text file with line numbers and pagination. Use this instead of cat/head/tail in terminal. Output format: 'LINE_NUM|CONTENT'. Suggests similar filenames if not found. Use offset and limit for large files. Reads exceeding ~100K characters are truncated on a line boundary and return a next_offset; continue with offset to read the rest. Documents auto-extract to readable text: .ipynb, Office (.docx/.xlsx/.pptx and legacy .doc/.ppt/.xls), PDF (text layer), OpenDocument, RTF, EPUB. Cannot read images/binary — use vision_analyze for images.",
"parameters": {
"type": "object",
"properties": {
@@ -2800,7 +2808,28 @@ def _handle_search_files(args, **kw):
output_mode=args.get("output_mode", "content"), context=args.get("context", 0), task_id=tid)
registry.register(name="read_file", toolset="file", schema=READ_FILE_SCHEMA, handler=_handle_read_file, check_fn=_check_file_reqs, emoji="📖", max_result_size_chars=100_000)
def _read_file_schema_overrides():
"""One-word capability upgrade: "PDF (text layer)" → "PDF (scanned or
text)" when hosted OCR has a trusted route (see
read_extract.hosted_ocr_available). Config/env probe only — no
network at schema-build time. Compaction's tool refresh (#97073)
picks up a key added mid-session.
"""
try:
from tools.read_extract import hosted_ocr_available
if hosted_ocr_available():
return {
"description": READ_FILE_SCHEMA["description"].replace(
"PDF (text layer)", "PDF (scanned or text)"
)
}
except Exception: # noqa: BLE001
pass
return {}
registry.register(name="read_file", toolset="file", schema=READ_FILE_SCHEMA, handler=_handle_read_file, check_fn=_check_file_reqs, emoji="📖", max_result_size_chars=100_000, dynamic_schema_overrides=_read_file_schema_overrides)
registry.register(name="write_file", toolset="file", schema=WRITE_FILE_SCHEMA, handler=_handle_write_file, check_fn=_check_file_reqs, emoji="✍️", max_result_size_chars=100_000)
registry.register(name="patch", toolset="file", schema=PATCH_SCHEMA, handler=_handle_patch, check_fn=_check_file_reqs, emoji="🔧", max_result_size_chars=100_000)
registry.register(name="search_files", toolset="file", schema=SEARCH_FILES_SCHEMA, handler=_handle_search_files, check_fn=_check_file_reqs, emoji="🔎", max_result_size_chars=100_000)

View File

@@ -292,10 +292,10 @@ LAZY_DEPS: dict[str, tuple[str, ...]] = {
# the stdlib .ipynb/.docx/.xlsx to PDF, legacy Office (.doc/.ppt/.xls),
# OpenDocument, RTF, and EPUB. Installed on first read of such a file;
# the call site uses prompt=False so read_file never blocks on a prompt.
# NOTE: lazy-only for now — no pyproject `doc-extract` extra until the
# package clears the uv exclude-newer 14-day quarantine (first release
# 2026-08-04); add the mirrored extra then.
"tool.doc_extract": ("firecrawl-anydoc==0.1.6",),
# NOTE: bundled in core pyproject dependencies since the hosted-OCR
# wiring (keep this lazy pin in lockstep with pyproject) — this entry
# survives as the self-heal path for lean/partial installs.
"tool.doc_extract": ("firecrawl-anydoc==0.2.4",), # lockstep with pyproject
# Computer Use (cua-driver) — the MCP client SDK used to spawn and talk
# to the cua-driver process over stdio. Matches the `mcp` / `computer-use`
# extras in pyproject.toml. The one-liner installer pulls this in via

View File

@@ -162,10 +162,105 @@ def extract_document_bytes(data: bytes, path: str) -> str:
pass
def _anydoc_missing_error(path: str) -> str:
"""Teaching error for anydoc-gated formats when the converter is absent.
Response-time hint (#95681 pattern): the schema no longer lists the
anydoc-gated formats or the availability caveat — a session that never
touches a .doc/.odt/.epub never pays for the explanation, and one that
does gets the full story here, with the fix.
"""
return (
f"Cannot convert {path!r}: this format needs the optional anydoc "
"converter, which is not installed (install blocked or first "
"attempt failed; retried every 5 minutes). Fix: `pip install "
"firecrawl-anydoc` in Hermes's environment, or convert the file "
"yourself via terminal (e.g. libreoffice --headless --convert-to "
"txt)."
)
def _hosted_ocr_config() -> tuple:
"""Resolve hosted-OCR settings: (enabled, api_key, api_url).
Maintainer decision: the ONLY route is a direct ``FIRECRAWL_API_KEY``
(anydoc defaults api_url to https://api.firecrawl.dev). The Nous
managed gateway is NOT used — its Parse proxy was live-probed broken
(uniform HTTP 500, 2026-08-28) while scrape/search worked; revisit
when the gateway grows Parse support. ``file_tools.hosted_ocr``:
false disables even with a key; true/unset → enabled iff key
present. Never raises.
"""
api_key = os.environ.get("FIRECRAWL_API_KEY") or None
enabled = api_key is not None
try:
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
section = cfg.get("file_tools") if isinstance(cfg, dict) else None
if isinstance(section, dict) and section.get("hosted_ocr") is False:
enabled = False
except Exception: # noqa: BLE001
pass
return enabled, api_key, None
def hosted_ocr_available() -> bool:
"""Public probe for read_file's schema line: is hosted OCR unlocked?
Maintainer decision: ONE gate — a direct ``FIRECRAWL_API_KEY`` in the
environment. Nothing else unlocks the "PDF (scanned or text)" wording
(not the Nous gateway — Parse proxy live-probed broken 2026-08-28 —
and not config assertions). ``file_tools.hosted_ocr: false`` still
disables. Env probe only — no network at schema-build time; a key
that fails at conversion time lands in the NEEDS-OCR warning.
"""
try:
if not os.environ.get("FIRECRAWL_API_KEY"):
return False
try:
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
section = cfg.get("file_tools") if isinstance(cfg, dict) else None
if isinstance(section, dict) and section.get("hosted_ocr") is False:
return False
except Exception: # noqa: BLE001
pass
return True
except Exception: # noqa: BLE001
return False
def _needs_ocr_warning(path: str, pages, hosted_error: str = "") -> str:
"""Typed replacement for the heuristic coverage note on full-OCR PDFs.
Fired when anydoc raises NeedsOcrError and hosted OCR is disabled,
unavailable, or failed. Maintainer-directed shape: hint at CHECKING
for an OCR skill (never name one — none is guaranteed to exist), and
never advertise the hosted_ocr config knob — when hosted fails or is
absent, a skill or ignoring the gap are the paths that exist.
"""
page_list = ", ".join(str(p) for p in pages) if pages else "unknown"
msg = (
f"[NEEDS OCR: pages {page_list} of this PDF are scanned images "
"with no text layer — their content is MISSING below. "
)
if hosted_error:
msg += f"Hosted OCR was attempted and failed ({hosted_error}). "
msg += (
"If the missing pages matter: render just those pages with "
f"`pdftoppm -jpeg -r 150 -f <first> -l <last> '{path}' /tmp/page` "
"and inspect via vision_analyze, or check whether an OCR skill is "
"available (skills_list)."
)
return msg + "]\n"
def _extract_anydoc(path: str) -> str:
mod = _anydoc()
if mod is None:
raise ExtractionError(f"Unsupported document type: {path!r}")
raise ExtractionError(_anydoc_missing_error(path))
try:
size = os.path.getsize(path)
except OSError as exc:
@@ -174,11 +269,32 @@ def _extract_anydoc(path: str) -> str:
raise ExtractionError(
f"Document too large to convert ({size:,} bytes, limit is {MAX_ANYDOC_BYTES:,})"
)
needs_ocr = getattr(mod, "NeedsOcrError", None)
try:
text = mod.to_markdown(path)
except OSError as exc:
raise ExtractionError(str(exc)) from exc
except Exception as exc:
if needs_ocr is not None and isinstance(exc, needs_ocr):
# Typed scanned-pages signal (anydoc >= 0.2). Try hosted OCR
# when a Firecrawl route exists; otherwise teach recovery.
pages = list(getattr(exc, "pages", []) or [])
enabled, api_key, api_url = _hosted_ocr_config()
hosted_error = ""
if enabled:
try:
kwargs = {"ocr": "hosted"}
if api_key:
kwargs["api_key"] = api_key
if api_url:
kwargs["api_url"] = api_url
text = mod.to_markdown(path, **kwargs)
return text.rstrip("\n") + "\n"
except Exception as hosted_exc: # noqa: BLE001
hosted_error = f"{type(hosted_exc).__name__}: {hosted_exc}"
# No route / disabled / hosted failed: whole doc is scans —
# nothing to extract, so the warning IS the result.
return _needs_ocr_warning(path, pages, hosted_error)
# anydoc raises one ConvertError subclass per failure mode
# (Unsupported, Malformed, Encrypted, ResourceLimit, MissingPart).
# Any of them means "no meaningful text": fall back to the normal
@@ -192,6 +308,9 @@ def _extract_anydoc(path: str) -> str:
if note:
# Prepend: read_file paginates the extraction, so a footer on a
# long document would sit on a page the model may never fetch.
# This heuristic note survives for PARTIAL coverage gaps —
# documents with a text layer plus some scanned pages, which
# convert without raising NeedsOcrError.
text = note + text
return text
@@ -335,7 +454,7 @@ def _pdf_coverage_note(path: str, display_path: Optional[str] = None) -> str:
def _extract_anydoc_bytes(data: bytes, path: str) -> str:
mod = _anydoc()
if mod is None:
raise ExtractionError(f"Unsupported document type: {path!r}")
raise ExtractionError(_anydoc_missing_error(path))
if len(data) > MAX_ANYDOC_BYTES:
raise ExtractionError(
f"Document too large to convert ({len(data):,} bytes, limit is {MAX_ANYDOC_BYTES:,})"

36
uv.lock generated
View File

@@ -13,17 +13,18 @@ exclude-newer-span = "P14D"
[options.exclude-newer-package]
setuptools = false
h2 = false
huggingface-hub = false
vercel = false
pillow = false
unpaddedbase64 = false
h2 = false
defusedxml = false
aiohttp = false
cryptography = false
mcp = false
python-olm = false
firecrawl-anydoc = false
nemo-relay = false
huggingface-hub = false
[manifest]
overrides = [
@@ -1268,6 +1269,21 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/e5/4c/93d0f85318da65923e4b91c1c2ff03d8a458cbefebe3bc612a6693c7906d/fire-0.7.1-py3-none-any.whl", hash = "sha256:e43fd8a5033a9001e7e2973bab96070694b9f12f2e0ecf96d4683971b5ab1882", size = 115945, upload-time = "2025-08-16T20:20:22.87Z" },
]
[[package]]
name = "firecrawl-anydoc"
version = "0.2.4"
source = { registry = "https://pypi.org/simple" }
sdist = { url = "https://files.pythonhosted.org/packages/35/54/ee43b1661a954e53eaed85878a7ccc37ed73e02e6577bb26437ec5f9de94/firecrawl_anydoc-0.2.4.tar.gz", hash = "sha256:3e29460272fea81cde08fd5af11f6b0f1ff05919214ddc939867f72362c83032", size = 270706, upload-time = "2026-08-27T19:13:58.156Z" }
wheels = [
{ url = "https://files.pythonhosted.org/packages/25/f1/5616b8d703b2bf5688eda86ede4b35620356d9c60346fe2d3efbec2a872d/firecrawl_anydoc-0.2.4-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:e3745a8b108d6575beafd37bcefe52e3f6c7ab44138e05cbdf4c6adf63caba84", size = 3441281, upload-time = "2026-08-27T19:13:43.408Z" },
{ url = "https://files.pythonhosted.org/packages/59/7d/4b2a466c8953af3b9d94471081b04ce8636680111402d2140105f98c02e5/firecrawl_anydoc-0.2.4-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:b2a01eecaa0b99f99ff6749a098f8c7a5250d41db9c53673eb93536ad575bfed", size = 3262365, upload-time = "2026-08-27T19:13:46.362Z" },
{ url = "https://files.pythonhosted.org/packages/77/b5/5784fc514f58ca19328faee701d2cc2e25961389c2d4908b2a51a1071ee4/firecrawl_anydoc-0.2.4-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:17720ba093820654162891d7995fd2dbae948b16fe1c174ba799bddfc0b92ad6", size = 3265363, upload-time = "2026-08-27T19:13:48.425Z" },
{ url = "https://files.pythonhosted.org/packages/fc/16/feeca9705bfdb237f1cb69ede0b373b144c0d51df4297e595a74b815557e/firecrawl_anydoc-0.2.4-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:0e5aed01bf6d4e5c588d3363888293f2ceeb9feb00449a32e0ba993797cf0bd3", size = 3506751, upload-time = "2026-08-27T19:13:50.372Z" },
{ url = "https://files.pythonhosted.org/packages/14/28/b4d0d3c5345cdd65c0b9113c6e6ff1b717fe360dccb658db2a9952315f9c/firecrawl_anydoc-0.2.4-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:dd162b5eb64478fcf02f53abe7765e26f4350ba657f36c25f055c4cc3dbade0a", size = 3443075, upload-time = "2026-08-27T19:13:52.417Z" },
{ url = "https://files.pythonhosted.org/packages/cb/29/50c52b0ea5dd645aea6d195da913c8d9e136601eec465bd9a590edd250c7/firecrawl_anydoc-0.2.4-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:e57c79093c1592929dd8dfaf6c60b49eefe631721d96fb833c3f465472b37b9f", size = 3742860, upload-time = "2026-08-27T19:13:54.474Z" },
{ url = "https://files.pythonhosted.org/packages/65/34/6d54fed906055e5507dd94efd63f4a05021f103dc684ea62a3b0876b7595/firecrawl_anydoc-0.2.4-cp310-abi3-win_amd64.whl", hash = "sha256:7c9ffc81946e6954c124560ed1d395e0103a1a08fdc0f6be382e2f35daf6dbc9", size = 3583555, upload-time = "2026-08-27T19:13:56.698Z" },
]
[[package]]
name = "firecrawl-py"
version = "4.17.0"
@@ -1578,6 +1594,7 @@ dependencies = [
{ name = "cryptography" },
{ name = "fastapi" },
{ name = "fire" },
{ name = "firecrawl-anydoc" },
{ name = "httpx", extra = ["socks"] },
{ name = "jinja2" },
{ name = "markdown" },
@@ -1844,6 +1861,7 @@ requires-dist = [
{ name = "fastapi", marker = "extra == 'web'", specifier = "==0.133.1" },
{ name = "faster-whisper", marker = "extra == 'voice'", specifier = "==1.2.1" },
{ name = "fire", specifier = "==0.7.1" },
{ name = "firecrawl-anydoc", specifier = "==0.2.4" },
{ name = "firecrawl-py", marker = "extra == 'firecrawl'", specifier = "==4.17.0" },
{ name = "google-api-python-client", marker = "extra == 'google'", specifier = "==2.194.0" },
{ name = "google-auth", marker = "extra == 'google'", specifier = "==2.55.1" },
@@ -4052,7 +4070,7 @@ resolution-markers = [
"python_full_version < '3.12'",
]
dependencies = [
{ name = "numpy", marker = "python_full_version < '3.12'" },
{ name = "numpy" },
]
sdist = { url = "https://files.pythonhosted.org/packages/7a/97/5a3609c4f8d58b039179648e62dd220f89864f56f7357f5d4f45c29eb2cc/scipy-1.17.1.tar.gz", hash = "sha256:95d8e012d8cb8816c226aef832200b1d45109ed4464303e997c5b13122b297c0", size = 30573822, upload-time = "2026-02-23T00:26:24.851Z" }
wheels = [
@@ -4107,7 +4125,7 @@ resolution-markers = [
"python_full_version == '3.12.*'",
]
dependencies = [
{ name = "numpy", marker = "python_full_version >= '3.12'" },
{ name = "numpy" },
]
sdist = { url = "https://files.pythonhosted.org/packages/a7/25/c2700dfaf6442b4effaa91af24ebce5dc9d31bb4a69706313aae70d72cd0/scipy-1.18.0.tar.gz", hash = "sha256:67b2ad2ad54c72ca6d04975a9b2df8c3638c34ddd5b28738e94fc2b57929d378", size = 30774447, upload-time = "2026-06-19T15:01:43.456Z" }
wheels = [
@@ -4755,11 +4773,11 @@ name = "vercel-workers"
version = "0.0.25"
source = { registry = "https://pypi.org/simple" }
dependencies = [
{ name = "anyio", marker = "python_full_version >= '3.12'" },
{ name = "httpx", marker = "python_full_version >= '3.12'" },
{ name = "pydantic", marker = "python_full_version >= '3.12'" },
{ name = "python-dotenv", marker = "python_full_version >= '3.12'" },
{ name = "vercel", marker = "python_full_version >= '3.12'" },
{ name = "anyio" },
{ name = "httpx" },
{ name = "pydantic" },
{ name = "python-dotenv" },
{ name = "vercel" },
]
sdist = { url = "https://files.pythonhosted.org/packages/30/df/04d37021ad7ca53b7599c313e411d91623c7a005c741f491d1eefb7a9f0c/vercel_workers-0.0.25.tar.gz", hash = "sha256:212ded01400b524be51d251df49f801caf115ad7d48cca7eb168cbeceda3def3", size = 64149, upload-time = "2026-06-20T19:26:27.177Z" }
wheels = [