simplify(compat): web_tools — drop 35 re-exports/aliases, repoint 2 callers + 6 test files

This commit is contained in:
Teknium
2026-09-03 13:08:53 -07:00
parent 00a3cfe5c9
commit 2ed36eff40
10 changed files with 45 additions and 43 deletions

View File

@@ -13,6 +13,8 @@ from typing import Any, Dict, List, Optional
import httpx
from plugins.web._common import BaseWebSearchProvider, keyless_extract, keyless_search, lazy_ensure, search_fail, search_ok, setup_schema
from tools import managed_tool_gateway as _gateway
from tools import tool_backend_helpers as _backend_helpers
from tools.url_safety import is_safe_url
# Module-level (cheap import) so tests can monkeypatch the policy gate on this module.
from tools.website_policy import check_website_access
@@ -21,8 +23,7 @@ logger = logging.getLogger(__name__)
_FIRECRAWL_CLOUD_API_URL = "https://api.firecrawl.dev"
# The SDK costs ~200ms of imports on a cold CLI; defer to first use. tools.web_tools
# re-exports ``Firecrawl`` so ``patch("tools.web_tools.Firecrawl")`` keeps working.
# The SDK costs ~200ms of imports on a cold CLI; defer to first use (tests patch ``Firecrawl`` here).
_FIRECRAWL_CLS_CACHE: Optional[type] = None
@@ -55,8 +56,7 @@ Firecrawl = _FirecrawlProxy()
# --- Client construction (direct vs managed-gateway) ---------------------------
def _wt():
"""Client cache slots and gateway/token helpers are read through tools.web_tools so tests
that reset ``_firecrawl_client`` or patch ``_peek_nous_access_token`` there see their changes."""
"""Client cache slots live on tools.web_tools so tests that reset ``_firecrawl_client`` there see it."""
import tools.web_tools as _mod
return _mod
@@ -92,7 +92,7 @@ def _use_keyless_ring() -> bool:
from tools.tool_backend_helpers import NOUS_MANAGED_PROVIDER, read_selection
from plugins.web.keyless_mcp import use_keyless
# Both probes are optional layers: a failing probe never blocks the ring.
for probe in (lambda: read_selection("web") == NOUS_MANAGED_PROVIDER, lambda: _wt()._is_tool_gateway_ready() and not _is_explicit_firecrawl_selection()):
for probe in (lambda: read_selection("web") == NOUS_MANAGED_PROVIDER, lambda: _is_tool_gateway_ready() and not _is_explicit_firecrawl_selection()):
try:
if probe():
return False
@@ -118,17 +118,17 @@ class _KeylessFirecrawlClient:
def _get_firecrawl_gateway_url() -> str:
return _wt().build_vendor_gateway_url("firecrawl")
return _gateway.build_vendor_gateway_url("firecrawl")
def _is_tool_gateway_ready() -> bool:
"""True when gateway URL + Nous Subscriber token are available."""
return _wt().resolve_managed_tool_gateway("firecrawl", token_reader=_wt()._peek_nous_access_token) is not None
return _gateway.resolve_managed_tool_gateway("firecrawl", token_reader=_gateway.peek_nous_access_token) is not None
def check_firecrawl_api_key() -> bool:
"""True when the route selected via ``hermes tools`` (or, on a never-configured
install, either route) is usable. Re-exported by tools.web_tools."""
install, either route) is usable."""
from tools.tool_backend_helpers import NOUS_MANAGED_PROVIDER, read_selection
selected = read_selection("web")
if selected == NOUS_MANAGED_PROVIDER:
@@ -137,7 +137,7 @@ def check_firecrawl_api_key() -> bool:
def _firecrawl_backend_help_suffix() -> str:
return ", or use the Nous Tool Gateway via your subscription (FIRECRAWL_GATEWAY_URL or TOOL_GATEWAY_DOMAIN)" if _wt().managed_nous_tools_enabled() else ""
return ", or use the Nous Tool Gateway via your subscription (FIRECRAWL_GATEWAY_URL or TOOL_GATEWAY_DOMAIN)" if _backend_helpers.managed_nous_tools_enabled() else ""
def _get_firecrawl_client() -> Any:
@@ -151,16 +151,16 @@ def _get_firecrawl_client() -> Any:
direct_config = _get_direct_firecrawl_config()
def _managed():
gw = wt.resolve_managed_tool_gateway("firecrawl", token_reader=wt._read_nous_access_token)
gw = _gateway.resolve_managed_tool_gateway("firecrawl", token_reader=_gateway.read_nous_access_token)
if gw is None:
return None
return "sdk", {"api_key": gw.nous_user_token, "api_url": gw.gateway_origin}, ("tool-gateway", gw.gateway_origin, gw.nous_user_token)
def _unconfigured_message() -> str:
message = "Web tools are not configured. Set FIRECRAWL_API_KEY for cloud Firecrawl or set FIRECRAWL_API_URL for a self-hosted Firecrawl instance."
if wt.managed_nous_tools_enabled():
if _backend_helpers.managed_nous_tools_enabled():
return message + " With your Nous subscription you can also use the Tool Gateway. run `hermes tools` and select Nous Subscription as the web provider."
return message + " " + wt.nous_tool_gateway_unavailable_message("managed Firecrawl web tools")
return message + " " + _backend_helpers.nous_tool_gateway_unavailable_message("managed Firecrawl web tools")
# (resolved config, log detail, error message) per selection state; the message is built lazily.
if selected == NOUS_MANAGED_PROVIDER:
@@ -181,7 +181,7 @@ def _get_firecrawl_client() -> Any:
cached = getattr(wt, "_firecrawl_client", None)
if cached is not None and getattr(wt, "_firecrawl_client_config", None) == client_config:
return cached
wt._firecrawl_client = _KeylessFirecrawlClient(api_url=kwargs["api_url"]) if client_mode == "keyless" else wt.Firecrawl(**kwargs)
wt._firecrawl_client = _KeylessFirecrawlClient(api_url=kwargs["api_url"]) if client_mode == "keyless" else Firecrawl(**kwargs)
wt._firecrawl_client_config = client_config
return wt._firecrawl_client

View File

@@ -10,26 +10,27 @@ from __future__ import annotations
import re
import tools.web_tools as wt
from tools import web_tools_truncate
def test_store_full_text_is_bounded(tmp_path, monkeypatch):
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
# Force the cache dir under the temp home.
from hermes_constants import get_hermes_dir # noqa: F401
huge = "x\n" * (wt.MAX_STORED_TEXT_CHARS) # > MAX_STORED_TEXT_CHARS chars
assert len(huge) > wt.MAX_STORED_TEXT_CHARS
path = wt._store_full_text("https://example.com/big", huge)
huge = "x\n" * (web_tools_truncate.MAX_STORED_TEXT_CHARS) # > MAX_STORED_TEXT_CHARS chars
assert len(huge) > web_tools_truncate.MAX_STORED_TEXT_CHARS
path = web_tools_truncate._store_full_text("https://example.com/big", huge)
assert path is not None
stored = open(path, encoding="utf-8").read()
# Stored copy capped (+ short marker), not the full unbounded blob.
assert len(stored) <= wt.MAX_STORED_TEXT_CHARS + 200
assert len(stored) <= web_tools_truncate.MAX_STORED_TEXT_CHARS + 200
assert "stored copy truncated" in stored
def test_small_page_not_truncated_no_footer(tmp_path, monkeypatch):
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
content = "short page\nwith a few lines\n"
model_text, truncated = wt._truncate_with_footer(
model_text, truncated = web_tools_truncate._truncate_with_footer(
content, "https://example.com/s", char_limit=15000
)
assert not truncated

View File

@@ -19,6 +19,7 @@ from unittest.mock import patch
import pytest
import tools.web_tools as web_tools
from tools import web_tools_rescue
from plugins.web import keyless_mcp
from plugins.web.keenable.provider import KeenableWebSearchProvider
@@ -73,33 +74,33 @@ def _ring_ok(vendor="exa"):
class TestEligibility:
def test_keyed_ring_vendor_is_eligible(self):
assert web_tools._rescue_eligible(_KeyedBoomProvider()) is True
assert web_tools_rescue._rescue_eligible(_KeyedBoomProvider()) is True
def test_keyless_mode_ring_vendor_not_eligible(self, monkeypatch):
# No key: the keenable call already rode the ring; no double-walk.
monkeypatch.setattr(
"agent.web_search_provider.get_provider_env", lambda name: ""
)
assert web_tools._rescue_eligible(KeenableWebSearchProvider()) is False
assert web_tools_rescue._rescue_eligible(KeenableWebSearchProvider()) is False
def test_non_ring_backend_is_eligible(self):
class _SearxProvider(_KeyedBoomProvider):
name = "searxng"
assert web_tools._rescue_eligible(_SearxProvider()) is True
assert web_tools_rescue._rescue_eligible(_SearxProvider()) is True
def test_config_gate_disables(self, monkeypatch):
monkeypatch.setattr(
web_tools, "_load_web_config",
lambda: {"backend": "keenable", "keyless_rescue": False},
)
assert web_tools._rescue_eligible(_KeyedBoomProvider()) is False
assert web_tools_rescue._rescue_eligible(_KeyedBoomProvider()) is False
def test_keyless_fallback_off_disables(self, monkeypatch):
monkeypatch.setattr(
"agent.web_search_registry._keyless_tier_enabled", lambda: False
)
assert web_tools._rescue_eligible(_KeyedBoomProvider()) is False
assert web_tools_rescue._rescue_eligible(_KeyedBoomProvider()) is False
class TestSearchRescue:
@@ -211,7 +212,7 @@ class TestExtractRescue:
with patch.object(
keyless_mcp, "extract_with_failover", return_value=good
):
out = web_tools._rescue_extract("keenable", ["https://a"], failed)
out = web_tools_rescue._rescue_extract("keenable", ["https://a"], failed)
assert out[0]["metadata"]["rescued_from"] == "keenable"
assert "HTTP 500" in out[0]["metadata"]["backend_error"]

View File

@@ -190,7 +190,7 @@ def test_extract_cache_provider_participates_in_key(_isolated_cache):
def test_extract_cache_oversized_page_not_indexed(_isolated_cache):
import tools.web_tools as wt
from tools import web_tools_truncate as wt
big = "x" * (wt.MAX_STORED_TEXT_CHARS + 1)
extract_cache_put("https://big.com", big)
assert extract_cache_get("https://big.com") is None

View File

@@ -92,7 +92,7 @@ class TestNormalizeTavilySearchResults:
"""Test search result normalization."""
def test_basic_normalization(self):
from tools.web_tools import _normalize_tavily_search_results
from plugins.web.tavily.provider import _normalize_tavily_search_results
raw = {
"results": [
{"title": "Python Docs", "url": "https://docs.python.org", "content": "Official docs", "score": 0.9},
@@ -111,7 +111,7 @@ class TestNormalizeTavilySearchResults:
def test_missing_fields(self):
from tools.web_tools import _normalize_tavily_search_results
from plugins.web.tavily.provider import _normalize_tavily_search_results
result = _normalize_tavily_search_results({"results": [{}]})
web = result["data"]["web"]
assert web[0]["title"] == ""
@@ -125,7 +125,7 @@ class TestNormalizeTavilyDocuments:
"""Test extract document normalization."""
def test_basic_document(self):
from tools.web_tools import _normalize_tavily_documents
from plugins.web.tavily.provider import _normalize_tavily_documents
raw = {
"results": [{
"url": "https://example.com",
@@ -143,7 +143,7 @@ class TestNormalizeTavilyDocuments:
def test_fallback_url(self):
from tools.web_tools import _normalize_tavily_documents
from plugins.web.tavily.provider import _normalize_tavily_documents
raw = {"results": [{"content": "data"}]}
docs = _normalize_tavily_documents(raw, fallback_url="https://fallback.com")
assert docs[0]["url"] == "https://fallback.com"

View File

@@ -12,13 +12,14 @@ from unittest.mock import patch
import pytest
import tools.web_tools as wt
from tools import web_tools_truncate
class TestImageConversion:
def test_markdown_base64_image_keeps_alt_drops_blob(self):
blob = "A" * 5000
text = f"before ![a cat]( data:image/png;base64,{blob}) after"
out = wt.convert_base64_images_to_links(text)
out = web_tools_truncate.convert_base64_images_to_links(text)
assert "[IMAGE: a cat]" in out
assert "base64" not in out
assert blob not in out
@@ -27,16 +28,16 @@ class TestImageConversion:
def test_bare_and_parenthesised_base64_become_placeholder(self):
blob = "Z" * 3000
bare = wt.convert_base64_images_to_links(f"data:image/gif;base64,{blob}")
bare = web_tools_truncate.convert_base64_images_to_links(f"data:image/gif;base64,{blob}")
assert bare == "[IMAGE]"
paren = wt.convert_base64_images_to_links(f"(data:image/gif;base64,{blob})")
paren = web_tools_truncate.convert_base64_images_to_links(f"(data:image/gif;base64,{blob})")
assert paren == "[IMAGE]"
class TestTruncation:
def test_short_content_returned_whole(self):
content = "# Title\n\nshort body\n"
out, truncated = wt._truncate_with_footer(content, "https://e.com", 15000)
out, truncated = web_tools_truncate._truncate_with_footer(content, "https://e.com", 15000)
assert out == content
assert truncated is False
@@ -44,7 +45,7 @@ class TestTruncation:
def test_truncation_stores_full_text_readable(self, tmp_path, monkeypatch):
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
body = "UNIQUE_MIDDLE_MARKER\n" + ("\n".join(f"row {i}" for i in range(5000)))
out, truncated = wt._truncate_with_footer(body, "https://example.com/doc", 3000)
out, truncated = web_tools_truncate._truncate_with_footer(body, "https://example.com/doc", 3000)
assert truncated is True
# Extract the stored path from the footer and confirm full text is there.
path_line = next(ln for ln in out.splitlines() if "Full text saved to:" in ln)
@@ -58,12 +59,12 @@ class TestTruncation:
class TestCharLimitConfig:
def test_default_when_unset(self):
with patch("tools.web_tools._load_web_config", return_value={}):
assert wt._get_extract_char_limit() == wt.DEFAULT_EXTRACT_CHAR_LIMIT
assert web_tools_truncate._get_extract_char_limit() == web_tools_truncate.DEFAULT_EXTRACT_CHAR_LIMIT
def test_bad_value_falls_back(self):
with patch("tools.web_tools._load_web_config", return_value={"extract_char_limit": "nope"}):
assert wt._get_extract_char_limit() == wt.DEFAULT_EXTRACT_CHAR_LIMIT
assert web_tools_truncate._get_extract_char_limit() == web_tools_truncate.DEFAULT_EXTRACT_CHAR_LIMIT
class TestEndToEnd:

View File

@@ -288,7 +288,7 @@ def extract_cache_put(
if not content or not _cacheable(url):
return
try:
from tools.web_tools import MAX_STORED_TEXT_CHARS
from tools.web_tools_truncate import MAX_STORED_TEXT_CHARS
file_path = _entry_file_path(url, format, provider)
if len(content) > MAX_STORED_TEXT_CHARS or file_path is None:
return

View File

@@ -3,8 +3,7 @@
Order of controls (each is a gate, never skipped by a cache hit): secret-URL
refusal -> SSRF filter (in web_tools.web_extract_tool) -> provider resolution
(strict selection) -> per-URL website policy -> disk cache -> vendor call with
one-shot keyless rescue. Names are re-imported by tools/web_tools.py so
``tools.web_tools._validate_extract_urls`` etc. keep working; logs under the origin logger.
one-shot keyless rescue. Logs under the origin (tools.web_tools) logger.
"""
import asyncio

View File

@@ -2,8 +2,8 @@
Stateless by design: a rescue routes THIS call through the free-tier ring (plugins/web/keyless_mcp.py);
the next web_search/web_extract call attempts the chosen backend again. Callers must never cache a
rescue-served response, or the one-shot rescue becomes sticky for a whole TTL. Names are re-imported by
tools/web_tools.py (``tools.web_tools._rescue_eligible``); logs under the origin logger.
rescue-served response, or the one-shot rescue becomes sticky for a whole TTL. Logs under the origin
(tools.web_tools) logger.
"""
import logging

View File

@@ -2,8 +2,8 @@
Pages at or under the char budget are returned whole; larger pages become a head+tail window plus a
footer that says how much is shown, where the full text is stored (cache/web) and the exact read_file
call that pages the omitted middle. Inline base64 images become ``[IMAGE: alt]`` placeholders. Names are
re-imported by tools/web_tools.py (``tools.web_tools.MAX_STORED_TEXT_CHARS``); logs under the origin logger.
call that pages the omitted middle. Inline base64 images become ``[IMAGE: alt]`` placeholders. Logs under the
origin (tools.web_tools) logger.
"""
import logging