From 2ed36eff403d6dd62e34d53faacaa6097dba0b08 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Thu, 3 Sep 2026 13:08:53 -0700 Subject: [PATCH] =?UTF-8?q?simplify(compat):=20web=5Ftools=20=E2=80=94=20d?= =?UTF-8?q?rop=2035=20re-exports/aliases,=20repoint=202=20callers=20+=206?= =?UTF-8?q?=20test=20files?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- plugins/web/firecrawl/provider.py | 26 +++++++++++----------- tests/tools/test_web_extract_robustness.py | 11 ++++----- tests/tools/test_web_keyless_rescue.py | 13 ++++++----- tests/tools/test_web_result_cache.py | 2 +- tests/tools/test_web_tools_tavily.py | 8 +++---- tests/tools/test_web_tools_truncate.py | 15 +++++++------ tools/web_result_cache.py | 2 +- tools/web_tools_extract.py | 3 +-- tools/web_tools_rescue.py | 4 ++-- tools/web_tools_truncate.py | 4 ++-- 10 files changed, 45 insertions(+), 43 deletions(-) diff --git a/plugins/web/firecrawl/provider.py b/plugins/web/firecrawl/provider.py index a1e42503b2..25f024eb20 100644 --- a/plugins/web/firecrawl/provider.py +++ b/plugins/web/firecrawl/provider.py @@ -13,6 +13,8 @@ from typing import Any, Dict, List, Optional import httpx from plugins.web._common import BaseWebSearchProvider, keyless_extract, keyless_search, lazy_ensure, search_fail, search_ok, setup_schema +from tools import managed_tool_gateway as _gateway +from tools import tool_backend_helpers as _backend_helpers from tools.url_safety import is_safe_url # Module-level (cheap import) so tests can monkeypatch the policy gate on this module. from tools.website_policy import check_website_access @@ -21,8 +23,7 @@ logger = logging.getLogger(__name__) _FIRECRAWL_CLOUD_API_URL = "https://api.firecrawl.dev" -# The SDK costs ~200ms of imports on a cold CLI; defer to first use. tools.web_tools -# re-exports ``Firecrawl`` so ``patch("tools.web_tools.Firecrawl")`` keeps working. +# The SDK costs ~200ms of imports on a cold CLI; defer to first use (tests patch ``Firecrawl`` here). _FIRECRAWL_CLS_CACHE: Optional[type] = None @@ -55,8 +56,7 @@ Firecrawl = _FirecrawlProxy() # --- Client construction (direct vs managed-gateway) --------------------------- def _wt(): - """Client cache slots and gateway/token helpers are read through tools.web_tools so tests - that reset ``_firecrawl_client`` or patch ``_peek_nous_access_token`` there see their changes.""" + """Client cache slots live on tools.web_tools so tests that reset ``_firecrawl_client`` there see it.""" import tools.web_tools as _mod return _mod @@ -92,7 +92,7 @@ def _use_keyless_ring() -> bool: from tools.tool_backend_helpers import NOUS_MANAGED_PROVIDER, read_selection from plugins.web.keyless_mcp import use_keyless # Both probes are optional layers: a failing probe never blocks the ring. - for probe in (lambda: read_selection("web") == NOUS_MANAGED_PROVIDER, lambda: _wt()._is_tool_gateway_ready() and not _is_explicit_firecrawl_selection()): + for probe in (lambda: read_selection("web") == NOUS_MANAGED_PROVIDER, lambda: _is_tool_gateway_ready() and not _is_explicit_firecrawl_selection()): try: if probe(): return False @@ -118,17 +118,17 @@ class _KeylessFirecrawlClient: def _get_firecrawl_gateway_url() -> str: - return _wt().build_vendor_gateway_url("firecrawl") + return _gateway.build_vendor_gateway_url("firecrawl") def _is_tool_gateway_ready() -> bool: """True when gateway URL + Nous Subscriber token are available.""" - return _wt().resolve_managed_tool_gateway("firecrawl", token_reader=_wt()._peek_nous_access_token) is not None + return _gateway.resolve_managed_tool_gateway("firecrawl", token_reader=_gateway.peek_nous_access_token) is not None def check_firecrawl_api_key() -> bool: """True when the route selected via ``hermes tools`` (or, on a never-configured - install, either route) is usable. Re-exported by tools.web_tools.""" + install, either route) is usable.""" from tools.tool_backend_helpers import NOUS_MANAGED_PROVIDER, read_selection selected = read_selection("web") if selected == NOUS_MANAGED_PROVIDER: @@ -137,7 +137,7 @@ def check_firecrawl_api_key() -> bool: def _firecrawl_backend_help_suffix() -> str: - return ", or use the Nous Tool Gateway via your subscription (FIRECRAWL_GATEWAY_URL or TOOL_GATEWAY_DOMAIN)" if _wt().managed_nous_tools_enabled() else "" + return ", or use the Nous Tool Gateway via your subscription (FIRECRAWL_GATEWAY_URL or TOOL_GATEWAY_DOMAIN)" if _backend_helpers.managed_nous_tools_enabled() else "" def _get_firecrawl_client() -> Any: @@ -151,16 +151,16 @@ def _get_firecrawl_client() -> Any: direct_config = _get_direct_firecrawl_config() def _managed(): - gw = wt.resolve_managed_tool_gateway("firecrawl", token_reader=wt._read_nous_access_token) + gw = _gateway.resolve_managed_tool_gateway("firecrawl", token_reader=_gateway.read_nous_access_token) if gw is None: return None return "sdk", {"api_key": gw.nous_user_token, "api_url": gw.gateway_origin}, ("tool-gateway", gw.gateway_origin, gw.nous_user_token) def _unconfigured_message() -> str: message = "Web tools are not configured. Set FIRECRAWL_API_KEY for cloud Firecrawl or set FIRECRAWL_API_URL for a self-hosted Firecrawl instance." - if wt.managed_nous_tools_enabled(): + if _backend_helpers.managed_nous_tools_enabled(): return message + " With your Nous subscription you can also use the Tool Gateway. run `hermes tools` and select Nous Subscription as the web provider." - return message + " " + wt.nous_tool_gateway_unavailable_message("managed Firecrawl web tools") + return message + " " + _backend_helpers.nous_tool_gateway_unavailable_message("managed Firecrawl web tools") # (resolved config, log detail, error message) per selection state; the message is built lazily. if selected == NOUS_MANAGED_PROVIDER: @@ -181,7 +181,7 @@ def _get_firecrawl_client() -> Any: cached = getattr(wt, "_firecrawl_client", None) if cached is not None and getattr(wt, "_firecrawl_client_config", None) == client_config: return cached - wt._firecrawl_client = _KeylessFirecrawlClient(api_url=kwargs["api_url"]) if client_mode == "keyless" else wt.Firecrawl(**kwargs) + wt._firecrawl_client = _KeylessFirecrawlClient(api_url=kwargs["api_url"]) if client_mode == "keyless" else Firecrawl(**kwargs) wt._firecrawl_client_config = client_config return wt._firecrawl_client diff --git a/tests/tools/test_web_extract_robustness.py b/tests/tools/test_web_extract_robustness.py index 120cee0cf6..9404d01fbb 100644 --- a/tests/tools/test_web_extract_robustness.py +++ b/tests/tools/test_web_extract_robustness.py @@ -10,26 +10,27 @@ from __future__ import annotations import re import tools.web_tools as wt +from tools import web_tools_truncate def test_store_full_text_is_bounded(tmp_path, monkeypatch): monkeypatch.setenv("HERMES_HOME", str(tmp_path)) # Force the cache dir under the temp home. from hermes_constants import get_hermes_dir # noqa: F401 - huge = "x\n" * (wt.MAX_STORED_TEXT_CHARS) # > MAX_STORED_TEXT_CHARS chars - assert len(huge) > wt.MAX_STORED_TEXT_CHARS - path = wt._store_full_text("https://example.com/big", huge) + huge = "x\n" * (web_tools_truncate.MAX_STORED_TEXT_CHARS) # > MAX_STORED_TEXT_CHARS chars + assert len(huge) > web_tools_truncate.MAX_STORED_TEXT_CHARS + path = web_tools_truncate._store_full_text("https://example.com/big", huge) assert path is not None stored = open(path, encoding="utf-8").read() # Stored copy capped (+ short marker), not the full unbounded blob. - assert len(stored) <= wt.MAX_STORED_TEXT_CHARS + 200 + assert len(stored) <= web_tools_truncate.MAX_STORED_TEXT_CHARS + 200 assert "stored copy truncated" in stored def test_small_page_not_truncated_no_footer(tmp_path, monkeypatch): monkeypatch.setenv("HERMES_HOME", str(tmp_path)) content = "short page\nwith a few lines\n" - model_text, truncated = wt._truncate_with_footer( + model_text, truncated = web_tools_truncate._truncate_with_footer( content, "https://example.com/s", char_limit=15000 ) assert not truncated diff --git a/tests/tools/test_web_keyless_rescue.py b/tests/tools/test_web_keyless_rescue.py index 408200006f..fdb1b73e62 100644 --- a/tests/tools/test_web_keyless_rescue.py +++ b/tests/tools/test_web_keyless_rescue.py @@ -19,6 +19,7 @@ from unittest.mock import patch import pytest import tools.web_tools as web_tools +from tools import web_tools_rescue from plugins.web import keyless_mcp from plugins.web.keenable.provider import KeenableWebSearchProvider @@ -73,33 +74,33 @@ def _ring_ok(vendor="exa"): class TestEligibility: def test_keyed_ring_vendor_is_eligible(self): - assert web_tools._rescue_eligible(_KeyedBoomProvider()) is True + assert web_tools_rescue._rescue_eligible(_KeyedBoomProvider()) is True def test_keyless_mode_ring_vendor_not_eligible(self, monkeypatch): # No key: the keenable call already rode the ring; no double-walk. monkeypatch.setattr( "agent.web_search_provider.get_provider_env", lambda name: "" ) - assert web_tools._rescue_eligible(KeenableWebSearchProvider()) is False + assert web_tools_rescue._rescue_eligible(KeenableWebSearchProvider()) is False def test_non_ring_backend_is_eligible(self): class _SearxProvider(_KeyedBoomProvider): name = "searxng" - assert web_tools._rescue_eligible(_SearxProvider()) is True + assert web_tools_rescue._rescue_eligible(_SearxProvider()) is True def test_config_gate_disables(self, monkeypatch): monkeypatch.setattr( web_tools, "_load_web_config", lambda: {"backend": "keenable", "keyless_rescue": False}, ) - assert web_tools._rescue_eligible(_KeyedBoomProvider()) is False + assert web_tools_rescue._rescue_eligible(_KeyedBoomProvider()) is False def test_keyless_fallback_off_disables(self, monkeypatch): monkeypatch.setattr( "agent.web_search_registry._keyless_tier_enabled", lambda: False ) - assert web_tools._rescue_eligible(_KeyedBoomProvider()) is False + assert web_tools_rescue._rescue_eligible(_KeyedBoomProvider()) is False class TestSearchRescue: @@ -211,7 +212,7 @@ class TestExtractRescue: with patch.object( keyless_mcp, "extract_with_failover", return_value=good ): - out = web_tools._rescue_extract("keenable", ["https://a"], failed) + out = web_tools_rescue._rescue_extract("keenable", ["https://a"], failed) assert out[0]["metadata"]["rescued_from"] == "keenable" assert "HTTP 500" in out[0]["metadata"]["backend_error"] diff --git a/tests/tools/test_web_result_cache.py b/tests/tools/test_web_result_cache.py index b9a40be73a..3a804baf7a 100644 --- a/tests/tools/test_web_result_cache.py +++ b/tests/tools/test_web_result_cache.py @@ -190,7 +190,7 @@ def test_extract_cache_provider_participates_in_key(_isolated_cache): def test_extract_cache_oversized_page_not_indexed(_isolated_cache): - import tools.web_tools as wt + from tools import web_tools_truncate as wt big = "x" * (wt.MAX_STORED_TEXT_CHARS + 1) extract_cache_put("https://big.com", big) assert extract_cache_get("https://big.com") is None diff --git a/tests/tools/test_web_tools_tavily.py b/tests/tools/test_web_tools_tavily.py index b6fe37c59b..b7bc7296e0 100644 --- a/tests/tools/test_web_tools_tavily.py +++ b/tests/tools/test_web_tools_tavily.py @@ -92,7 +92,7 @@ class TestNormalizeTavilySearchResults: """Test search result normalization.""" def test_basic_normalization(self): - from tools.web_tools import _normalize_tavily_search_results + from plugins.web.tavily.provider import _normalize_tavily_search_results raw = { "results": [ {"title": "Python Docs", "url": "https://docs.python.org", "content": "Official docs", "score": 0.9}, @@ -111,7 +111,7 @@ class TestNormalizeTavilySearchResults: def test_missing_fields(self): - from tools.web_tools import _normalize_tavily_search_results + from plugins.web.tavily.provider import _normalize_tavily_search_results result = _normalize_tavily_search_results({"results": [{}]}) web = result["data"]["web"] assert web[0]["title"] == "" @@ -125,7 +125,7 @@ class TestNormalizeTavilyDocuments: """Test extract document normalization.""" def test_basic_document(self): - from tools.web_tools import _normalize_tavily_documents + from plugins.web.tavily.provider import _normalize_tavily_documents raw = { "results": [{ "url": "https://example.com", @@ -143,7 +143,7 @@ class TestNormalizeTavilyDocuments: def test_fallback_url(self): - from tools.web_tools import _normalize_tavily_documents + from plugins.web.tavily.provider import _normalize_tavily_documents raw = {"results": [{"content": "data"}]} docs = _normalize_tavily_documents(raw, fallback_url="https://fallback.com") assert docs[0]["url"] == "https://fallback.com" diff --git a/tests/tools/test_web_tools_truncate.py b/tests/tools/test_web_tools_truncate.py index d1dab454fe..8d0a6f2812 100644 --- a/tests/tools/test_web_tools_truncate.py +++ b/tests/tools/test_web_tools_truncate.py @@ -12,13 +12,14 @@ from unittest.mock import patch import pytest import tools.web_tools as wt +from tools import web_tools_truncate class TestImageConversion: def test_markdown_base64_image_keeps_alt_drops_blob(self): blob = "A" * 5000 text = f"before ![a cat]( data:image/png;base64,{blob}) after" - out = wt.convert_base64_images_to_links(text) + out = web_tools_truncate.convert_base64_images_to_links(text) assert "[IMAGE: a cat]" in out assert "base64" not in out assert blob not in out @@ -27,16 +28,16 @@ class TestImageConversion: def test_bare_and_parenthesised_base64_become_placeholder(self): blob = "Z" * 3000 - bare = wt.convert_base64_images_to_links(f"data:image/gif;base64,{blob}") + bare = web_tools_truncate.convert_base64_images_to_links(f"data:image/gif;base64,{blob}") assert bare == "[IMAGE]" - paren = wt.convert_base64_images_to_links(f"(data:image/gif;base64,{blob})") + paren = web_tools_truncate.convert_base64_images_to_links(f"(data:image/gif;base64,{blob})") assert paren == "[IMAGE]" class TestTruncation: def test_short_content_returned_whole(self): content = "# Title\n\nshort body\n" - out, truncated = wt._truncate_with_footer(content, "https://e.com", 15000) + out, truncated = web_tools_truncate._truncate_with_footer(content, "https://e.com", 15000) assert out == content assert truncated is False @@ -44,7 +45,7 @@ class TestTruncation: def test_truncation_stores_full_text_readable(self, tmp_path, monkeypatch): monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes")) body = "UNIQUE_MIDDLE_MARKER\n" + ("\n".join(f"row {i}" for i in range(5000))) - out, truncated = wt._truncate_with_footer(body, "https://example.com/doc", 3000) + out, truncated = web_tools_truncate._truncate_with_footer(body, "https://example.com/doc", 3000) assert truncated is True # Extract the stored path from the footer and confirm full text is there. path_line = next(ln for ln in out.splitlines() if "Full text saved to:" in ln) @@ -58,12 +59,12 @@ class TestTruncation: class TestCharLimitConfig: def test_default_when_unset(self): with patch("tools.web_tools._load_web_config", return_value={}): - assert wt._get_extract_char_limit() == wt.DEFAULT_EXTRACT_CHAR_LIMIT + assert web_tools_truncate._get_extract_char_limit() == web_tools_truncate.DEFAULT_EXTRACT_CHAR_LIMIT def test_bad_value_falls_back(self): with patch("tools.web_tools._load_web_config", return_value={"extract_char_limit": "nope"}): - assert wt._get_extract_char_limit() == wt.DEFAULT_EXTRACT_CHAR_LIMIT + assert web_tools_truncate._get_extract_char_limit() == web_tools_truncate.DEFAULT_EXTRACT_CHAR_LIMIT class TestEndToEnd: diff --git a/tools/web_result_cache.py b/tools/web_result_cache.py index a544ebd9c9..1eda5ff315 100644 --- a/tools/web_result_cache.py +++ b/tools/web_result_cache.py @@ -288,7 +288,7 @@ def extract_cache_put( if not content or not _cacheable(url): return try: - from tools.web_tools import MAX_STORED_TEXT_CHARS + from tools.web_tools_truncate import MAX_STORED_TEXT_CHARS file_path = _entry_file_path(url, format, provider) if len(content) > MAX_STORED_TEXT_CHARS or file_path is None: return diff --git a/tools/web_tools_extract.py b/tools/web_tools_extract.py index 57642c48c5..ba0c1c485c 100644 --- a/tools/web_tools_extract.py +++ b/tools/web_tools_extract.py @@ -3,8 +3,7 @@ Order of controls (each is a gate, never skipped by a cache hit): secret-URL refusal -> SSRF filter (in web_tools.web_extract_tool) -> provider resolution (strict selection) -> per-URL website policy -> disk cache -> vendor call with -one-shot keyless rescue. Names are re-imported by tools/web_tools.py so -``tools.web_tools._validate_extract_urls`` etc. keep working; logs under the origin logger. +one-shot keyless rescue. Logs under the origin (tools.web_tools) logger. """ import asyncio diff --git a/tools/web_tools_rescue.py b/tools/web_tools_rescue.py index bfa57c6360..ade3332ecc 100644 --- a/tools/web_tools_rescue.py +++ b/tools/web_tools_rescue.py @@ -2,8 +2,8 @@ Stateless by design: a rescue routes THIS call through the free-tier ring (plugins/web/keyless_mcp.py); the next web_search/web_extract call attempts the chosen backend again. Callers must never cache a -rescue-served response, or the one-shot rescue becomes sticky for a whole TTL. Names are re-imported by -tools/web_tools.py (``tools.web_tools._rescue_eligible``); logs under the origin logger. +rescue-served response, or the one-shot rescue becomes sticky for a whole TTL. Logs under the origin +(tools.web_tools) logger. """ import logging diff --git a/tools/web_tools_truncate.py b/tools/web_tools_truncate.py index a2b69b4f5c..9920f4055b 100644 --- a/tools/web_tools_truncate.py +++ b/tools/web_tools_truncate.py @@ -2,8 +2,8 @@ Pages at or under the char budget are returned whole; larger pages become a head+tail window plus a footer that says how much is shown, where the full text is stored (cache/web) and the exact read_file -call that pages the omitted middle. Inline base64 images become ``[IMAGE: alt]`` placeholders. Names are -re-imported by tools/web_tools.py (``tools.web_tools.MAX_STORED_TEXT_CHARS``); logs under the origin logger. +call that pages the omitted middle. Inline base64 images become ``[IMAGE: alt]`` placeholders. Logs under the +origin (tools.web_tools) logger. """ import logging