`tests/hermes_cli/test_plugins_cmd.py::TestNoAutoActivation::test_compressor_default_ignores_plugin`
fails on every Windows machine:
UnicodeDecodeError: 'charmap' codec can't decode byte 0x8f in
position 47744: character maps to <undefined>
The test reads `run_agent.py` back as text to assert a removed comment is
gone, but called `open()` with no `encoding=`. Python then falls back to
the locale preferred encoding, which is cp1252 on a default Windows
install rather than UTF-8. `run_agent.py` contains nine bytes cp1252
leaves undefined, so the read raises before the assertion is reached. On
Linux and macOS the preferred encoding is UTF-8 and the same line is
fine, which is why CI never caught it.
That one line is the only active failure. The rest of this change closes
the same gap in the files it touches, which `scripts/check-windows-footguns.py`
flags and which the #71014 read_text campaign has been working through
elsewhere in the tree:
- `tests/hermes_cli/test_plugins_cmd.py`: nine bare `write_text`/`read_text`
calls writing YAML manifests, config and plugin sources
- `tests/tools/test_web_tools_truncate.py`: reads stored extracted web text,
which is arbitrary content from the internet
- `tests/stress/test_atypical_scenarios.py`: writes and reads worker task
ids and a barrier file
All three files are now clean under `check-windows-footguns.py`.
Reads go through `Path.read_text(encoding="utf-8")` rather than
`open(...).read()`, which also closes the handle instead of leaving it to
the garbage collector. On Windows a live handle blocks tmpdir cleanup, so
that part is not cosmetic either.
No new test. The repaired test is the regression coverage: it fails
before this change and passes after, on Windows.
112 lines
4.2 KiB
Python
112 lines
4.2 KiB
Python
"""Unit tests for the truncate-and-store web_extract path (no LLM).
|
|
|
|
Covers convert_base64_images_to_links, _truncate_with_footer, _store_full_text,
|
|
_get_extract_char_limit, and the end-to-end web_extract_tool truncation behavior.
|
|
"""
|
|
import asyncio
|
|
import json
|
|
import os
|
|
from pathlib import Path
|
|
from unittest.mock import patch
|
|
|
|
import pytest
|
|
|
|
import tools.web_tools as wt
|
|
|
|
|
|
class TestImageConversion:
|
|
def test_markdown_base64_image_keeps_alt_drops_blob(self):
|
|
blob = "A" * 5000
|
|
text = f"before  after"
|
|
out = wt.convert_base64_images_to_links(text)
|
|
assert "[IMAGE: a cat]" in out
|
|
assert "base64" not in out
|
|
assert blob not in out
|
|
assert "before" in out and "after" in out
|
|
|
|
|
|
def test_bare_and_parenthesised_base64_become_placeholder(self):
|
|
blob = "Z" * 3000
|
|
bare = wt.convert_base64_images_to_links(f"data:image/gif;base64,{blob}")
|
|
assert bare == "[IMAGE]"
|
|
paren = wt.convert_base64_images_to_links(f"(data:image/gif;base64,{blob})")
|
|
assert paren == "[IMAGE]"
|
|
|
|
|
|
class TestTruncation:
|
|
def test_short_content_returned_whole(self):
|
|
content = "# Title\n\nshort body\n"
|
|
out, truncated = wt._truncate_with_footer(content, "https://e.com", 15000)
|
|
assert out == content
|
|
assert truncated is False
|
|
|
|
|
|
def test_truncation_stores_full_text_readable(self, tmp_path, monkeypatch):
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
body = "UNIQUE_MIDDLE_MARKER\n" + ("\n".join(f"row {i}" for i in range(5000)))
|
|
out, truncated = wt._truncate_with_footer(body, "https://example.com/doc", 3000)
|
|
assert truncated is True
|
|
# Extract the stored path from the footer and confirm full text is there.
|
|
path_line = next(ln for ln in out.splitlines() if "Full text saved to:" in ln)
|
|
stored_path = path_line.split("Full text saved to:", 1)[1].strip()
|
|
assert os.path.exists(stored_path)
|
|
full = Path(stored_path).read_text(encoding="utf-8")
|
|
assert "UNIQUE_MIDDLE_MARKER" in full
|
|
assert "row 2500" in full # the omitted-middle row is in the stored file
|
|
|
|
|
|
class TestCharLimitConfig:
|
|
def test_default_when_unset(self):
|
|
with patch("tools.web_tools._load_web_config", return_value={}):
|
|
assert wt._get_extract_char_limit() == wt.DEFAULT_EXTRACT_CHAR_LIMIT
|
|
|
|
|
|
def test_bad_value_falls_back(self):
|
|
with patch("tools.web_tools._load_web_config", return_value={"extract_char_limit": "nope"}):
|
|
assert wt._get_extract_char_limit() == wt.DEFAULT_EXTRACT_CHAR_LIMIT
|
|
|
|
|
|
class TestEndToEnd:
|
|
def test_web_extract_truncates_large_page_no_llm(self, tmp_path, monkeypatch):
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
big = "\n".join(f"para {i} " + "y" * 80 for i in range(3000))
|
|
|
|
class FakeProvider:
|
|
name = "fake"
|
|
display_name = "Fake"
|
|
|
|
def supports_extract(self):
|
|
return True
|
|
|
|
async def extract(self, urls, **kwargs):
|
|
return [{"url": urls[0], "title": "Big Page", "content": big,
|
|
"raw_content": big, "metadata": {}}]
|
|
|
|
with patch("tools.web_tools._ensure_web_plugins_loaded"), \
|
|
patch("tools.web_tools._get_extract_backend", return_value="fake"), \
|
|
patch("tools.web_tools.async_is_safe_url", new=_AsyncTrue()), \
|
|
patch("agent.web_search_registry.get_provider", return_value=FakeProvider()):
|
|
result = json.loads(asyncio.new_event_loop().run_until_complete(
|
|
wt.web_extract_tool(["https://example.com/big"], char_limit=5000)
|
|
))
|
|
|
|
assert "results" in result
|
|
content = result["results"][0]["content"]
|
|
assert "[TRUNCATED]" in content
|
|
assert "Full text saved to:" in content
|
|
# No LLM was involved: para 0 (head) and the last para (tail) are verbatim.
|
|
assert "para 0 " in content
|
|
assert "para 2999 " in content
|
|
|
|
|
|
def _make_awaitable(value):
|
|
async def _coro(*a, **k):
|
|
return value
|
|
return _coro()
|
|
|
|
|
|
class _AsyncTrue:
|
|
"""Async callable that always returns True (re-awaitable per call)."""
|
|
async def __call__(self, *a, **k):
|
|
return True
|