Files
hermes-agent/tests/agent/test_overflow_partial_terminal.py
teknium1 d10bb2ab6f test: make tests/ mirror the source tree; drop issue numbers from filenames
`scripts/run_tests.sh tests/<dir>/` is how a change gets its regression
coverage run, so a test filed under the wrong directory is a test nobody
runs when that code changes. Two kinds of drift had accumulated.

Parallel directories for one source package, folded into the mirror:
  tests/acp        -> tests/acp_adapter   (its __init__/conftest move with it)
  tests/cli        -> tests/hermes_cli    (prompt_toolkit fixture merged into
                                           hermes_cli/conftest.py)
  tests/run_agent  -> tests/agent         (backoff fixture becomes
                                           agent/conftest.py)
  tests/relay      -> tests/gateway/relay
  tests/state      -> tests/hermes_state

246 loose files at tests/ root, routed by the package they import/patch:
hermes_cli, hermes_state, agent, gateway, tools, plugins, tui_gateway, cron.
Installer and desktop-update script tests go to tests/scripts/{install,
desktop_update}/. 43 tests of root-level modules (batch_runner, utils,
hermes_constants, packaging) stay at the root.

Filenames drop their issue numbers (95 files: test_89315_x.py -> test_x.py);
the number stays in the module docstring where it has context.

Collisions: test_cli_skin_integration.py existed in both tests/ and tests/cli
with different subsets — merged into one (10 tests, all kept);
run_agent/test_pre_compress_memory_context.py -> agent/..._handoff.py;
tests/test_account_usage.py -> agent/test_account_usage_fetch.py;
tests/test_web_server.py -> hermes_cli/test_web_server_ws_ping.py.
Deleted: test_minisweagent_path.py (empty since PR #2804),
test_model_picker_scroll.py (tested a private copy of the logic, imported
nothing), test_process_loop_event_loop_warning.py (asserted asyncio behaviour,
imported nothing from Hermes).

Repo-root path arithmetic (Path(__file__).parents[N], dirname chains) is
bumped for the 202 files that changed depth and verified by evaluating every
such expression against the new location. classify_changes' desktop-updater
lane prefix, tests-os.yml's ignore glob and every in-tree path comment follow
the moves. tests/test_tests_tree_layout.py keeps the tree from drifting back.
2026-09-13 09:18:02 -07:00

163 lines
6.9 KiB
Python

"""Regression tests for #106260: a context-overflow error after partial stream
delivery must NOT seed a continuation stub with the recovered text.
Seeding tens of KB of partial content as a length-continuation stub makes every
later request larger, so a session whose transcript cannot fit (compression
failed / protect_last_n covers it) loops forever growing the context. The stub
is instead marked terminal (content empty) and the loop ends the turn via the
recovery contract.
"""
from types import SimpleNamespace
from unittest.mock import MagicMock, patch
import pytest
from hermes_constants import PARTIAL_STREAM_STUB_ID, FINISH_REASON_LENGTH
def _make_agent():
from run_agent import AIAgent
agent = AIAgent(
api_key="test-key",
base_url="https://example.com/v1",
model="test/model",
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
)
agent.api_mode = "chat_completions"
agent._interrupt_requested = False
return agent
def _make_stream_chunk(content=None, finish_reason=None):
delta = SimpleNamespace(content=content, tool_calls=None,
reasoning_content=None, reasoning=None)
choice = SimpleNamespace(index=0, delta=delta, finish_reason=finish_reason)
return SimpleNamespace(choices=[choice], model=None, usage=None)
class TestOverflowTerminalStub:
@patch("run_agent.AIAgent._create_request_openai_client")
@patch("run_agent.AIAgent._close_request_openai_client")
def test_partial_stream_overflow_error_returns_terminal_stub(
self, _mock_close, mock_create, monkeypatch,
):
"""A stream that delivered text then hit the provider's
maximum-context-length error must return an EMPTY stub marked terminal,
not a continuation stub carrying the recovered text (#106260)."""
def _overflowing_stream():
yield _make_stream_chunk(content="Here's my long partial answer ...")
raise RuntimeError(
"This model's maximum context length is 128000 tokens. "
"However, you requested 140000 tokens."
)
mock_client = MagicMock()
mock_client.chat.completions.create.side_effect = lambda *a, **kw: _overflowing_stream()
mock_create.return_value = mock_client
agent = _make_agent()
agent._current_streamed_assistant_text = "Here's my long partial answer ..."
monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
response = agent._interruptible_streaming_api_call({})
assert response.id == PARTIAL_STREAM_STUB_ID
assert getattr(response, "_overflow_terminal", False) is True
# The recovered text must not be seeded for continuation.
assert response.choices[0].message.content is None
@patch("run_agent.AIAgent._create_request_openai_client")
@patch("run_agent.AIAgent._close_request_openai_client")
def test_payload_too_large_partial_is_not_made_terminal(
self, _mock_close, mock_create, monkeypatch,
):
"""Review P1 (andrexibiza): payload_too_large (413) has its own byte-scored
recovery owner and must NOT be collapsed into the context-overflow terminal
contract — the partial keeps its normal continuation stub."""
def _overflowing_stream():
yield _make_stream_chunk(content="some media-heavy partial output ...")
raise RuntimeError(
"Request payload too large (413): 5MB exceeds the 4MB limit"
)
mock_client = MagicMock()
mock_client.chat.completions.create.side_effect = lambda *a, **kw: _overflowing_stream()
mock_create.return_value = mock_client
agent = _make_agent()
agent._current_streamed_assistant_text = "some media-heavy partial output ..."
monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
response = agent._interruptible_streaming_api_call({})
assert response.id == PARTIAL_STREAM_STUB_ID
assert getattr(response, "_overflow_terminal", False) is False
# The recovered text is preserved for the normal continuation path.
assert response.choices[0].message.content == "some media-heavy partial output ..."
class TestRecoverFromTruncationOverflowTerminal:
def _mock_agent(self):
agent = MagicMock()
agent._vprint = MagicMock()
agent._flush_status_buffer = MagicMock()
agent._cleanup_task_resources = MagicMock()
agent._persist_session = MagicMock()
return agent
def _response(self, overflow_terminal=True, content="recovered text"):
return SimpleNamespace(
id=PARTIAL_STREAM_STUB_ID,
_overflow_terminal=overflow_terminal,
_dropped_tool_names=None,
choices=[SimpleNamespace(
index=0,
message=SimpleNamespace(role="assistant", content=content,
tool_calls=None, reasoning_content=None),
finish_reason=FINISH_REASON_LENGTH,
)],
)
def test_overflow_terminal_stub_ends_turn_without_continuation(self):
from agent.turn_truncation import (
_CONTEXT_OVERFLOW_PARTIAL_FINAL,
recover_from_truncation,
)
agent = self._mock_agent()
# The overflow fired right after a tool batch: the transcript tail is a
# raw tool result, which strict providers reject as tool -> user.
messages = [
{"role": "user", "content": "go"},
{"role": "assistant", "content": None,
"tool_calls": [{"id": "c1", "type": "function",
"function": {"name": "read_file", "arguments": "{}"}}]},
{"role": "tool", "tool_call_id": "c1", "content": "big file"},
]
verdict = recover_from_truncation(
agent, self._response(), FINISH_REASON_LENGTH, MagicMock(),
messages=messages, conversation_history=None, api_kwargs={},
api_call_count=0, effective_task_id=None, current_turn_user_idx=None,
length_continue_retries=0, truncated_response_parts=[],
truncated_tool_call_retries=0, retry_count=0, compression_attempts=0,
)
assert verdict.action == "return"
result = verdict.result or {}
assert result.get("failed") is True
assert result.get("final_response") == _CONTEXT_OVERFLOW_PARTIAL_FINAL
# #98722 typed bit: the gateway consumes this to reset/move future input
# to a clean session instead of leaving the bloated one authoritative.
assert result.get("compression_exhausted") is True
# The interrupted tool tail is closed so the next user turn alternates;
# no fragment or nudge was appended.
assert messages[-1]["role"] == "assistant"
assert messages[-1]["content"] == _CONTEXT_OVERFLOW_PARTIAL_FINAL
assert len(messages) == 4
assert result.get("messages") is messages