Files
hermes-agent/tests/agent/test_fallback_reasoning_override.py
teknium1 d10bb2ab6f test: make tests/ mirror the source tree; drop issue numbers from filenames
`scripts/run_tests.sh tests/<dir>/` is how a change gets its regression
coverage run, so a test filed under the wrong directory is a test nobody
runs when that code changes. Two kinds of drift had accumulated.

Parallel directories for one source package, folded into the mirror:
  tests/acp        -> tests/acp_adapter   (its __init__/conftest move with it)
  tests/cli        -> tests/hermes_cli    (prompt_toolkit fixture merged into
                                           hermes_cli/conftest.py)
  tests/run_agent  -> tests/agent         (backoff fixture becomes
                                           agent/conftest.py)
  tests/relay      -> tests/gateway/relay
  tests/state      -> tests/hermes_state

246 loose files at tests/ root, routed by the package they import/patch:
hermes_cli, hermes_state, agent, gateway, tools, plugins, tui_gateway, cron.
Installer and desktop-update script tests go to tests/scripts/{install,
desktop_update}/. 43 tests of root-level modules (batch_runner, utils,
hermes_constants, packaging) stay at the root.

Filenames drop their issue numbers (95 files: test_89315_x.py -> test_x.py);
the number stays in the module docstring where it has context.

Collisions: test_cli_skin_integration.py existed in both tests/ and tests/cli
with different subsets — merged into one (10 tests, all kept);
run_agent/test_pre_compress_memory_context.py -> agent/..._handoff.py;
tests/test_account_usage.py -> agent/test_account_usage_fetch.py;
tests/test_web_server.py -> hermes_cli/test_web_server_ws_ping.py.
Deleted: test_minisweagent_path.py (empty since PR #2804),
test_model_picker_scroll.py (tested a private copy of the logic, imported
nothing), test_process_loop_event_loop_warning.py (asserted asyncio behaviour,
imported nothing from Hermes).

Repo-root path arithmetic (Path(__file__).parents[N], dirname chains) is
bumped for the 202 files that changed depth and verified by evaluating every
such expression against the new location. classify_changes' desktop-updater
lane prefix, tests-os.yml's ignore glob and every in-tree path comment follow
the moves. tests/test_tests_tree_layout.py keeps the tree from drifting back.
2026-09-13 09:18:02 -07:00

138 lines
6.0 KiB
Python

"""Tests for per-model reasoning_effort override during fallback activation.
Tests that try_activate_fallback re-resolves reasoning_config when
swapping to a fallback model, so per-model overrides are honored even
during error recovery.
"""
import pytest
from unittest.mock import MagicMock, patch
class TestFallbackReasoningOverride:
"""Test try_activate_fallback re-resolves reasoning_config."""
def test_fallback_re_resolves_reasoning_config(self):
"""When fallback activates, reasoning_config should be re-resolved.
We test the resolution logic directly rather than spinning up a
full try_activate_fallback (which requires extensive agent setup).
The production code calls resolve_per_model_reasoning_effort with
the fallback model string — we verify that works correctly.
"""
from hermes_constants import resolve_per_model_reasoning_effort
# Simulate: primary was gemini-flash (medium), fallback to claude-opus-4.5 (xhigh)
overrides = {
"claude-opus-4.5": "xhigh",
"gemini-flash": "medium",
}
# Fallback model lookup
fb_result = resolve_per_model_reasoning_effort("claude-opus-4.5", overrides)
assert fb_result is not None
assert fb_result["effort"] == "xhigh"
# Primary model lookup (for comparison)
primary_result = resolve_per_model_reasoning_effort("gemini-flash", overrides)
assert primary_result is not None
assert primary_result["effort"] == "medium"
# The key point: fallback result differs from primary
assert fb_result["effort"] != primary_result["effort"]
def test_fallback_to_model_without_override_uses_global(self):
"""Fallback to a model with no override should resolve to None (→ global)."""
from hermes_constants import resolve_per_model_reasoning_effort
overrides = {"claude-opus-4.5": "xhigh"}
# Fallback to gpt-5 which has no override
result = resolve_per_model_reasoning_effort("gpt-5", overrides)
assert result is None # caller falls back to global
def test_fallback_recovery_restores_primary_reasoning(self):
"""After fallback + restore_primary_runtime, reasoning_config returns to primary's value.
This tests the integration of Task 6 (_primary_runtime snapshot) with
Task 6b (fallback re-resolution). The full cycle:
1. Primary model = gemini-flash, reasoning = medium
2. /model switch → _primary_runtime captures reasoning_config
3. Fallback activates → reasoning re-resolved for fallback model
4. restore_primary_runtime → reasoning_config restored from snapshot
"""
from agent.agent_runtime_helpers import restore_primary_runtime
agent = MagicMock()
# Simulate: _primary_runtime was captured during /model switch
agent._primary_runtime = {
"model": "gemini-flash",
"provider": "google",
"base_url": "",
"api_mode": "openai",
"api_key": "key",
"client_kwargs": {},
"use_prompt_caching": False,
"use_native_cache_layout": False,
"runtime_capabilities": {"native_compaction": True},
"reasoning_config": {"enabled": True, "effort": "medium"},
"compressor_model": "gemini-flash",
"compressor_base_url": "",
"compressor_api_key": "",
"compressor_provider": "",
"compressor_context_length": 0,
"compressor_api_mode": "",
"compressor_threshold_tokens": 0,
}
agent._fallback_activated = True
agent._fallback_index = 0
agent._fallback_chain = []
agent._fallback_model = None
agent._transport_cache = {}
agent._config_context_length = None
agent._rate_limited_until = 0
# During fallback, reasoning was changed to xhigh (fallback model's override)
agent.model = "claude-opus-4.5"
agent.provider = "anthropic"
agent.reasoning_config = {"enabled": True, "effort": "xhigh"}
agent.runtime_capabilities = {"native_compaction": False}
agent.context_compressor = MagicMock()
agent.base_url = ""
agent._anthropic_prompt_cache_policy = MagicMock(return_value=(False, False))
agent._create_openai_client = MagicMock(return_value=MagicMock())
agent._ensure_lmstudio_runtime_loaded = MagicMock()
result = restore_primary_runtime(agent)
assert result is True
# reasoning_config should be restored to primary's value (medium)
assert agent.reasoning_config == {"enabled": True, "effort": "medium"}
assert agent.runtime_capabilities == {"native_compaction": True}
def test_fallback_global_fallback_with_yaml_false(self):
"""Fallback global fallback must not coerce YAML boolean False.
Regression: ``or ""`` turned False into "", silently re-enabling
thinking. The raw value must pass through so
parse_reasoning_effort(False) returns {'enabled': False}.
The production code in try_activate_fallback does:
_fb_global_effort = _fb_agent_cfg.get("reasoning_effort", "")
agent.reasoning_config = parse_reasoning_effort(_fb_global_effort)
We verify that passing the raw False (not coerced "") produces
the disabled config.
"""
from hermes_constants import parse_reasoning_effort
# Simulate: no per-model override matches, global is YAML False
_fb_agent_cfg = {"reasoning_effort": False}
# This is the exact line from try_activate_fallback's else branch.
# The bug was: _fb_global_effort = _fb_agent_cfg.get(...) or ""
# which turned False into "". The fix passes the raw value.
_fb_global_effort = _fb_agent_cfg.get("reasoning_effort", "")
result = parse_reasoning_effort(_fb_global_effort)
assert result is not None
assert result.get("enabled") is False