Files
hermes-agent/tests/hermes_cli/test_local_runtime.py
teknium1 379f14b553 test(local-runtime): racing-callers fake publishes a modern state record
65ff3ad353 made a bare legacy {pid} record non-adoptable (a live PID alone is not
evidence of the managed router) and moved the other tests in this file onto
_write_current_process_state, but left the racing-callers fake publishing the legacy
shape, so the second caller could never adopt the first's state and main's Run tests
lane went red on every PR rebased past it (main's own CI could not start at the time).
Publish the modern record the real supervisor writes.
2026-09-28 12:18:58 -07:00

964 lines
40 KiB
Python

"""Contract tests for hermes_cli.local_runtime — Rollouts 1+2.
Per the design's verification plan: relationships and contracts, no
change-detector tests, real imports against temp HERMES_HOME (the autouse
fixture isolates it). The stub HTTP server speaks just enough llama-server
(/props, /health, /models, /v1/chat/completions, /metrics, /slots) to
exercise detection fingerprinting and supervisor logic without a GPU.
"""
from __future__ import annotations
import json
import logging
import os
import threading
from http.server import BaseHTTPRequestHandler, HTTPServer
import pytest
from hermes_cli.local_runtime.binaries import select_backend
from hermes_cli.local_runtime.detect import DetectedServer, probe_port
def _write_current_process_state(path: Path, *, base_url: str, api_key: str) -> None:
"""Publish a modern state record for the process hosting the test stub."""
import psutil
proc = psutil.Process()
parent = proc.parent()
assert parent is not None
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps({
"base_url": base_url,
"api_key": api_key,
"pid": proc.pid,
"create_time": proc.create_time(),
"executable": proc.exe(),
"owner_pid": parent.pid,
"owner_create_time": parent.create_time(),
}), encoding="utf-8")
# ── stub llama-server ────────────────────────────────────────
class _StubHandler(BaseHTTPRequestHandler):
"""Minimal llama-server imitation; behavior driven by class attrs."""
props: dict = {}
models: dict | None = None
require_auth = False
chat_answer = "Paris"
requests_processing = 0
slots: list = []
slots_error = 0
metrics_error = 0
def _send(self, code: int, body: dict | str | None = None) -> None:
raw = (json.dumps(body) if isinstance(body, dict) else (body or "")).encode()
self.send_response(code)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(raw)))
self.end_headers()
self.wfile.write(raw)
def do_GET(self): # noqa: N802
if self.require_auth and "Authorization" not in self.headers:
self._send(401, {})
return
path = self.path.split("?")[0] # router telemetry uses ?model=
if path == "/props":
self._send(200, self.props)
elif path == "/health":
self._send(200, {"status": "ok"})
elif path == "/models":
if self.models is None:
self._send(404, {})
else:
self._send(200, self.models)
elif path == "/metrics":
if self.metrics_error:
self._send(self.metrics_error, {})
else:
self._send(
200, f"llamacpp:requests_processing {self.requests_processing}\n"
)
elif path == "/slots":
if self.slots_error:
self._send(self.slots_error, {})
return
raw = json.dumps(self.slots).encode()
self.send_response(200)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(raw)))
self.end_headers()
self.wfile.write(raw)
else:
self._send(404, {})
def do_POST(self): # noqa: N802
if self.path == "/v1/chat/completions":
self._send(200, {"choices": [{"message": {
"role": "assistant", "content": self.chat_answer}}]})
elif self.path == "/models/load":
self._send(200, {"success": True})
elif self.path == "/models/unload":
type(self).unloaded = getattr(type(self), "unloaded", [])
length = int(self.headers.get("Content-Length", 0))
body = json.loads(self.rfile.read(length)) if length else {}
type(self).unloaded.append(body.get("model"))
self._send(200, {"success": True})
else:
self._send(404, {})
def log_message(self, *args): # silence
pass
@pytest.fixture
def stub_server():
"""Yields (port, handler_class); handler attrs are per-test mutable."""
class Handler(_StubHandler):
props = {}
models = None
require_auth = False
slots = []
server = HTTPServer(("127.0.0.1", 0), Handler)
thread = threading.Thread(target=server.serve_forever, daemon=True)
thread.start()
yield server.server_address[1], Handler
server.shutdown()
# ── detection (Rollout 1) ────────────────────────────────────
def test_probe_fingerprints_real_llama_server(stub_server):
port, handler = stub_server
handler.props = {
"build_info": "b10290-c8e03ce81",
"model_path": "C:/models/some model with spaces.gguf",
"default_generation_settings": {"n_ctx": 65536},
}
handler.models = {"data": [{"id": "m", "status": {"value": "unloaded"}}]}
hit = probe_port(port)
assert isinstance(hit, DetectedServer)
assert hit.base_url == f"http://127.0.0.1:{port}/v1"
assert hit.build_info.startswith("b10290")
assert hit.n_ctx == 65536
assert hit.router_mode is True
assert hit.auth_required is False
def test_probe_rejects_non_llama_openai_server(stub_server):
# Answers /props with no build_info (e.g. some other local service).
port, handler = stub_server
handler.props = {"something": "else"}
assert probe_port(port) is None
def test_probe_single_model_mode_is_not_router(stub_server):
port, handler = stub_server
handler.props = {"build_info": "b10290-x", "model_path": "m.gguf"}
handler.models = None # /models 404s in plain (non-router) mode
hit = probe_port(port)
assert hit is not None
assert hit.router_mode is False
def test_probe_auth_required_still_detected(stub_server):
port, handler = stub_server
handler.require_auth = True
hit = probe_port(port)
assert hit is not None
assert hit.auth_required is True
def test_probe_dead_port_returns_none():
# Bind-then-close to get a port that is definitely closed.
import socket
with socket.socket() as s:
s.bind(("127.0.0.1", 0))
dead_port = s.getsockname()[1]
assert probe_port(dead_port) is None
# ── binary resolver (Rollout 2) ──────────────────────────────
@pytest.mark.parametrize("vendor,os_name,expected", [
("NVIDIA GeForce RTX 5090", "win", "cuda"),
("nvidia", "ubuntu", "cuda"),
("AMD Radeon RX 7900", "win", "vulkan"),
("intel", "win", "vulkan"),
(None, "win", "cpu"),
("", "ubuntu", "cpu"),
("nvidia", "macos", "metal"), # macOS is Metal regardless
(None, "macos", "metal"),
])
def test_backend_selection(vendor, os_name, expected):
assert select_backend(vendor, os_name=os_name) == expected
# ── supervisor contracts (stubbed; no GPU) ───────────────────
def _make_supervisor(tmp_path, port):
"""Supervisor pointed at the stub: skip spawn, drive HTTP logic only."""
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
sup = LlamaServerSupervisor(
binary=tmp_path / "llama-server", models_dir=tmp_path, port=port)
return sup
def test_touch_generate_is_the_readiness_proof(stub_server, tmp_path):
port, handler = stub_server
sup = _make_supervisor(tmp_path, port)
handler.chat_answer = "Paris"
assert sup.touch_generate("m") is True
handler.chat_answer = "I cannot answer that."
assert sup.touch_generate("m") is False
def test_touch_generate_scans_reasoning_content(stub_server, tmp_path):
"""Reasoning models answer inside reasoning_content (receipted pitfall)."""
port, handler = stub_server
sup = _make_supervisor(tmp_path, port)
class ReasoningHandler(handler): # type: ignore[valid-type]
def do_POST(self): # noqa: N802
if self.path == "/v1/chat/completions":
self._send(200, {"choices": [{"message": {
"role": "assistant", "content": "",
"reasoning_content": "The capital of France is Paris."}}]})
else:
self._send(404, {})
# Swap handler class on the live stub server socket is overkill; just
# verify the scan logic path via the normal handler with empty content.
handler.chat_answer = ""
assert sup.touch_generate("m") is False # empty content, no reasoning field
def test_ensure_model_ready_unknown_model_raises(stub_server, tmp_path):
port, handler = stub_server
handler.models = {"data": [{"id": "present", "status": {"value": "unloaded"}}]}
sup = _make_supervisor(tmp_path, port)
with pytest.raises(KeyError):
sup.ensure_model_ready("absent")
def test_is_idle_requires_no_busy_slots_and_zero_processing(stub_server, tmp_path):
port, handler = stub_server
sup = _make_supervisor(tmp_path, port)
# Router telemetry is per-child (?model=); a loaded model must exist for
# is_idle to have anything to check.
handler.models = {"data": [{"id": "m", "status": {"value": "loaded"}}]}
handler.slots = [{"id": 0, "is_processing": False}]
handler.requests_processing = 0
assert sup.is_idle() is True
handler.slots = [{"id": 0, "is_processing": True}]
assert sup.is_idle() is False
handler.slots = [{"id": 0, "is_processing": False}]
handler.requests_processing = 2
assert sup.is_idle() is False
def test_base_url_dials_loopback_ip_never_localhost(tmp_path):
"""C12: localhost costs ~2s/request on Windows."""
sup = _make_supervisor(tmp_path, 9999)
assert "127.0.0.1" in sup.base_url
assert "localhost" not in sup.base_url
# ── provider integration (existing alias mechanism, no new plugin) ──
def test_llamacpp_aliases_route_to_custom_profile():
"""Design + maintainer direction: llamacpp fits the EXISTING provider
mechanism — the aliases already resolve to the keyless custom profile;
no parallel provider plugin exists."""
from providers import get_provider_profile
for alias in ("llamacpp", "llama.cpp", "llama-cpp"):
profile = get_provider_profile(alias)
assert profile is not None, alias
assert profile.name == "custom"
assert profile.env_vars == () # credential is reachability
def test_llamacpp_endpoint_resolution_prefers_managed(tmp_path, monkeypatch, stub_server):
"""provider: llamacpp with a live managed server resolves to it,
api-key included."""
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
port, handler = stub_server
from hermes_cli.local_runtime import endpoint as ep
from hermes_cli.local_runtime.supervisor import state_path
_write_current_process_state(
state_path(), base_url=f"http://127.0.0.1:{port}/v1", api_key="sk-managed")
resolved = ep.resolve_llamacpp_endpoint()
assert resolved == {"base_url": f"http://127.0.0.1:{port}/v1", "api_key": "sk-managed"}
def test_llamacpp_endpoint_stale_state_falls_through(tmp_path, monkeypatch):
"""A crashed-without-cleanup state file (dead pid, dead endpoint) must
not blackhole requests: state ignored -> detection (none here) -> None."""
import socket
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
with socket.socket() as s:
s.bind(("127.0.0.1", 0))
dead_port = s.getsockname()[1]
from hermes_cli.local_runtime import endpoint as ep
from hermes_cli.local_runtime.detect import DEFAULT_PROBE_PORTS
from hermes_cli.local_runtime.supervisor import state_path
state_path().parent.mkdir(parents=True, exist_ok=True)
state_path().write_text(json.dumps({
"base_url": f"http://127.0.0.1:{dead_port}/v1", "api_key": "sk-x", "pid": 1,
}), encoding="utf-8")
monkeypatch.setattr(ep, "_pid_alive", lambda pid: False)
# Keep detection away from any real server on 8080 during the test.
monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS",
(dead_port,))
assert ep.resolve_llamacpp_endpoint() is None
assert DEFAULT_PROBE_PORTS # (import kept honest)
def test_llamacpp_dead_server_raises_friendly_error(tmp_path, monkeypatch):
"""A llamacpp send with no server must say WHY in user terms, not fall
through to the generic custom path (which lands on a cloud provider
with a placeholder key and surfaces as a baffling '401 Invalid API
key'). Message tracks the off switch: enabled = probably starting;
disabled = the user turned it off."""
import pytest
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli import runtime_provider as rp
monkeypatch.setattr(
"hermes_cli.local_runtime.endpoint.resolve_llamacpp_endpoint",
lambda *a, **k: None)
monkeypatch.setattr(
"hermes_cli.config.load_config",
lambda: {"local_runtime": {"enabled": False}})
with pytest.raises(ValueError, match="turned off"):
rp._resolve_named_custom_runtime(requested_provider="llamacpp")
monkeypatch.setattr(
"hermes_cli.config.load_config",
lambda: {"local_runtime": {"enabled": True}})
with pytest.raises(ValueError, match="isn't running"):
rp._resolve_named_custom_runtime(requested_provider="llamacpp")
# An explicit base_url is the user pointing at a specific server —
# that path keeps its own error reporting, never this one.
result = rp._resolve_named_custom_runtime(
requested_provider="llamacpp",
explicit_base_url="http://127.0.0.1:9999/v1")
assert result is None or result.get("base_url", "").startswith("http://127.0.0.1:9999")
def test_llamacpp_endpoint_starting_server_resolves(tmp_path, monkeypatch):
"""The restart race: state written at spawn, server not yet healthy,
supervisor child alive — resolution must return the endpoint (a
STARTING server is configured, not missing credentials; this exact
race threw the app back to onboarding on the first restart test)."""
import socket
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
with socket.socket() as s:
s.bind(("127.0.0.1", 0))
not_listening = s.getsockname()[1]
from hermes_cli.local_runtime import endpoint as ep
from hermes_cli.local_runtime.supervisor import state_path
state_path().parent.mkdir(parents=True, exist_ok=True)
state_path().write_text(json.dumps({
"base_url": f"http://127.0.0.1:{not_listening}/v1",
"api_key": "sk-starting", "pid": 4242,
}), encoding="utf-8")
monkeypatch.setattr(
"hermes_cli.local_runtime.recovery.legacy_recorded_process", lambda state: object())
resolved = ep.resolve_llamacpp_endpoint()
assert resolved is not None
assert resolved["api_key"] == "sk-starting"
def test_llamacpp_endpoint_waits_for_boot_in_flight(tmp_path, monkeypatch):
"""The SECOND restart race (no state file at all yet): a fresh backend's
readiness probe resolves before the lifespan boot thread has even
spawned the server. With the runtime enabled+installed, resolution must
poll briefly and pick up the state file when the boot thread writes it
— not report unconfigured."""
import threading
import time as _time
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime import endpoint as ep
from hermes_cli.local_runtime.supervisor import state_path
# Boot is in flight: runtime enabled + binary installed.
monkeypatch.setattr(ep, "_boot_in_flight", lambda config: True)
monkeypatch.setattr(
"hermes_cli.local_runtime.recovery.legacy_recorded_process", lambda state: object())
# Nothing detected externally.
monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS", ())
def _late_writer():
_time.sleep(0.6)
state_path().parent.mkdir(parents=True, exist_ok=True)
state_path().write_text(json.dumps({
"base_url": "http://127.0.0.1:59999/v1",
"api_key": "sk-boot", "pid": 777,
}), encoding="utf-8")
t = threading.Thread(target=_late_writer)
t.start()
try:
resolved = ep.resolve_llamacpp_endpoint(wait_for_boot_s=5.0)
finally:
t.join()
assert resolved is not None
assert resolved["api_key"] == "sk-boot"
def test_resolution_kicks_boot_when_no_thread_is_booting(tmp_path, monkeypatch):
"""The dead-router-mid-flight case: runtime enabled+installed, but no
state file and NO lifespan boot thread running (the router died after
backend start — tree-killed with a stale backend, or the stable port
was owned by another install and the ownership guard refused it).
Resolution must not just wait for a boot that nobody is doing — it
kicks ensure_local_runtime itself and picks up the state file that
boot writes."""
import time as _time
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime import bootstrap as bs
from hermes_cli.local_runtime import endpoint as ep
from hermes_cli.local_runtime.supervisor import state_path
monkeypatch.setattr(ep, "_boot_in_flight", lambda config: True)
monkeypatch.setattr(
"hermes_cli.local_runtime.recovery.legacy_recorded_process", lambda state: object())
monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS", ())
def _fake_ensure(config, force=False):
_time.sleep(0.3) # a real spawn takes a moment
state_path().parent.mkdir(parents=True, exist_ok=True)
state_path().write_text(json.dumps({
"base_url": "http://127.0.0.1:59998/v1",
"api_key": "sk-kicked", "pid": 778,
}), encoding="utf-8")
monkeypatch.setattr(bs, "ensure_local_runtime", _fake_ensure)
resolved = ep.resolve_llamacpp_endpoint(config={}, wait_for_boot_s=5.0)
assert resolved is not None
assert resolved["api_key"] == "sk-kicked"
def test_boot_in_flight_real_gate(tmp_path, monkeypatch):
"""_boot_in_flight exercised FOR REAL (the previous regression test
monkeypatched it — and the real one threw TypeError on every call,
silently disabling the boot wait). Enabled + installed PM engine
-> True; either missing -> False."""
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime import endpoint as ep
monkeypatch.setattr("hermes_cli.local_runtime.binaries.installed_engine", lambda: None)
enabled = {"local_runtime": {"enabled": True}}
assert ep._boot_in_flight(enabled) is False
monkeypatch.setattr("hermes_cli.local_runtime.binaries.installed_engine", lambda: object())
assert ep._boot_in_flight(enabled) is True
# Disabled -> False even when installed.
assert ep._boot_in_flight({"local_runtime": {"enabled": False}}) is False
def test_idle_sweep_unloads_idle_models(tmp_path, monkeypatch, stub_server):
"""Residency v2 contract: after the idle threshold, idle loaded models
unload — no exemptions; demand reloads anything the user returns to.
Idleness is the C5 contract (no busy slots)."""
port, handler = stub_server
handler.models = {"data": [
{"id": "model-a", "status": {"value": "loaded"}},
{"id": "model-b", "status": {"value": "loaded"}},
]}
handler.slots = [] # everyone idle per C5
handler.requests_processing = 0
handler.unloaded = []
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
sup = LlamaServerSupervisor(tmp_path / "i", tmp_path / "m", port=port)
t0 = 1000.0
# First sweep: starts the idle clocks, nothing unloads yet.
assert sup.sweep_idle(now=t0) == []
# Before the threshold: still nothing.
assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S - 1) == []
# Past the threshold: both idle models unload.
assert sorted(sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S + 1)) == ["model-a", "model-b"]
assert sorted(handler.unloaded) == ["model-a", "model-b"]
def test_idle_sweep_busy_model_resets_clock(tmp_path, monkeypatch, stub_server):
"""A model seen busy (C5: busy slot) restarts its idle clock — an
active conversation never trips the sweep."""
port, handler = stub_server
handler.models = {"data": [{"id": "side-m", "status": {"value": "loaded"}}]}
handler.unloaded = []
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
sup = LlamaServerSupervisor(tmp_path / "i", tmp_path / "m", port=port)
t0 = 1000.0
handler.slots = [] # idle: clock starts
assert sup.sweep_idle(now=t0) == []
handler.slots = [{"is_processing": True}] # busy mid-window
assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S) == []
handler.slots = [] # idle again: clock restarts, not expired
assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S + 10) == []
assert handler.unloaded == []
def test_idle_sweep_probe_failure_keeps_clock(tmp_path, monkeypatch, stub_server):
"""A failed telemetry probe is not activity: /slots or /metrics errors must keep the
idle clock instead of resetting it, so one flaky probe per sweep can't pin a resident
model (and its VRAM) for hours."""
port, handler = stub_server
handler.models = {"data": [{"id": "stuck-m", "status": {"value": "loaded"}}]}
handler.slots = []
handler.unloaded = []
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
sup = LlamaServerSupervisor(tmp_path / "i", tmp_path / "m", port=port)
t0 = 1000.0
assert sup.sweep_idle(now=t0) == [] # clock starts
handler.metrics_error = 500 # probe fails mid-window
assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S + 1) == [] # kept, not reset
assert handler.unloaded == []
handler.metrics_error = 0 # telemetry recovers
# The clock survived the failures: unload happens at the first healthy sweep.
assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S + 2) == ["stuck-m"]
assert handler.unloaded == ["stuck-m"]
def test_idle_sweep_busy_after_probe_failure_still_resets_clock(
tmp_path, monkeypatch, stub_server
):
"""A kept clock must not mask real activity: a confirmed busy slot still resets it."""
port, handler = stub_server
handler.models = {"data": [{"id": "m", "status": {"value": "loaded"}}]}
handler.unloaded = []
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
sup = LlamaServerSupervisor(tmp_path / "i", tmp_path / "m", port=port)
t0 = 1000.0
assert sup.sweep_idle(now=t0) == [] # clock starts
handler.slots_error = 500 # probe fails: clock kept
assert sup.sweep_idle(now=t0 + 100) == []
handler.slots_error = 0
handler.slots = [{"is_processing": True}] # confirmed busy: clock resets
assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S + 5) == []
handler.slots = []
# Fresh clock since the busy sighting — not the original t0 one.
assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S + 6) == []
assert handler.unloaded == []
def test_staged_models_requires_every_split_part(tmp_path, monkeypatch):
"""A split GGUF mid-download must NOT count as staged: the picker, the
catalog's 'downloaded' flag, and the router's model list all read
staged_models(), and a first part with missing continuations is not
servable. Single files and complete splits count; continuation parts
never count as their own model."""
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
import hermes_cli.local_runtime.bootstrap as bs
mdir = bs.models_dir()
mdir.mkdir(parents=True, exist_ok=True)
(mdir / "Single-Q4_K_M.gguf").touch()
# Complete split: both parts present.
(mdir / "Whole-Q4-00001-of-00002.gguf").touch()
(mdir / "Whole-Q4-00002-of-00002.gguf").touch()
# Mid-download split: first part only, of three.
(mdir / "Partial-Q4-00001-of-00003.gguf").touch()
assert bs.staged_model_ids() == ["Single-Q4_K_M", "Whole-Q4"]
def test_bootstrap_skips_boot_with_no_staged_models(tmp_path, monkeypatch):
"""Residency: enabled + installed but zero staged models -> no server
boot (nothing to serve; the walked-away story)."""
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
import hermes_cli.local_runtime.bootstrap as bs
monkeypatch.setattr(bs, "_SUPERVISOR", None)
called = {"spawn": False}
def _boom(*a, **k):
called["spawn"] = True
raise AssertionError("must not reach install/spawn")
monkeypatch.setattr("hermes_cli.local_runtime.binaries.installed_engine", _boom)
result = bs.ensure_local_runtime({"local_runtime": {"enabled": True}})
assert result is None
assert called["spawn"] is False
def test_endpoint_identity_stable_across_supervisor_instances(tmp_path, monkeypatch):
"""Round-7 contract: base_url AND api_key survive a restart as a unit.
Two supervisor constructions (= two backend boots) must agree on both —
sessions persist the resolved pair, so either piece rotating strands
every resumed session (connection error / HTTP 401)."""
import socket as _socket
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime import supervisor as sup_mod
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
# A test-owned default port: the production default may legitimately be
# held by a live managed server on the dev machine.
with _socket.socket() as s:
s.bind(("127.0.0.1", 0))
test_port = s.getsockname()[1]
monkeypatch.setattr(sup_mod, "_DEFAULT_PORT", test_port)
first = LlamaServerSupervisor(tmp_path / "install", tmp_path / "models")
second = LlamaServerSupervisor(tmp_path / "install", tmp_path / "models")
assert first.api_key == second.api_key
assert len(first.api_key) >= 16
assert first.port == second.port == test_port
# The key is persisted, not per-process state.
key_file = tmp_path / ".hermes" / "runtimes" / "llamacpp" / ".api_key"
assert key_file.exists()
assert key_file.read_text(encoding="utf-8").strip() == first.api_key
def test_llamacpp_endpoint_no_wait_when_not_enabled(tmp_path, monkeypatch):
"""No boot in flight (runtime disabled/uninstalled): resolution returns
None promptly instead of burning the wait budget."""
import time as _time
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime import endpoint as ep
monkeypatch.setattr(ep, "_boot_in_flight", lambda config: False)
monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS", ())
t0 = _time.monotonic()
assert ep.resolve_llamacpp_endpoint(wait_for_boot_s=8.0) is None
assert _time.monotonic() - t0 < 3.0
def test_switch_model_explicit_llamacpp_provider(tmp_path, monkeypatch, stub_server):
"""The desktop dropdown path: switch_model(explicit_provider='llamacpp')
must resolve the managed provider — not 'Unknown provider' (the
desktop-review symptom). E2E through the real pipeline against a stub server."""
port, handler = stub_server
handler.models = {"data": [{"id": "stub-model-a", "owned_by": "llamacpp"}]}
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime.supervisor import state_path
_write_current_process_state(
state_path(), base_url=f"http://127.0.0.1:{port}/v1", api_key="sk-managed")
from hermes_cli.model_switch import switch_model
result = switch_model(
"stub-model-a",
current_provider="nous",
current_model="Hermes-4.5",
current_base_url="",
explicit_provider="llamacpp",
)
assert result.success, result.error_message
assert f"127.0.0.1:{port}" in (result.base_url or "")
assert result.api_key == "sk-managed"
def test_runtime_provider_seam_llamacpp_alias(tmp_path, monkeypatch, stub_server):
"""End to end through the REAL resolver: provider='llamacpp' with no
base_url lands on the managed endpoint with source='local-runtime'."""
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
port, handler = stub_server
from hermes_cli.local_runtime.supervisor import state_path
_write_current_process_state(
state_path(), base_url=f"http://127.0.0.1:{port}/v1", api_key="sk-managed")
from hermes_cli.runtime_provider import _resolve_named_custom_runtime
runtime = _resolve_named_custom_runtime(requested_provider="llamacpp")
assert runtime is not None
assert runtime["source"] == "local-runtime"
assert runtime["base_url"] == f"http://127.0.0.1:{port}/v1"
assert runtime["api_key"] == "sk-managed"
assert runtime["provider"] == "custom"
def test_configured_llamacpp_provider_wins_over_managed_alias(tmp_path, monkeypatch):
"""A providers.llamacpp endpoint is explicit configuration, not a managed-runtime request (#116143)."""
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
monkeypatch.setattr(
"hermes_cli.config.load_config",
lambda: {
"providers": {
"llamacpp": {
"base_url": "http://127.0.0.1:8081/v1",
"default_model": "configured-model",
}
}
},
)
def _managed_alias_must_not_run(*args, **kwargs):
raise AssertionError("configured providers.llamacpp must resolve before managed detection")
monkeypatch.setattr(
"hermes_cli.local_runtime.endpoint.resolve_llamacpp_endpoint",
_managed_alias_must_not_run,
)
from hermes_cli.runtime_provider import _resolve_named_custom_runtime
runtime = _resolve_named_custom_runtime(requested_provider="llamacpp")
assert runtime is not None
assert runtime["base_url"] == "http://127.0.0.1:8081/v1"
assert runtime["model"] == "configured-model"
assert runtime["source"].startswith("custom_provider:")
def test_runtime_provider_seam_explicit_base_url_wins(tmp_path, monkeypatch):
"""A user-specified base_url must never be overridden by the managed
endpoint — pointing at a specific server means that server."""
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime.supervisor import state_path
state_path().parent.mkdir(parents=True, exist_ok=True)
state_path().write_text(json.dumps({
"base_url": "http://127.0.0.1:1/v1", "api_key": "sk-managed", "pid": 1,
}), encoding="utf-8")
from hermes_cli.runtime_provider import _resolve_named_custom_runtime
runtime = _resolve_named_custom_runtime(
requested_provider="llamacpp",
explicit_base_url="http://127.0.0.1:9999/v1")
assert runtime is not None
assert runtime["base_url"] == "http://127.0.0.1:9999/v1"
assert runtime["source"] != "local-runtime"
def _stage_local_model(home, model_id):
models = home / "models"
models.mkdir(parents=True, exist_ok=True)
(models / f"{model_id}.gguf").write_bytes(b"GGUF")
def test_staged_local_model_resolves_without_a_running_server(tmp_path, monkeypatch):
"""The picker's Local row exists from staged GGUFs alone (contract: selectable before the server
runs — selection starts it through the runtime seam). Selecting one must reach that seam instead
of dying at the provider gate with "Unknown provider 'llamacpp'": the row's id and the resolver's
are one definition, so an id the picker offers always resolves (#116249)."""
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
_stage_local_model(tmp_path / ".hermes", "Qwen3.8-27B-IQ3_S-mtp")
monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS", ())
from hermes_cli.providers import LLAMACPP_PROVIDER_ID, resolve_provider_full
pdef = resolve_provider_full(LLAMACPP_PROVIDER_ID, {}, [])
assert pdef is not None, "staged model, but the picker's own provider id does not resolve"
assert pdef.id == LLAMACPP_PROVIDER_ID
from hermes_cli.model_switch import switch_model
result = switch_model("Qwen3.8-27B-IQ3_S-mtp", current_provider="nous",
current_model="Hermes-4.5", current_base_url="",
explicit_provider=LLAMACPP_PROVIDER_ID)
assert result.success is False # no server anywhere; the seam reports it
error = result.error_message or ""
assert "Unknown provider" not in error
assert "local model server" in error.lower(), error
def test_external_server_on_a_configured_detect_port_is_used(tmp_path, monkeypatch, stub_server):
"""A llama-server the user runs on a fixed non-default port is what `provider: llamacpp` resolves
to when that port is declared in local_runtime.detect_ports — the knob is documented for exactly
this, and every provider path called the endpoint resolver without a config (#116143, #116249)."""
port, handler = stub_server
handler.props = {"build_info": "b10964-test", "model_path": "/models/ext-model.gguf",
"default_generation_settings": {"n_ctx": 4096}}
handler.models = {"data": [{"id": "ext-model", "owned_by": "llamacpp"}]}
home = tmp_path / ".hermes"
monkeypatch.setenv("HERMES_HOME", str(home))
_stage_local_model(home, "ext-model")
(home / "config.yaml").write_text(
f"local_runtime:\n enabled: false\n detect_ports: [{port}]\n", encoding="utf-8")
monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS", ())
from hermes_cli.model_switch import switch_model
result = switch_model("ext-model", current_provider="nous", current_model="Hermes-4.5",
current_base_url="", explicit_provider="llamacpp")
assert result.success, result.error_message
assert result.base_url == f"http://127.0.0.1:{port}/v1"
# ── bootstrap contracts ──────────────────────────────────────
def test_bootstrap_disabled_is_noop(tmp_path, monkeypatch):
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime import bootstrap
monkeypatch.setattr(bootstrap, "_SUPERVISOR", None)
assert bootstrap.ensure_local_runtime({"local_runtime": {"enabled": False}}) is None
assert bootstrap.ensure_local_runtime({}) is None
assert bootstrap.ensure_local_runtime(None) is None
def test_bootstrap_reuses_running_server(tmp_path, monkeypatch, stub_server):
"""A live state file (another process supervising) short-circuits the
install/spawn path entirely."""
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
port, handler = stub_server
from hermes_cli.local_runtime import bootstrap
from hermes_cli.local_runtime.supervisor import state_path
monkeypatch.setattr(bootstrap, "_SUPERVISOR", None)
state_path().parent.mkdir(parents=True, exist_ok=True)
state_path().write_text(json.dumps({
"base_url": f"http://127.0.0.1:{port}/v1", "api_key": "k", "pid": os.getpid(),
}), encoding="utf-8")
called = []
monkeypatch.setattr(
"hermes_cli.local_runtime.binaries.installed_engine",
lambda *a, **k: called.append(1))
assert bootstrap.ensure_local_runtime({"local_runtime": {"enabled": True}}) is None
assert called == []
def test_bootstrap_failure_never_raises(tmp_path, monkeypatch):
"""Session start must survive a broken runtime: failures log + return
None, chat falls back to configured providers."""
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime import bootstrap
monkeypatch.setattr(bootstrap, "_SUPERVISOR", None)
monkeypatch.setattr(bootstrap, "_detect_gpu_vendor", lambda: None)
models = bootstrap.models_dir()
models.mkdir(parents=True, exist_ok=True)
(models / "test.gguf").touch()
def boom(*a, **k):
raise RuntimeError("no network")
monkeypatch.setattr(
"hermes_cli.local_runtime.binaries.installed_engine", boom)
result = bootstrap.ensure_local_runtime({"local_runtime": {"enabled": True}})
assert result is None # no exception escaped
def test_ensure_local_runtime_serializes_racing_callers(tmp_path, monkeypatch):
"""Cross-process boot race (#116682): two backends starting in the same second must not
both spawn a router on the stable port. Neither caller here ever sees the other's
in-process ``_SUPERVISOR`` (each opens its own fd for the boot lock, exactly like two
separate OS processes would) — only the cross-process file lock can serialize them."""
import time as _time
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime import bootstrap
from hermes_cli.local_runtime import supervisor as sup_mod
monkeypatch.setattr(bootstrap, "_SUPERVISOR", None)
mdir = bootstrap.models_dir()
mdir.mkdir(parents=True, exist_ok=True)
(mdir / "stub-Q4_K_M.gguf").touch()
monkeypatch.setattr(bootstrap, "_generate_presets", lambda *a, **k: None)
monkeypatch.setattr(bootstrap, "_presets_stale", lambda: False)
monkeypatch.setattr(bootstrap, "_detect_gpu_vendor", lambda: None)
from hermes_cli.local_runtime.binaries import Engine
monkeypatch.setattr("hermes_cli.local_runtime.binaries.installed_engine",
lambda backend: Engine("cpu", "b1", tmp_path / "llama-server"))
spawns = []
class _FakeSupervisor:
def __init__(self, *a, **k):
self.port = 18434
self.api_key = "k"
self.proc = None
@property
def base_url(self):
return f"http://127.0.0.1:{self.port}/v1"
def start(self, timeout_s=120):
spawns.append(1)
_time.sleep(0.3) # widen the window the other caller races into
# A bare legacy ``{pid}`` record is no longer adoptable (a live PID is not evidence);
# publish what a real router publishes so the second caller can adopt it.
_write_current_process_state(
sup_mod.state_path(), base_url=self.base_url, api_key=self.api_key)
monkeypatch.setattr(sup_mod, "LlamaServerSupervisor", _FakeSupervisor)
barrier = threading.Barrier(2)
results = []
def _boot():
barrier.wait()
results.append(bootstrap.ensure_local_runtime({"local_runtime": {"enabled": True}}))
threads = [threading.Thread(target=_boot) for _ in range(2)]
for t in threads:
t.start()
for t in threads:
t.join(timeout=10)
assert len(spawns) == 1, (
"both racing callers spawned a router instead of the second adopting the "
"first's published state (#116682)")
def test_ensure_local_runtime_proceeds_when_boot_lock_is_unwritable(tmp_path, monkeypatch, caplog):
"""The boot lock lives outside the body's ``try/except``: an unwritable runtimes dir must
degrade to a warning and an unlocked boot, never an OSError out of session start."""
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
from hermes_cli.local_runtime import bootstrap
monkeypatch.setattr(bootstrap, "_SUPERVISOR", None)
mdir = bootstrap.models_dir()
mdir.mkdir(parents=True, exist_ok=True)
(mdir / "stub-Q4_K_M.gguf").touch()
blocker = tmp_path / "not-a-dir"
blocker.write_text("", encoding="utf-8")
monkeypatch.setattr(bootstrap, "runtimes_root", lambda: blocker / "runtimes") # mkdir -> OSError
monkeypatch.setattr("hermes_cli.local_runtime.endpoint._state_endpoint", lambda: None)
monkeypatch.setattr("hermes_cli.local_runtime.binaries.installed_engine", lambda backend: None)
with caplog.at_level(logging.WARNING, logger=bootstrap.logger.name):
result = bootstrap.ensure_local_runtime({"local_runtime": {"enabled": True}})
assert result is None # no exception escaped
assert any("boot lock unavailable" in rec.getMessage() for rec in caplog.records)