65ff3ad353 made a bare legacy {pid} record non-adoptable (a live PID alone is not
evidence of the managed router) and moved the other tests in this file onto
_write_current_process_state, but left the racing-callers fake publishing the legacy
shape, so the second caller could never adopt the first's state and main's Run tests
lane went red on every PR rebased past it (main's own CI could not start at the time).
Publish the modern record the real supervisor writes.
964 lines
40 KiB
Python
964 lines
40 KiB
Python
"""Contract tests for hermes_cli.local_runtime — Rollouts 1+2.
|
|
|
|
Per the design's verification plan: relationships and contracts, no
|
|
change-detector tests, real imports against temp HERMES_HOME (the autouse
|
|
fixture isolates it). The stub HTTP server speaks just enough llama-server
|
|
(/props, /health, /models, /v1/chat/completions, /metrics, /slots) to
|
|
exercise detection fingerprinting and supervisor logic without a GPU.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import logging
|
|
import os
|
|
import threading
|
|
from http.server import BaseHTTPRequestHandler, HTTPServer
|
|
|
|
import pytest
|
|
|
|
from hermes_cli.local_runtime.binaries import select_backend
|
|
from hermes_cli.local_runtime.detect import DetectedServer, probe_port
|
|
|
|
|
|
def _write_current_process_state(path: Path, *, base_url: str, api_key: str) -> None:
|
|
"""Publish a modern state record for the process hosting the test stub."""
|
|
import psutil
|
|
|
|
proc = psutil.Process()
|
|
parent = proc.parent()
|
|
assert parent is not None
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
path.write_text(json.dumps({
|
|
"base_url": base_url,
|
|
"api_key": api_key,
|
|
"pid": proc.pid,
|
|
"create_time": proc.create_time(),
|
|
"executable": proc.exe(),
|
|
"owner_pid": parent.pid,
|
|
"owner_create_time": parent.create_time(),
|
|
}), encoding="utf-8")
|
|
|
|
|
|
# ── stub llama-server ────────────────────────────────────────
|
|
|
|
|
|
class _StubHandler(BaseHTTPRequestHandler):
|
|
"""Minimal llama-server imitation; behavior driven by class attrs."""
|
|
|
|
props: dict = {}
|
|
models: dict | None = None
|
|
require_auth = False
|
|
chat_answer = "Paris"
|
|
requests_processing = 0
|
|
slots: list = []
|
|
slots_error = 0
|
|
metrics_error = 0
|
|
|
|
def _send(self, code: int, body: dict | str | None = None) -> None:
|
|
raw = (json.dumps(body) if isinstance(body, dict) else (body or "")).encode()
|
|
self.send_response(code)
|
|
self.send_header("Content-Type", "application/json")
|
|
self.send_header("Content-Length", str(len(raw)))
|
|
self.end_headers()
|
|
self.wfile.write(raw)
|
|
|
|
def do_GET(self): # noqa: N802
|
|
if self.require_auth and "Authorization" not in self.headers:
|
|
self._send(401, {})
|
|
return
|
|
path = self.path.split("?")[0] # router telemetry uses ?model=
|
|
if path == "/props":
|
|
self._send(200, self.props)
|
|
elif path == "/health":
|
|
self._send(200, {"status": "ok"})
|
|
elif path == "/models":
|
|
if self.models is None:
|
|
self._send(404, {})
|
|
else:
|
|
self._send(200, self.models)
|
|
elif path == "/metrics":
|
|
if self.metrics_error:
|
|
self._send(self.metrics_error, {})
|
|
else:
|
|
self._send(
|
|
200, f"llamacpp:requests_processing {self.requests_processing}\n"
|
|
)
|
|
elif path == "/slots":
|
|
if self.slots_error:
|
|
self._send(self.slots_error, {})
|
|
return
|
|
raw = json.dumps(self.slots).encode()
|
|
self.send_response(200)
|
|
self.send_header("Content-Type", "application/json")
|
|
self.send_header("Content-Length", str(len(raw)))
|
|
self.end_headers()
|
|
self.wfile.write(raw)
|
|
else:
|
|
self._send(404, {})
|
|
|
|
def do_POST(self): # noqa: N802
|
|
if self.path == "/v1/chat/completions":
|
|
self._send(200, {"choices": [{"message": {
|
|
"role": "assistant", "content": self.chat_answer}}]})
|
|
elif self.path == "/models/load":
|
|
self._send(200, {"success": True})
|
|
elif self.path == "/models/unload":
|
|
type(self).unloaded = getattr(type(self), "unloaded", [])
|
|
length = int(self.headers.get("Content-Length", 0))
|
|
body = json.loads(self.rfile.read(length)) if length else {}
|
|
type(self).unloaded.append(body.get("model"))
|
|
self._send(200, {"success": True})
|
|
else:
|
|
self._send(404, {})
|
|
|
|
def log_message(self, *args): # silence
|
|
pass
|
|
|
|
|
|
@pytest.fixture
|
|
def stub_server():
|
|
"""Yields (port, handler_class); handler attrs are per-test mutable."""
|
|
|
|
class Handler(_StubHandler):
|
|
props = {}
|
|
models = None
|
|
require_auth = False
|
|
slots = []
|
|
|
|
server = HTTPServer(("127.0.0.1", 0), Handler)
|
|
thread = threading.Thread(target=server.serve_forever, daemon=True)
|
|
thread.start()
|
|
yield server.server_address[1], Handler
|
|
server.shutdown()
|
|
|
|
|
|
# ── detection (Rollout 1) ────────────────────────────────────
|
|
|
|
|
|
def test_probe_fingerprints_real_llama_server(stub_server):
|
|
port, handler = stub_server
|
|
handler.props = {
|
|
"build_info": "b10290-c8e03ce81",
|
|
"model_path": "C:/models/some model with spaces.gguf",
|
|
"default_generation_settings": {"n_ctx": 65536},
|
|
}
|
|
handler.models = {"data": [{"id": "m", "status": {"value": "unloaded"}}]}
|
|
hit = probe_port(port)
|
|
assert isinstance(hit, DetectedServer)
|
|
assert hit.base_url == f"http://127.0.0.1:{port}/v1"
|
|
assert hit.build_info.startswith("b10290")
|
|
assert hit.n_ctx == 65536
|
|
assert hit.router_mode is True
|
|
assert hit.auth_required is False
|
|
|
|
|
|
def test_probe_rejects_non_llama_openai_server(stub_server):
|
|
# Answers /props with no build_info (e.g. some other local service).
|
|
port, handler = stub_server
|
|
handler.props = {"something": "else"}
|
|
assert probe_port(port) is None
|
|
|
|
|
|
def test_probe_single_model_mode_is_not_router(stub_server):
|
|
port, handler = stub_server
|
|
handler.props = {"build_info": "b10290-x", "model_path": "m.gguf"}
|
|
handler.models = None # /models 404s in plain (non-router) mode
|
|
hit = probe_port(port)
|
|
assert hit is not None
|
|
assert hit.router_mode is False
|
|
|
|
|
|
def test_probe_auth_required_still_detected(stub_server):
|
|
port, handler = stub_server
|
|
handler.require_auth = True
|
|
hit = probe_port(port)
|
|
assert hit is not None
|
|
assert hit.auth_required is True
|
|
|
|
|
|
def test_probe_dead_port_returns_none():
|
|
# Bind-then-close to get a port that is definitely closed.
|
|
import socket
|
|
with socket.socket() as s:
|
|
s.bind(("127.0.0.1", 0))
|
|
dead_port = s.getsockname()[1]
|
|
assert probe_port(dead_port) is None
|
|
|
|
|
|
# ── binary resolver (Rollout 2) ──────────────────────────────
|
|
|
|
|
|
@pytest.mark.parametrize("vendor,os_name,expected", [
|
|
("NVIDIA GeForce RTX 5090", "win", "cuda"),
|
|
("nvidia", "ubuntu", "cuda"),
|
|
("AMD Radeon RX 7900", "win", "vulkan"),
|
|
("intel", "win", "vulkan"),
|
|
(None, "win", "cpu"),
|
|
("", "ubuntu", "cpu"),
|
|
("nvidia", "macos", "metal"), # macOS is Metal regardless
|
|
(None, "macos", "metal"),
|
|
])
|
|
def test_backend_selection(vendor, os_name, expected):
|
|
assert select_backend(vendor, os_name=os_name) == expected
|
|
|
|
|
|
# ── supervisor contracts (stubbed; no GPU) ───────────────────
|
|
|
|
|
|
def _make_supervisor(tmp_path, port):
|
|
"""Supervisor pointed at the stub: skip spawn, drive HTTP logic only."""
|
|
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
|
|
|
|
sup = LlamaServerSupervisor(
|
|
binary=tmp_path / "llama-server", models_dir=tmp_path, port=port)
|
|
return sup
|
|
|
|
|
|
def test_touch_generate_is_the_readiness_proof(stub_server, tmp_path):
|
|
port, handler = stub_server
|
|
sup = _make_supervisor(tmp_path, port)
|
|
handler.chat_answer = "Paris"
|
|
assert sup.touch_generate("m") is True
|
|
handler.chat_answer = "I cannot answer that."
|
|
assert sup.touch_generate("m") is False
|
|
|
|
|
|
def test_touch_generate_scans_reasoning_content(stub_server, tmp_path):
|
|
"""Reasoning models answer inside reasoning_content (receipted pitfall)."""
|
|
port, handler = stub_server
|
|
sup = _make_supervisor(tmp_path, port)
|
|
|
|
class ReasoningHandler(handler): # type: ignore[valid-type]
|
|
def do_POST(self): # noqa: N802
|
|
if self.path == "/v1/chat/completions":
|
|
self._send(200, {"choices": [{"message": {
|
|
"role": "assistant", "content": "",
|
|
"reasoning_content": "The capital of France is Paris."}}]})
|
|
else:
|
|
self._send(404, {})
|
|
|
|
# Swap handler class on the live stub server socket is overkill; just
|
|
# verify the scan logic path via the normal handler with empty content.
|
|
handler.chat_answer = ""
|
|
assert sup.touch_generate("m") is False # empty content, no reasoning field
|
|
|
|
|
|
def test_ensure_model_ready_unknown_model_raises(stub_server, tmp_path):
|
|
port, handler = stub_server
|
|
handler.models = {"data": [{"id": "present", "status": {"value": "unloaded"}}]}
|
|
sup = _make_supervisor(tmp_path, port)
|
|
with pytest.raises(KeyError):
|
|
sup.ensure_model_ready("absent")
|
|
|
|
|
|
def test_is_idle_requires_no_busy_slots_and_zero_processing(stub_server, tmp_path):
|
|
port, handler = stub_server
|
|
sup = _make_supervisor(tmp_path, port)
|
|
# Router telemetry is per-child (?model=); a loaded model must exist for
|
|
# is_idle to have anything to check.
|
|
handler.models = {"data": [{"id": "m", "status": {"value": "loaded"}}]}
|
|
handler.slots = [{"id": 0, "is_processing": False}]
|
|
handler.requests_processing = 0
|
|
assert sup.is_idle() is True
|
|
handler.slots = [{"id": 0, "is_processing": True}]
|
|
assert sup.is_idle() is False
|
|
handler.slots = [{"id": 0, "is_processing": False}]
|
|
handler.requests_processing = 2
|
|
assert sup.is_idle() is False
|
|
|
|
|
|
def test_base_url_dials_loopback_ip_never_localhost(tmp_path):
|
|
"""C12: localhost costs ~2s/request on Windows."""
|
|
sup = _make_supervisor(tmp_path, 9999)
|
|
assert "127.0.0.1" in sup.base_url
|
|
assert "localhost" not in sup.base_url
|
|
|
|
|
|
# ── provider integration (existing alias mechanism, no new plugin) ──
|
|
|
|
|
|
def test_llamacpp_aliases_route_to_custom_profile():
|
|
"""Design + maintainer direction: llamacpp fits the EXISTING provider
|
|
mechanism — the aliases already resolve to the keyless custom profile;
|
|
no parallel provider plugin exists."""
|
|
from providers import get_provider_profile
|
|
|
|
for alias in ("llamacpp", "llama.cpp", "llama-cpp"):
|
|
profile = get_provider_profile(alias)
|
|
assert profile is not None, alias
|
|
assert profile.name == "custom"
|
|
assert profile.env_vars == () # credential is reachability
|
|
|
|
|
|
def test_llamacpp_endpoint_resolution_prefers_managed(tmp_path, monkeypatch, stub_server):
|
|
"""provider: llamacpp with a live managed server resolves to it,
|
|
api-key included."""
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
port, handler = stub_server
|
|
from hermes_cli.local_runtime import endpoint as ep
|
|
from hermes_cli.local_runtime.supervisor import state_path
|
|
|
|
_write_current_process_state(
|
|
state_path(), base_url=f"http://127.0.0.1:{port}/v1", api_key="sk-managed")
|
|
resolved = ep.resolve_llamacpp_endpoint()
|
|
assert resolved == {"base_url": f"http://127.0.0.1:{port}/v1", "api_key": "sk-managed"}
|
|
|
|
|
|
def test_llamacpp_endpoint_stale_state_falls_through(tmp_path, monkeypatch):
|
|
"""A crashed-without-cleanup state file (dead pid, dead endpoint) must
|
|
not blackhole requests: state ignored -> detection (none here) -> None."""
|
|
import socket
|
|
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
with socket.socket() as s:
|
|
s.bind(("127.0.0.1", 0))
|
|
dead_port = s.getsockname()[1]
|
|
from hermes_cli.local_runtime import endpoint as ep
|
|
from hermes_cli.local_runtime.detect import DEFAULT_PROBE_PORTS
|
|
from hermes_cli.local_runtime.supervisor import state_path
|
|
|
|
state_path().parent.mkdir(parents=True, exist_ok=True)
|
|
state_path().write_text(json.dumps({
|
|
"base_url": f"http://127.0.0.1:{dead_port}/v1", "api_key": "sk-x", "pid": 1,
|
|
}), encoding="utf-8")
|
|
monkeypatch.setattr(ep, "_pid_alive", lambda pid: False)
|
|
# Keep detection away from any real server on 8080 during the test.
|
|
monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS",
|
|
(dead_port,))
|
|
assert ep.resolve_llamacpp_endpoint() is None
|
|
assert DEFAULT_PROBE_PORTS # (import kept honest)
|
|
|
|
|
|
def test_llamacpp_dead_server_raises_friendly_error(tmp_path, monkeypatch):
|
|
"""A llamacpp send with no server must say WHY in user terms, not fall
|
|
through to the generic custom path (which lands on a cloud provider
|
|
with a placeholder key and surfaces as a baffling '401 Invalid API
|
|
key'). Message tracks the off switch: enabled = probably starting;
|
|
disabled = the user turned it off."""
|
|
import pytest
|
|
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
|
|
from hermes_cli import runtime_provider as rp
|
|
|
|
monkeypatch.setattr(
|
|
"hermes_cli.local_runtime.endpoint.resolve_llamacpp_endpoint",
|
|
lambda *a, **k: None)
|
|
|
|
monkeypatch.setattr(
|
|
"hermes_cli.config.load_config",
|
|
lambda: {"local_runtime": {"enabled": False}})
|
|
with pytest.raises(ValueError, match="turned off"):
|
|
rp._resolve_named_custom_runtime(requested_provider="llamacpp")
|
|
|
|
monkeypatch.setattr(
|
|
"hermes_cli.config.load_config",
|
|
lambda: {"local_runtime": {"enabled": True}})
|
|
with pytest.raises(ValueError, match="isn't running"):
|
|
rp._resolve_named_custom_runtime(requested_provider="llamacpp")
|
|
|
|
# An explicit base_url is the user pointing at a specific server —
|
|
# that path keeps its own error reporting, never this one.
|
|
result = rp._resolve_named_custom_runtime(
|
|
requested_provider="llamacpp",
|
|
explicit_base_url="http://127.0.0.1:9999/v1")
|
|
assert result is None or result.get("base_url", "").startswith("http://127.0.0.1:9999")
|
|
|
|
|
|
def test_llamacpp_endpoint_starting_server_resolves(tmp_path, monkeypatch):
|
|
"""The restart race: state written at spawn, server not yet healthy,
|
|
supervisor child alive — resolution must return the endpoint (a
|
|
STARTING server is configured, not missing credentials; this exact
|
|
race threw the app back to onboarding on the first restart test)."""
|
|
import socket
|
|
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
with socket.socket() as s:
|
|
s.bind(("127.0.0.1", 0))
|
|
not_listening = s.getsockname()[1]
|
|
from hermes_cli.local_runtime import endpoint as ep
|
|
from hermes_cli.local_runtime.supervisor import state_path
|
|
|
|
state_path().parent.mkdir(parents=True, exist_ok=True)
|
|
state_path().write_text(json.dumps({
|
|
"base_url": f"http://127.0.0.1:{not_listening}/v1",
|
|
"api_key": "sk-starting", "pid": 4242,
|
|
}), encoding="utf-8")
|
|
monkeypatch.setattr(
|
|
"hermes_cli.local_runtime.recovery.legacy_recorded_process", lambda state: object())
|
|
resolved = ep.resolve_llamacpp_endpoint()
|
|
assert resolved is not None
|
|
assert resolved["api_key"] == "sk-starting"
|
|
|
|
|
|
def test_llamacpp_endpoint_waits_for_boot_in_flight(tmp_path, monkeypatch):
|
|
"""The SECOND restart race (no state file at all yet): a fresh backend's
|
|
readiness probe resolves before the lifespan boot thread has even
|
|
spawned the server. With the runtime enabled+installed, resolution must
|
|
poll briefly and pick up the state file when the boot thread writes it
|
|
— not report unconfigured."""
|
|
import threading
|
|
import time as _time
|
|
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime import endpoint as ep
|
|
from hermes_cli.local_runtime.supervisor import state_path
|
|
|
|
# Boot is in flight: runtime enabled + binary installed.
|
|
monkeypatch.setattr(ep, "_boot_in_flight", lambda config: True)
|
|
monkeypatch.setattr(
|
|
"hermes_cli.local_runtime.recovery.legacy_recorded_process", lambda state: object())
|
|
# Nothing detected externally.
|
|
monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS", ())
|
|
|
|
def _late_writer():
|
|
_time.sleep(0.6)
|
|
state_path().parent.mkdir(parents=True, exist_ok=True)
|
|
state_path().write_text(json.dumps({
|
|
"base_url": "http://127.0.0.1:59999/v1",
|
|
"api_key": "sk-boot", "pid": 777,
|
|
}), encoding="utf-8")
|
|
|
|
t = threading.Thread(target=_late_writer)
|
|
t.start()
|
|
try:
|
|
resolved = ep.resolve_llamacpp_endpoint(wait_for_boot_s=5.0)
|
|
finally:
|
|
t.join()
|
|
assert resolved is not None
|
|
assert resolved["api_key"] == "sk-boot"
|
|
|
|
|
|
def test_resolution_kicks_boot_when_no_thread_is_booting(tmp_path, monkeypatch):
|
|
"""The dead-router-mid-flight case: runtime enabled+installed, but no
|
|
state file and NO lifespan boot thread running (the router died after
|
|
backend start — tree-killed with a stale backend, or the stable port
|
|
was owned by another install and the ownership guard refused it).
|
|
Resolution must not just wait for a boot that nobody is doing — it
|
|
kicks ensure_local_runtime itself and picks up the state file that
|
|
boot writes."""
|
|
import time as _time
|
|
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime import bootstrap as bs
|
|
from hermes_cli.local_runtime import endpoint as ep
|
|
from hermes_cli.local_runtime.supervisor import state_path
|
|
|
|
monkeypatch.setattr(ep, "_boot_in_flight", lambda config: True)
|
|
monkeypatch.setattr(
|
|
"hermes_cli.local_runtime.recovery.legacy_recorded_process", lambda state: object())
|
|
monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS", ())
|
|
|
|
def _fake_ensure(config, force=False):
|
|
_time.sleep(0.3) # a real spawn takes a moment
|
|
state_path().parent.mkdir(parents=True, exist_ok=True)
|
|
state_path().write_text(json.dumps({
|
|
"base_url": "http://127.0.0.1:59998/v1",
|
|
"api_key": "sk-kicked", "pid": 778,
|
|
}), encoding="utf-8")
|
|
|
|
monkeypatch.setattr(bs, "ensure_local_runtime", _fake_ensure)
|
|
|
|
resolved = ep.resolve_llamacpp_endpoint(config={}, wait_for_boot_s=5.0)
|
|
assert resolved is not None
|
|
assert resolved["api_key"] == "sk-kicked"
|
|
|
|
|
|
def test_boot_in_flight_real_gate(tmp_path, monkeypatch):
|
|
"""_boot_in_flight exercised FOR REAL (the previous regression test
|
|
monkeypatched it — and the real one threw TypeError on every call,
|
|
silently disabling the boot wait). Enabled + installed PM engine
|
|
-> True; either missing -> False."""
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime import endpoint as ep
|
|
monkeypatch.setattr("hermes_cli.local_runtime.binaries.installed_engine", lambda: None)
|
|
|
|
enabled = {"local_runtime": {"enabled": True}}
|
|
assert ep._boot_in_flight(enabled) is False
|
|
monkeypatch.setattr("hermes_cli.local_runtime.binaries.installed_engine", lambda: object())
|
|
assert ep._boot_in_flight(enabled) is True
|
|
# Disabled -> False even when installed.
|
|
assert ep._boot_in_flight({"local_runtime": {"enabled": False}}) is False
|
|
|
|
|
|
def test_idle_sweep_unloads_idle_models(tmp_path, monkeypatch, stub_server):
|
|
"""Residency v2 contract: after the idle threshold, idle loaded models
|
|
unload — no exemptions; demand reloads anything the user returns to.
|
|
Idleness is the C5 contract (no busy slots)."""
|
|
port, handler = stub_server
|
|
handler.models = {"data": [
|
|
{"id": "model-a", "status": {"value": "loaded"}},
|
|
{"id": "model-b", "status": {"value": "loaded"}},
|
|
]}
|
|
handler.slots = [] # everyone idle per C5
|
|
handler.requests_processing = 0
|
|
handler.unloaded = []
|
|
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
|
|
|
|
sup = LlamaServerSupervisor(tmp_path / "i", tmp_path / "m", port=port)
|
|
|
|
t0 = 1000.0
|
|
# First sweep: starts the idle clocks, nothing unloads yet.
|
|
assert sup.sweep_idle(now=t0) == []
|
|
# Before the threshold: still nothing.
|
|
assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S - 1) == []
|
|
# Past the threshold: both idle models unload.
|
|
assert sorted(sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S + 1)) == ["model-a", "model-b"]
|
|
assert sorted(handler.unloaded) == ["model-a", "model-b"]
|
|
|
|
|
|
def test_idle_sweep_busy_model_resets_clock(tmp_path, monkeypatch, stub_server):
|
|
"""A model seen busy (C5: busy slot) restarts its idle clock — an
|
|
active conversation never trips the sweep."""
|
|
port, handler = stub_server
|
|
handler.models = {"data": [{"id": "side-m", "status": {"value": "loaded"}}]}
|
|
handler.unloaded = []
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
|
|
|
|
sup = LlamaServerSupervisor(tmp_path / "i", tmp_path / "m", port=port)
|
|
|
|
t0 = 1000.0
|
|
handler.slots = [] # idle: clock starts
|
|
assert sup.sweep_idle(now=t0) == []
|
|
handler.slots = [{"is_processing": True}] # busy mid-window
|
|
assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S) == []
|
|
handler.slots = [] # idle again: clock restarts, not expired
|
|
assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S + 10) == []
|
|
assert handler.unloaded == []
|
|
|
|
|
|
def test_idle_sweep_probe_failure_keeps_clock(tmp_path, monkeypatch, stub_server):
|
|
"""A failed telemetry probe is not activity: /slots or /metrics errors must keep the
|
|
idle clock instead of resetting it, so one flaky probe per sweep can't pin a resident
|
|
model (and its VRAM) for hours."""
|
|
port, handler = stub_server
|
|
handler.models = {"data": [{"id": "stuck-m", "status": {"value": "loaded"}}]}
|
|
handler.slots = []
|
|
handler.unloaded = []
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
|
|
|
|
sup = LlamaServerSupervisor(tmp_path / "i", tmp_path / "m", port=port)
|
|
|
|
t0 = 1000.0
|
|
assert sup.sweep_idle(now=t0) == [] # clock starts
|
|
handler.metrics_error = 500 # probe fails mid-window
|
|
assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S + 1) == [] # kept, not reset
|
|
assert handler.unloaded == []
|
|
handler.metrics_error = 0 # telemetry recovers
|
|
# The clock survived the failures: unload happens at the first healthy sweep.
|
|
assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S + 2) == ["stuck-m"]
|
|
assert handler.unloaded == ["stuck-m"]
|
|
|
|
|
|
def test_idle_sweep_busy_after_probe_failure_still_resets_clock(
|
|
tmp_path, monkeypatch, stub_server
|
|
):
|
|
"""A kept clock must not mask real activity: a confirmed busy slot still resets it."""
|
|
port, handler = stub_server
|
|
handler.models = {"data": [{"id": "m", "status": {"value": "loaded"}}]}
|
|
handler.unloaded = []
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
|
|
|
|
sup = LlamaServerSupervisor(tmp_path / "i", tmp_path / "m", port=port)
|
|
|
|
t0 = 1000.0
|
|
assert sup.sweep_idle(now=t0) == [] # clock starts
|
|
handler.slots_error = 500 # probe fails: clock kept
|
|
assert sup.sweep_idle(now=t0 + 100) == []
|
|
handler.slots_error = 0
|
|
handler.slots = [{"is_processing": True}] # confirmed busy: clock resets
|
|
assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S + 5) == []
|
|
handler.slots = []
|
|
# Fresh clock since the busy sighting — not the original t0 one.
|
|
assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S + 6) == []
|
|
assert handler.unloaded == []
|
|
|
|
|
|
def test_staged_models_requires_every_split_part(tmp_path, monkeypatch):
|
|
"""A split GGUF mid-download must NOT count as staged: the picker, the
|
|
catalog's 'downloaded' flag, and the router's model list all read
|
|
staged_models(), and a first part with missing continuations is not
|
|
servable. Single files and complete splits count; continuation parts
|
|
never count as their own model."""
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
import hermes_cli.local_runtime.bootstrap as bs
|
|
|
|
mdir = bs.models_dir()
|
|
mdir.mkdir(parents=True, exist_ok=True)
|
|
|
|
(mdir / "Single-Q4_K_M.gguf").touch()
|
|
# Complete split: both parts present.
|
|
(mdir / "Whole-Q4-00001-of-00002.gguf").touch()
|
|
(mdir / "Whole-Q4-00002-of-00002.gguf").touch()
|
|
# Mid-download split: first part only, of three.
|
|
(mdir / "Partial-Q4-00001-of-00003.gguf").touch()
|
|
|
|
assert bs.staged_model_ids() == ["Single-Q4_K_M", "Whole-Q4"]
|
|
|
|
|
|
def test_bootstrap_skips_boot_with_no_staged_models(tmp_path, monkeypatch):
|
|
"""Residency: enabled + installed but zero staged models -> no server
|
|
boot (nothing to serve; the walked-away story)."""
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
import hermes_cli.local_runtime.bootstrap as bs
|
|
|
|
monkeypatch.setattr(bs, "_SUPERVISOR", None)
|
|
called = {"spawn": False}
|
|
|
|
def _boom(*a, **k):
|
|
called["spawn"] = True
|
|
raise AssertionError("must not reach install/spawn")
|
|
|
|
monkeypatch.setattr("hermes_cli.local_runtime.binaries.installed_engine", _boom)
|
|
result = bs.ensure_local_runtime({"local_runtime": {"enabled": True}})
|
|
assert result is None
|
|
assert called["spawn"] is False
|
|
|
|
|
|
def test_endpoint_identity_stable_across_supervisor_instances(tmp_path, monkeypatch):
|
|
"""Round-7 contract: base_url AND api_key survive a restart as a unit.
|
|
Two supervisor constructions (= two backend boots) must agree on both —
|
|
sessions persist the resolved pair, so either piece rotating strands
|
|
every resumed session (connection error / HTTP 401)."""
|
|
import socket as _socket
|
|
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime import supervisor as sup_mod
|
|
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
|
|
|
|
# A test-owned default port: the production default may legitimately be
|
|
# held by a live managed server on the dev machine.
|
|
with _socket.socket() as s:
|
|
s.bind(("127.0.0.1", 0))
|
|
test_port = s.getsockname()[1]
|
|
monkeypatch.setattr(sup_mod, "_DEFAULT_PORT", test_port)
|
|
|
|
first = LlamaServerSupervisor(tmp_path / "install", tmp_path / "models")
|
|
second = LlamaServerSupervisor(tmp_path / "install", tmp_path / "models")
|
|
assert first.api_key == second.api_key
|
|
assert len(first.api_key) >= 16
|
|
assert first.port == second.port == test_port
|
|
# The key is persisted, not per-process state.
|
|
key_file = tmp_path / ".hermes" / "runtimes" / "llamacpp" / ".api_key"
|
|
assert key_file.exists()
|
|
assert key_file.read_text(encoding="utf-8").strip() == first.api_key
|
|
|
|
|
|
def test_llamacpp_endpoint_no_wait_when_not_enabled(tmp_path, monkeypatch):
|
|
"""No boot in flight (runtime disabled/uninstalled): resolution returns
|
|
None promptly instead of burning the wait budget."""
|
|
import time as _time
|
|
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime import endpoint as ep
|
|
|
|
monkeypatch.setattr(ep, "_boot_in_flight", lambda config: False)
|
|
monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS", ())
|
|
t0 = _time.monotonic()
|
|
assert ep.resolve_llamacpp_endpoint(wait_for_boot_s=8.0) is None
|
|
assert _time.monotonic() - t0 < 3.0
|
|
|
|
|
|
def test_switch_model_explicit_llamacpp_provider(tmp_path, monkeypatch, stub_server):
|
|
"""The desktop dropdown path: switch_model(explicit_provider='llamacpp')
|
|
must resolve the managed provider — not 'Unknown provider' (the
|
|
desktop-review symptom). E2E through the real pipeline against a stub server."""
|
|
port, handler = stub_server
|
|
handler.models = {"data": [{"id": "stub-model-a", "owned_by": "llamacpp"}]}
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime.supervisor import state_path
|
|
|
|
_write_current_process_state(
|
|
state_path(), base_url=f"http://127.0.0.1:{port}/v1", api_key="sk-managed")
|
|
|
|
from hermes_cli.model_switch import switch_model
|
|
|
|
result = switch_model(
|
|
"stub-model-a",
|
|
current_provider="nous",
|
|
current_model="Hermes-4.5",
|
|
current_base_url="",
|
|
explicit_provider="llamacpp",
|
|
)
|
|
assert result.success, result.error_message
|
|
assert f"127.0.0.1:{port}" in (result.base_url or "")
|
|
assert result.api_key == "sk-managed"
|
|
|
|
|
|
def test_runtime_provider_seam_llamacpp_alias(tmp_path, monkeypatch, stub_server):
|
|
"""End to end through the REAL resolver: provider='llamacpp' with no
|
|
base_url lands on the managed endpoint with source='local-runtime'."""
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
port, handler = stub_server
|
|
from hermes_cli.local_runtime.supervisor import state_path
|
|
|
|
_write_current_process_state(
|
|
state_path(), base_url=f"http://127.0.0.1:{port}/v1", api_key="sk-managed")
|
|
|
|
from hermes_cli.runtime_provider import _resolve_named_custom_runtime
|
|
|
|
runtime = _resolve_named_custom_runtime(requested_provider="llamacpp")
|
|
assert runtime is not None
|
|
assert runtime["source"] == "local-runtime"
|
|
assert runtime["base_url"] == f"http://127.0.0.1:{port}/v1"
|
|
assert runtime["api_key"] == "sk-managed"
|
|
assert runtime["provider"] == "custom"
|
|
|
|
|
|
def test_configured_llamacpp_provider_wins_over_managed_alias(tmp_path, monkeypatch):
|
|
"""A providers.llamacpp endpoint is explicit configuration, not a managed-runtime request (#116143)."""
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
monkeypatch.setattr(
|
|
"hermes_cli.config.load_config",
|
|
lambda: {
|
|
"providers": {
|
|
"llamacpp": {
|
|
"base_url": "http://127.0.0.1:8081/v1",
|
|
"default_model": "configured-model",
|
|
}
|
|
}
|
|
},
|
|
)
|
|
|
|
def _managed_alias_must_not_run(*args, **kwargs):
|
|
raise AssertionError("configured providers.llamacpp must resolve before managed detection")
|
|
|
|
monkeypatch.setattr(
|
|
"hermes_cli.local_runtime.endpoint.resolve_llamacpp_endpoint",
|
|
_managed_alias_must_not_run,
|
|
)
|
|
from hermes_cli.runtime_provider import _resolve_named_custom_runtime
|
|
|
|
runtime = _resolve_named_custom_runtime(requested_provider="llamacpp")
|
|
|
|
assert runtime is not None
|
|
assert runtime["base_url"] == "http://127.0.0.1:8081/v1"
|
|
assert runtime["model"] == "configured-model"
|
|
assert runtime["source"].startswith("custom_provider:")
|
|
|
|
|
|
def test_runtime_provider_seam_explicit_base_url_wins(tmp_path, monkeypatch):
|
|
"""A user-specified base_url must never be overridden by the managed
|
|
endpoint — pointing at a specific server means that server."""
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime.supervisor import state_path
|
|
|
|
state_path().parent.mkdir(parents=True, exist_ok=True)
|
|
state_path().write_text(json.dumps({
|
|
"base_url": "http://127.0.0.1:1/v1", "api_key": "sk-managed", "pid": 1,
|
|
}), encoding="utf-8")
|
|
|
|
from hermes_cli.runtime_provider import _resolve_named_custom_runtime
|
|
|
|
runtime = _resolve_named_custom_runtime(
|
|
requested_provider="llamacpp",
|
|
explicit_base_url="http://127.0.0.1:9999/v1")
|
|
assert runtime is not None
|
|
assert runtime["base_url"] == "http://127.0.0.1:9999/v1"
|
|
assert runtime["source"] != "local-runtime"
|
|
|
|
|
|
def _stage_local_model(home, model_id):
|
|
models = home / "models"
|
|
models.mkdir(parents=True, exist_ok=True)
|
|
(models / f"{model_id}.gguf").write_bytes(b"GGUF")
|
|
|
|
|
|
def test_staged_local_model_resolves_without_a_running_server(tmp_path, monkeypatch):
|
|
"""The picker's Local row exists from staged GGUFs alone (contract: selectable before the server
|
|
runs — selection starts it through the runtime seam). Selecting one must reach that seam instead
|
|
of dying at the provider gate with "Unknown provider 'llamacpp'": the row's id and the resolver's
|
|
are one definition, so an id the picker offers always resolves (#116249)."""
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
_stage_local_model(tmp_path / ".hermes", "Qwen3.8-27B-IQ3_S-mtp")
|
|
monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS", ())
|
|
|
|
from hermes_cli.providers import LLAMACPP_PROVIDER_ID, resolve_provider_full
|
|
|
|
pdef = resolve_provider_full(LLAMACPP_PROVIDER_ID, {}, [])
|
|
assert pdef is not None, "staged model, but the picker's own provider id does not resolve"
|
|
assert pdef.id == LLAMACPP_PROVIDER_ID
|
|
|
|
from hermes_cli.model_switch import switch_model
|
|
|
|
result = switch_model("Qwen3.8-27B-IQ3_S-mtp", current_provider="nous",
|
|
current_model="Hermes-4.5", current_base_url="",
|
|
explicit_provider=LLAMACPP_PROVIDER_ID)
|
|
assert result.success is False # no server anywhere; the seam reports it
|
|
error = result.error_message or ""
|
|
assert "Unknown provider" not in error
|
|
assert "local model server" in error.lower(), error
|
|
|
|
|
|
def test_external_server_on_a_configured_detect_port_is_used(tmp_path, monkeypatch, stub_server):
|
|
"""A llama-server the user runs on a fixed non-default port is what `provider: llamacpp` resolves
|
|
to when that port is declared in local_runtime.detect_ports — the knob is documented for exactly
|
|
this, and every provider path called the endpoint resolver without a config (#116143, #116249)."""
|
|
port, handler = stub_server
|
|
handler.props = {"build_info": "b10964-test", "model_path": "/models/ext-model.gguf",
|
|
"default_generation_settings": {"n_ctx": 4096}}
|
|
handler.models = {"data": [{"id": "ext-model", "owned_by": "llamacpp"}]}
|
|
home = tmp_path / ".hermes"
|
|
monkeypatch.setenv("HERMES_HOME", str(home))
|
|
_stage_local_model(home, "ext-model")
|
|
(home / "config.yaml").write_text(
|
|
f"local_runtime:\n enabled: false\n detect_ports: [{port}]\n", encoding="utf-8")
|
|
monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS", ())
|
|
|
|
from hermes_cli.model_switch import switch_model
|
|
|
|
result = switch_model("ext-model", current_provider="nous", current_model="Hermes-4.5",
|
|
current_base_url="", explicit_provider="llamacpp")
|
|
assert result.success, result.error_message
|
|
assert result.base_url == f"http://127.0.0.1:{port}/v1"
|
|
|
|
|
|
|
|
|
|
# ── bootstrap contracts ──────────────────────────────────────
|
|
|
|
|
|
def test_bootstrap_disabled_is_noop(tmp_path, monkeypatch):
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime import bootstrap
|
|
|
|
monkeypatch.setattr(bootstrap, "_SUPERVISOR", None)
|
|
assert bootstrap.ensure_local_runtime({"local_runtime": {"enabled": False}}) is None
|
|
assert bootstrap.ensure_local_runtime({}) is None
|
|
assert bootstrap.ensure_local_runtime(None) is None
|
|
|
|
|
|
def test_bootstrap_reuses_running_server(tmp_path, monkeypatch, stub_server):
|
|
"""A live state file (another process supervising) short-circuits the
|
|
install/spawn path entirely."""
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
port, handler = stub_server
|
|
from hermes_cli.local_runtime import bootstrap
|
|
from hermes_cli.local_runtime.supervisor import state_path
|
|
|
|
monkeypatch.setattr(bootstrap, "_SUPERVISOR", None)
|
|
state_path().parent.mkdir(parents=True, exist_ok=True)
|
|
state_path().write_text(json.dumps({
|
|
"base_url": f"http://127.0.0.1:{port}/v1", "api_key": "k", "pid": os.getpid(),
|
|
}), encoding="utf-8")
|
|
|
|
called = []
|
|
monkeypatch.setattr(
|
|
"hermes_cli.local_runtime.binaries.installed_engine",
|
|
lambda *a, **k: called.append(1))
|
|
assert bootstrap.ensure_local_runtime({"local_runtime": {"enabled": True}}) is None
|
|
assert called == []
|
|
|
|
|
|
def test_bootstrap_failure_never_raises(tmp_path, monkeypatch):
|
|
"""Session start must survive a broken runtime: failures log + return
|
|
None, chat falls back to configured providers."""
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime import bootstrap
|
|
|
|
monkeypatch.setattr(bootstrap, "_SUPERVISOR", None)
|
|
monkeypatch.setattr(bootstrap, "_detect_gpu_vendor", lambda: None)
|
|
models = bootstrap.models_dir()
|
|
models.mkdir(parents=True, exist_ok=True)
|
|
(models / "test.gguf").touch()
|
|
|
|
def boom(*a, **k):
|
|
raise RuntimeError("no network")
|
|
|
|
monkeypatch.setattr(
|
|
"hermes_cli.local_runtime.binaries.installed_engine", boom)
|
|
result = bootstrap.ensure_local_runtime({"local_runtime": {"enabled": True}})
|
|
assert result is None # no exception escaped
|
|
|
|
|
|
def test_ensure_local_runtime_serializes_racing_callers(tmp_path, monkeypatch):
|
|
"""Cross-process boot race (#116682): two backends starting in the same second must not
|
|
both spawn a router on the stable port. Neither caller here ever sees the other's
|
|
in-process ``_SUPERVISOR`` (each opens its own fd for the boot lock, exactly like two
|
|
separate OS processes would) — only the cross-process file lock can serialize them."""
|
|
import time as _time
|
|
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime import bootstrap
|
|
from hermes_cli.local_runtime import supervisor as sup_mod
|
|
|
|
monkeypatch.setattr(bootstrap, "_SUPERVISOR", None)
|
|
mdir = bootstrap.models_dir()
|
|
mdir.mkdir(parents=True, exist_ok=True)
|
|
(mdir / "stub-Q4_K_M.gguf").touch()
|
|
|
|
monkeypatch.setattr(bootstrap, "_generate_presets", lambda *a, **k: None)
|
|
monkeypatch.setattr(bootstrap, "_presets_stale", lambda: False)
|
|
monkeypatch.setattr(bootstrap, "_detect_gpu_vendor", lambda: None)
|
|
from hermes_cli.local_runtime.binaries import Engine
|
|
monkeypatch.setattr("hermes_cli.local_runtime.binaries.installed_engine",
|
|
lambda backend: Engine("cpu", "b1", tmp_path / "llama-server"))
|
|
|
|
spawns = []
|
|
|
|
class _FakeSupervisor:
|
|
def __init__(self, *a, **k):
|
|
self.port = 18434
|
|
self.api_key = "k"
|
|
self.proc = None
|
|
|
|
@property
|
|
def base_url(self):
|
|
return f"http://127.0.0.1:{self.port}/v1"
|
|
|
|
def start(self, timeout_s=120):
|
|
spawns.append(1)
|
|
_time.sleep(0.3) # widen the window the other caller races into
|
|
# A bare legacy ``{pid}`` record is no longer adoptable (a live PID is not evidence);
|
|
# publish what a real router publishes so the second caller can adopt it.
|
|
_write_current_process_state(
|
|
sup_mod.state_path(), base_url=self.base_url, api_key=self.api_key)
|
|
|
|
monkeypatch.setattr(sup_mod, "LlamaServerSupervisor", _FakeSupervisor)
|
|
|
|
barrier = threading.Barrier(2)
|
|
results = []
|
|
|
|
def _boot():
|
|
barrier.wait()
|
|
results.append(bootstrap.ensure_local_runtime({"local_runtime": {"enabled": True}}))
|
|
|
|
threads = [threading.Thread(target=_boot) for _ in range(2)]
|
|
for t in threads:
|
|
t.start()
|
|
for t in threads:
|
|
t.join(timeout=10)
|
|
|
|
assert len(spawns) == 1, (
|
|
"both racing callers spawned a router instead of the second adopting the "
|
|
"first's published state (#116682)")
|
|
|
|
|
|
def test_ensure_local_runtime_proceeds_when_boot_lock_is_unwritable(tmp_path, monkeypatch, caplog):
|
|
"""The boot lock lives outside the body's ``try/except``: an unwritable runtimes dir must
|
|
degrade to a warning and an unlocked boot, never an OSError out of session start."""
|
|
monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
|
|
from hermes_cli.local_runtime import bootstrap
|
|
|
|
monkeypatch.setattr(bootstrap, "_SUPERVISOR", None)
|
|
mdir = bootstrap.models_dir()
|
|
mdir.mkdir(parents=True, exist_ok=True)
|
|
(mdir / "stub-Q4_K_M.gguf").touch()
|
|
blocker = tmp_path / "not-a-dir"
|
|
blocker.write_text("", encoding="utf-8")
|
|
monkeypatch.setattr(bootstrap, "runtimes_root", lambda: blocker / "runtimes") # mkdir -> OSError
|
|
monkeypatch.setattr("hermes_cli.local_runtime.endpoint._state_endpoint", lambda: None)
|
|
monkeypatch.setattr("hermes_cli.local_runtime.binaries.installed_engine", lambda backend: None)
|
|
|
|
with caplog.at_level(logging.WARNING, logger=bootstrap.logger.name):
|
|
result = bootstrap.ensure_local_runtime({"local_runtime": {"enabled": True}})
|
|
|
|
assert result is None # no exception escaped
|
|
assert any("boot lock unavailable" in rec.getMessage() for rec in caplog.records)
|