Keeps three invariants across #116156 (@fangliquanflq) and #116292 (@Finn763): a configured providers.llamacpp entry wins over managed-server detection on the runtime path; an external llama-server on a local_runtime.detect_ports port is what /model > Local (switch_model with the picker's provider id) resolves to; a staged GGUF alone keeps the picker's id resolvable so the runtime seam, not the provider gate, reports a missing server. Drops the endpoint.py config auto-load from #116292: both production callers of resolve_llamacpp_endpoint() now pass the loaded config (#116156), so loading it again inside the resolver was defense-in-depth. Drops the change-detector test on the forwarded config object and the redundant negative control. Docs: local-models page now shows detect_ports and the providers.llamacpp override for a llama-server the user runs on a fixed port.
150 lines
5.5 KiB
Python
150 lines
5.5 KiB
Python
"""Endpoint resolution for llamacpp-alias requests (provider integration).
|
|
|
|
``provider: llamacpp`` with no explicit base_url resolves, in order, to the managed server (state
|
|
file), a detected external llama-server, or — during a backend boot race — the managed server once
|
|
its state file appears.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from contextlib import suppress
|
|
import json
|
|
import logging
|
|
import threading
|
|
import time
|
|
import urllib.request
|
|
|
|
LLAMACPP_ALIASES = frozenset({"llamacpp", "llama.cpp", "llama-cpp"})
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
def _pid_alive(pid: int) -> bool:
|
|
"""Liveness for the state file's supervisor-child pid: psutil when available, else True
|
|
(optimistic). On Windows ``os.kill(pid, 0)`` TERMINATES the process — never use it as a probe."""
|
|
if not pid or pid < 0:
|
|
return False
|
|
with suppress(Exception):
|
|
import psutil # type: ignore
|
|
|
|
return psutil.pid_exists(pid)
|
|
return True
|
|
|
|
|
|
def _state_endpoint() -> dict | None:
|
|
from hermes_cli.local_runtime.recovery import is_modern, read_state, recorded_process
|
|
|
|
state = read_state()
|
|
base_url = state.get("base_url", "")
|
|
if not isinstance(base_url, str) or not base_url:
|
|
return None
|
|
if is_modern(state):
|
|
if recorded_process(state) is None:
|
|
return None
|
|
else:
|
|
# Preserve the legacy endpoint shape, with malformed PID values rejected.
|
|
try:
|
|
pid = state.get("pid")
|
|
if isinstance(pid, bool) or not _pid_alive(int(pid or 0)):
|
|
return None
|
|
except (TypeError, ValueError, OverflowError):
|
|
return None
|
|
return {"base_url": base_url, "api_key": state.get("api_key", "")}
|
|
|
|
|
|
def managed_root() -> "tuple[str, str] | None":
|
|
"""(base_root, api_key) of the managed router, or None. Resolved through the
|
|
ownership-guarded reader, not a raw state-file read: on the shared stable port a foreign
|
|
install's server answers /health for anyone, and a raw read would attach callers to someone
|
|
else's server."""
|
|
with suppress(Exception):
|
|
state = _state_endpoint()
|
|
if state is None:
|
|
return None
|
|
base = str(state.get("base_url", "")).rsplit("/v1", 1)[0]
|
|
return (base, str(state.get("api_key", ""))) if base else None
|
|
return None
|
|
|
|
|
|
def managed_get_json(base: str, api_key: str, route: str, timeout_s: float) -> object:
|
|
"""Authenticated GET against the managed router; raises on any transport/decode failure."""
|
|
req = urllib.request.Request(f"{base}{route}",
|
|
headers={"Authorization": f"Bearer {api_key}"})
|
|
with urllib.request.urlopen(req, timeout=timeout_s) as r:
|
|
return json.loads(r.read())
|
|
|
|
|
|
def resolve_llamacpp_endpoint(config: dict | None = None,
|
|
wait_for_boot_s: float = 8.0) -> dict | None:
|
|
"""Managed-first, detection-second endpoint for llamacpp aliases.
|
|
|
|
Boot-race rung: on a fresh backend start there is NO state file yet — the lifespan boot thread
|
|
is still spawning the server (≈1-3 s) while the desktop's readiness probe fires the moment the
|
|
WebSocket connects.
|
|
"""
|
|
managed = _state_endpoint()
|
|
if managed:
|
|
return managed
|
|
|
|
from hermes_cli.local_runtime.detect import detect_server
|
|
|
|
ports = ((config or {}).get("local_runtime") or {}).get("detect_ports") or []
|
|
hit = detect_server(extra_ports=tuple(int(p) for p in ports))
|
|
if hit and not hit.auth_required:
|
|
return {"base_url": hit.base_url, "api_key": ""}
|
|
|
|
if wait_for_boot_s > 0 and _boot_in_flight(config):
|
|
_kick_managed_boot(config)
|
|
deadline = time.monotonic() + wait_for_boot_s
|
|
while time.monotonic() < deadline:
|
|
time.sleep(0.25)
|
|
managed = _state_endpoint()
|
|
if managed:
|
|
return managed
|
|
return None
|
|
|
|
|
|
_KICK_LOCK = threading.Lock()
|
|
|
|
|
|
def _load_config_if_none(config: dict | None) -> dict | None:
|
|
if config is not None:
|
|
return config
|
|
from hermes_cli.config import load_config
|
|
|
|
return load_config()
|
|
|
|
|
|
def _kick_managed_boot(config: dict | None) -> None:
|
|
"""Actively start the managed server when resolution finds it missing — the wait loop assumes
|
|
some OTHER thread is bringing it up, which is true only at backend start."""
|
|
if not _KICK_LOCK.acquire(blocking=False):
|
|
return # a kick is already in flight
|
|
|
|
def _boot() -> None:
|
|
try:
|
|
from hermes_cli.local_runtime.bootstrap import ensure_local_runtime
|
|
|
|
ensure_local_runtime(_load_config_if_none(config))
|
|
except Exception: # noqa: BLE001 — best-effort; resolution falls back
|
|
logger.warning("on-demand managed-server boot failed", exc_info=True)
|
|
finally:
|
|
_KICK_LOCK.release()
|
|
|
|
threading.Thread(target=_boot, daemon=True,
|
|
name="lr-on-demand-boot").start()
|
|
|
|
|
|
def _boot_in_flight(config: dict | None) -> bool:
|
|
"""True when the managed runtime is enabled and installed (a verified-manifest scan under
|
|
runtimes_root(), NOT a bare ``server_binary()`` call — that needs an install_dir, and calling
|
|
it bare once made this gate throw-and-return False forever, disabling the boot wait)."""
|
|
with suppress(Exception):
|
|
config = _load_config_if_none(config)
|
|
if not ((config or {}).get("local_runtime") or {}).get("enabled"):
|
|
return False
|
|
from hermes_cli.local_runtime.binaries import manifest_verified, runtimes_root
|
|
|
|
return any(manifest_verified(m) for m in runtimes_root().glob("*/*/manifest.json"))
|
|
return False
|