* refactor(plugins): remove the Sep 2026 decomposition compat layer on schedule The PLUGIN-COMPAT layer (2776813df3+d63e380324+0a5164cebe) kept pre-#102117 import paths alive for external plugins until 2026-09-14. That window closed two weeks ago; since then the loader has already been skipping plugins that use the old paths. This removes the layer itself: - 328 appended `PLUGIN-COMPAT` blocks (lazy `__getattr__` pointer tables, re-exported third-party names, restored dead definitions) and the three re-export stub modules (gateway/startup_watchdog, hermes_cli/observability/relay_runtime, tools/environments/modal_utils) - COMPAT_MANIFEST.md, compat_manifest.json, scripts/check_compat_pointers.py and its lint step - the reporting surfaces: CLI banner notice, `hermes plugins compat`, the `hermes doctor` section, the post-update notice, the Desktop one-time dialog, the loader's pre-import skip and the `plugins.allow_deprecated_imports` escape hatch An external plugin that still imports an old path now fails to load with its ImportError as the reason in `hermes plugins list`, the same path as any broken plugin. hermes_cli/plugin_compat.py stays as three inert stubs (compat_report, removal_in_effect, summary_lines): an already-running pre-removal `hermes update` lazy-imports them after the checkout swap (tests/compat/old_updater_surface.json). In-tree fallout, both already dead: hermes_cli/setup.py::_check_espeak_ng (no callers; its `shutil` came from a compat block) and gateway/config.py::SessionResetPolicy ("retained solely for the scheduled plugin-compat window"). Two test_run_agent patches targeted the removed `run_agent.handle_function_call` pointer; they now patch `model_tools.handle_function_call`, the seam production reads, like every sibling test in that file. * chore: retrigger CI (zero-job startup_failure phantom) * test: drop resolution allowlist rows for the two deleted which() sites hermes_cli/setup.py::_check_espeak_ng (dead) and tools/skillevaluator_scan.py::scanner_available (a restored definition inside a PLUGIN-COMPAT block) no longer exist; the stale-row gate requires their allowlist entries go with them.
495 lines
21 KiB
Python
495 lines
21 KiB
Python
"""Bootstrap for the managed runtime: config -> installed binaries -> running supervised server.
|
|
|
|
One public call, ``ensure_local_runtime(config)``, safe at any session start: disabled or
|
|
already-running -> no-op; enabled -> serve the installed build under a supervisor. Kept
|
|
import-light: callers gate on config before importing so disabled sessions never pay the import.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from contextlib import contextmanager, suppress
|
|
import logging
|
|
import os
|
|
import signal
|
|
import subprocess
|
|
import time
|
|
from pathlib import Path
|
|
|
|
from hermes_cli.local_runtime.binaries import runtimes_root
|
|
from hermes_cli.local_runtime.gguf import SPLIT_PART_RE, model_id_from_stem
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
_SUPERVISOR = None # process-wide singleton; one router per Hermes process
|
|
|
|
|
|
def _detect_gpu_vendor() -> str | None:
|
|
"""Best-effort GPU vendor for backend selection. NVIDIA via nvidia-smi resolved by the hardware
|
|
probe's PATH-independent ladder (a stripped service PATH must not demote an NVIDIA box to
|
|
vulkan/cpu); anything else defers to select_backend's fallback ladder."""
|
|
from hermes_cli.local_runtime.hardware import _nvidia_smi_path
|
|
|
|
smi = _nvidia_smi_path()
|
|
if smi is None:
|
|
return None
|
|
with suppress(OSError, subprocess.TimeoutExpired):
|
|
out = subprocess.run(
|
|
[smi, "--query-gpu=name", "--format=csv,noheader"],
|
|
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=10)
|
|
if out.returncode == 0 and out.stdout.strip():
|
|
return "nvidia " + out.stdout.strip().splitlines()[0]
|
|
return None
|
|
|
|
|
|
def models_dir() -> Path:
|
|
"""Machine-scoped, deliberately NOT profile-scoped: a 20 GB GGUF is a machine asset, and every
|
|
profile shares the one managed server that serves it (same rule as runtimes_root())."""
|
|
from hermes_constants import get_default_hermes_root
|
|
|
|
return get_default_hermes_root() / "models"
|
|
|
|
|
|
def assets_dir() -> Path:
|
|
"""Non-model companion files (mmproj projectors, spec-decode drafts). A subdirectory so the
|
|
router's model listing — and staged_models() — never mistakes an asset for a servable model."""
|
|
return models_dir() / "assets"
|
|
|
|
|
|
def staged_in(models_dir: Path, *, require_complete: bool = True) -> "list[Path]":
|
|
"""Servable GGUFs in a directory: single files, plus split GGUFs once by their first part.
|
|
With ``require_complete`` a split counts only when EVERY part is on disk — a mid-download split
|
|
is not servable and must not surface anywhere as a model."""
|
|
files = sorted(models_dir.glob("*.gguf"))
|
|
names = {p.name for p in files}
|
|
out = []
|
|
for p in files:
|
|
m = SPLIT_PART_RE.search(p.name)
|
|
if m is None:
|
|
out.append(p)
|
|
continue
|
|
if m.group(1) != "00001":
|
|
continue
|
|
stem, total = p.name[: m.start()], int(m.group(2))
|
|
if not require_complete or all(f"{stem}-{i:05d}-of-{m.group(2)}.gguf" in names
|
|
for i in range(2, total + 1)):
|
|
out.append(p)
|
|
return out
|
|
|
|
|
|
def adopt_legacy_models() -> "list[Path]":
|
|
"""Move GGUFs left in the old per-profile ``<profile home>/models`` layout (and its assets/)
|
|
into the machine-scoped dirs, so everything downstream keeps reading one directory.
|
|
|
|
``os.rename`` only: within one filesystem it is instant even for a 20 GB model, while
|
|
``shutil.move`` silently degrades to a copy across devices. A cross-device profile dir is left
|
|
in place with a warning rather than copying tens of GB at session start. A name that already
|
|
exists in the destination is left alone (check-then-rename: the only window is two processes
|
|
adopting two profiles' same-named file at once, and a same name is the same catalog variant).
|
|
Two processes racing on one file are harmless: the loser's rename finds the source gone and
|
|
skips it. Returns the new paths of the moved files."""
|
|
from hermes_constants import get_default_hermes_root, named_profile_has_identity
|
|
|
|
profiles_root = get_default_hermes_root() / "profiles"
|
|
if not profiles_root.is_dir():
|
|
return []
|
|
moved: list[Path] = []
|
|
for home in sorted(profiles_root.iterdir()):
|
|
old = home / "models"
|
|
if home.name.startswith(".") or not old.is_dir() or not named_profile_has_identity(home):
|
|
continue
|
|
for src_dir, dest_dir in ((old, models_dir()), (old / "assets", assets_dir())):
|
|
for src in sorted(src_dir.glob("*.gguf")):
|
|
dest = dest_dir / src.name
|
|
if dest.exists():
|
|
logger.warning("legacy model %s not moved: %s already exists", src, dest)
|
|
continue
|
|
try:
|
|
dest_dir.mkdir(parents=True, exist_ok=True)
|
|
os.rename(src, dest)
|
|
except FileNotFoundError:
|
|
continue
|
|
except OSError as exc:
|
|
logger.warning("legacy model %s not moved to %s: %s", src, dest_dir, exc)
|
|
continue
|
|
moved.append(dest)
|
|
for emptied in (old / "assets", old):
|
|
with suppress(OSError):
|
|
emptied.rmdir()
|
|
if moved:
|
|
logger.info("moved %d legacy model file(s) into %s", len(moved), models_dir())
|
|
return moved
|
|
|
|
|
|
def staged_models() -> "list[Path]":
|
|
"""Servable staged models (continuation parts, incomplete splits and assets/ never count)."""
|
|
return staged_in(models_dir())
|
|
|
|
|
|
def staged_model_ids() -> "list[str]":
|
|
return [model_id_from_stem(p.stem) for p in staged_models()]
|
|
|
|
|
|
def _presets_stale() -> bool:
|
|
"""True when a staged model has no section in the preset INI — it would autoload with stock
|
|
fit instead of a policy decision."""
|
|
with suppress(Exception):
|
|
from hermes_cli.local_runtime.presets import read_preset_decisions
|
|
|
|
known = read_preset_decisions()
|
|
return any(mid not in known or (not known[mid].refusal and not (known[mid].keys or {}).get("model"))
|
|
for mid in staged_model_ids())
|
|
return False
|
|
|
|
|
|
def _stop_state_server(state: dict) -> None:
|
|
"""Best-effort stop of the server the state file points at (an incumbent this process doesn't
|
|
supervise). The state pid is ours by contract — the file only ever describes the managed
|
|
server."""
|
|
from hermes_cli.local_runtime.endpoint import _pid_alive
|
|
|
|
try:
|
|
pid = int(state.get("pid"))
|
|
if pid <= 0:
|
|
return
|
|
os.kill(pid, signal.SIGTERM)
|
|
except (TypeError, ValueError, OSError):
|
|
return
|
|
# Give it a moment to release the port and the GPU. Liveness via psutil — on Windows
|
|
# os.kill(pid, 0) TERMINATES the process, it is not a probe.
|
|
for _ in range(50):
|
|
if not _pid_alive(pid):
|
|
return
|
|
time.sleep(0.1)
|
|
|
|
|
|
def refresh_local_runtime() -> bool:
|
|
"""Restart the managed server so it rescans the models directory. The router's model list is
|
|
SPAWN-ONLY: a GGUF added after start is invisible to GET /models and 400s on completion, so
|
|
anything that changes the staged set while the server runs must bounce it."""
|
|
global _SUPERVISOR
|
|
try:
|
|
from hermes_cli.config import load_config
|
|
|
|
if _SUPERVISOR is None:
|
|
from hermes_cli.local_runtime.endpoint import _state_endpoint
|
|
|
|
state = _state_endpoint()
|
|
if state is None:
|
|
return False
|
|
logger.info("bouncing adopted llama-server (pid=%s) to rescan models", state.get("pid"))
|
|
_stop_state_server(state)
|
|
else:
|
|
shutdown_local_runtime()
|
|
return ensure_local_runtime(load_config(), force=True) is not None
|
|
except Exception as exc: # noqa: BLE001
|
|
logger.warning("local runtime refresh failed: %s", exc)
|
|
return False
|
|
|
|
|
|
def _admitted_models_max(mdir: Path, configured: int) -> int:
|
|
"""Residency cap to hand the router: derived from the hardware budget, ``models_max`` as a ceiling.
|
|
|
|
A cap of "four" on a card that holds one model is how a second child ends up paged (WDDM) and
|
|
silently slow — llama.cpp evicts its LRU before an incoming load only when the cap says the
|
|
card is full. A probe miss or an unpriceable model keeps the configured count: this must never
|
|
block a boot.
|
|
"""
|
|
try:
|
|
from hermes_cli.local_runtime.hardware import probe_budget
|
|
from hermes_cli.local_runtime.presets import admitted_residency_count
|
|
|
|
cap = admitted_residency_count(mdir, probe_budget(planning=True), configured)
|
|
except Exception as exc: # noqa: BLE001
|
|
logger.warning("residency cap probe failed (%s); using models_max=%s", exc, configured)
|
|
return configured
|
|
if cap != configured:
|
|
logger.info("residency cap: %s resident model(s) on this card (models_max=%s)",
|
|
cap, configured)
|
|
return cap
|
|
|
|
|
|
def _generate_presets(mdir: Path, preset_path: Path) -> Path | None:
|
|
"""Write the launch-policy INI for every staged model; returns the path to hand the router.
|
|
|
|
Priced against CAPACITY, not live free VRAM: this runs while the outgoing server instance may
|
|
still hold the card (restart, refresh after a download), and its memory is freed before the new
|
|
instance loads anything. Pricing against live-free once pinned a fitting model's weights to CPU.
|
|
|
|
The window is then narrowed to what fits beside other programs' GPU memory
|
|
(``hardware.launch_budget``), which never moves weights to the CPU.
|
|
|
|
Degradation ladder on failure: a STALE policy still beats no policy — stock fit (f16 KV at max
|
|
context, no placement) is the silent-busy-wait failure on Windows. Keep serving with the
|
|
previous INI when one exists; only a first boot with no INI at all falls to stock fit.
|
|
"""
|
|
from hermes_cli.local_runtime.hardware import probe_budget
|
|
from hermes_cli.local_runtime.presets import generate_presets
|
|
|
|
try:
|
|
capacity = probe_budget(planning=True)
|
|
for entry in generate_presets(mdir, capacity, preset_path, live=_launch_budget(capacity)):
|
|
if entry.refusal:
|
|
logger.warning("model refused by physics check: %s", entry.refusal)
|
|
return preset_path
|
|
except Exception as exc: # noqa: BLE001 — policy failure must not block serving
|
|
if preset_path.exists():
|
|
logger.error("preset generation failed (%s); serving with the "
|
|
"PREVIOUS launch policies — models staged since "
|
|
"the last successful generation run unpoliced "
|
|
"until this is fixed", exc)
|
|
return preset_path
|
|
logger.error("preset generation failed (%s) and no previous "
|
|
"policy file exists; router runs stock fit", exc)
|
|
return None
|
|
|
|
|
|
def _launch_budget(capacity, own_bytes: int = 0):
|
|
"""``hardware.launch_budget``, or None when the probe fails: a boot never waits on it."""
|
|
from hermes_cli.local_runtime.hardware import launch_budget
|
|
|
|
try:
|
|
return launch_budget(capacity, own_bytes=own_bytes)
|
|
except Exception as exc: # noqa: BLE001
|
|
logger.warning("free GPU memory probe failed (%s); launch windows use capacity", exc)
|
|
return None
|
|
|
|
|
|
# Router statuses that hold GPU memory. The preset file changes only while every model is out of
|
|
# them: the router unloads a loaded model whose launch flags change on reload.
|
|
_HOLDS_MEMORY = frozenset({"loaded", "loading", "sleeping", "ready"})
|
|
|
|
|
|
def refit_idle_presets(sup) -> bool:
|
|
"""Re-plan launch windows against free GPU memory while no model is loaded.
|
|
|
|
Boot plans every window once, but models load later (on demand, after an idle unload), and by
|
|
then other programs may hold more or less of the card. Returns True when the router was given
|
|
new presets. A request that starts a load between the last status check and the reload is
|
|
unloaded by the router and fails once; that gap is a few milliseconds, and only when a window
|
|
changes.
|
|
"""
|
|
from hermes_cli.local_runtime.hardware import probe_budget
|
|
from hermes_cli.local_runtime.presets import plan_presets, render_presets
|
|
|
|
path = sup.preset_path
|
|
if path is None or not path.exists() or _any_holds_memory(sup):
|
|
return False
|
|
capacity = probe_budget(planning=True)
|
|
live = _launch_budget(capacity)
|
|
if live is None:
|
|
return False
|
|
last = sup._refit_usable
|
|
if last is not None and abs(live.usable_vram_bytes - last) < _REFIT_STEP_BYTES:
|
|
return False
|
|
text = render_presets(plan_presets(models_dir(), capacity, live=live))
|
|
# Held across the write and the reload so a restart can't spawn between them; everything
|
|
# slow ran above.
|
|
with sup._lifecycle_lock:
|
|
if sup.proc is None or sup.proc.poll() is not None or _any_holds_memory(sup):
|
|
return False
|
|
sup._refit_usable = live.usable_vram_bytes
|
|
if text == path.read_text(encoding="utf-8-sig"):
|
|
return False
|
|
from utils import atomic_write_text
|
|
|
|
atomic_write_text(path, text, tmp_prefix=f".{path.name}_", mode=0o600)
|
|
sup.reload_presets()
|
|
logger.info("launch windows re-planned for %.1f GiB of free GPU memory",
|
|
live.usable_vram_bytes / (1 << 30))
|
|
return True
|
|
|
|
|
|
# Smaller changes in free memory don't move a window by a ladder rung; skipping them keeps the
|
|
# idle loop from re-reading every model header each pass.
|
|
_REFIT_STEP_BYTES = 256 << 20
|
|
|
|
|
|
def _any_holds_memory(sup) -> bool:
|
|
return any(status in _HOLDS_MEMORY for status in sup.models(timeout_s=5).values())
|
|
|
|
|
|
def _try_lock_boot_fd(fd: int) -> bool:
|
|
"""Non-blocking exclusive attempt; portable across fcntl/msvcrt."""
|
|
if os.name == "nt":
|
|
import msvcrt
|
|
|
|
if os.fstat(fd).st_size == 0:
|
|
os.write(fd, b"\0")
|
|
os.lseek(fd, 0, os.SEEK_SET)
|
|
try:
|
|
msvcrt.locking(fd, msvcrt.LK_NBLCK, 1)
|
|
return True
|
|
except OSError:
|
|
return False
|
|
else:
|
|
import fcntl
|
|
|
|
try:
|
|
fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
return True
|
|
except OSError:
|
|
return False
|
|
|
|
|
|
def _unlock_boot_fd(fd: int) -> None:
|
|
if os.name == "nt":
|
|
import msvcrt
|
|
|
|
os.lseek(fd, 0, os.SEEK_SET)
|
|
msvcrt.locking(fd, msvcrt.LK_UNLCK, 1)
|
|
else:
|
|
import fcntl
|
|
|
|
fcntl.flock(fd, fcntl.LOCK_UN)
|
|
|
|
|
|
@contextmanager
|
|
def _cross_process_boot_lock(timeout_s: float = 130.0):
|
|
"""Serialize the state-check-then-spawn sequence across every Hermes process on this
|
|
machine — the ``_SUPERVISOR`` singleton above only rules out a race within ONE process.
|
|
Two profiles booting in the same second each see no ``server.json`` yet and each spawn a
|
|
router on the stable port (#116682); an OS-held lock makes the second caller wait for the
|
|
first to publish its state file, so it re-checks and adopts instead of spawning a duplicate.
|
|
Bounded, not indefinite: never hang session start dead if the lock is somehow stuck, and
|
|
never raise into session start: an unwritable runtimes dir (or a foreign-owned lock file)
|
|
proceeds unlocked with a warning, like the contention timeout."""
|
|
path = runtimes_root() / "boot.lock"
|
|
try:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
fd = os.open(path, os.O_CREAT | os.O_RDWR, 0o600)
|
|
except OSError as exc:
|
|
logger.warning("boot lock unavailable (%s); proceeding without it", exc)
|
|
yield
|
|
return
|
|
try:
|
|
deadline = time.monotonic() + timeout_s
|
|
while not _try_lock_boot_fd(fd):
|
|
if time.monotonic() >= deadline:
|
|
logger.warning("boot lock contended past %.0fs; proceeding without it", timeout_s)
|
|
break
|
|
time.sleep(0.2)
|
|
try:
|
|
yield
|
|
finally:
|
|
with suppress(OSError):
|
|
_unlock_boot_fd(fd)
|
|
finally:
|
|
os.close(fd)
|
|
|
|
|
|
def ensure_local_runtime(config: dict, force: bool = False) -> "object | None":
|
|
"""Idempotent boot of the managed runtime. Returns the supervisor (or None when
|
|
disabled/unavailable). Never raises into a session start — failures log and return None; chat
|
|
falls back to configured providers."""
|
|
global _SUPERVISOR
|
|
section = (config or {}).get("local_runtime") or {}
|
|
if not force and not section.get("enabled"):
|
|
return None
|
|
if _SUPERVISOR is not None:
|
|
return _SUPERVISOR
|
|
|
|
try:
|
|
adopt_legacy_models()
|
|
except OSError as exc: # an unreadable profiles dir must not block serving what's staged
|
|
logger.warning("legacy model adoption failed: %s", exc)
|
|
# Residency: no staged models means nothing to serve — don't boot an empty server (delete
|
|
# your last model and boots stop). force boots as ever.
|
|
if not force and not staged_models():
|
|
logger.info("local runtime enabled but no models staged; not booting")
|
|
return None
|
|
|
|
# Another Hermes process may already be supervising — reuse via state, but ONLY while its
|
|
# launch policy still covers every staged model. A server whose preset file predates a
|
|
# download serves the new model with no policy at all (--models-autoload + stock fit). A stale
|
|
# incumbent gets stopped and replaced by a fresh boot with regenerated presets; sessions ride
|
|
# through like any other supervised restart (stable port + persisted key).
|
|
#
|
|
# The state check and the spawn below run under a cross-process lock: two backends racing
|
|
# to boot (#116682) must not both find no state file and both spawn a router on the stable
|
|
# port — the loser waits here, then re-checks state and adopts the winner's server instead.
|
|
from hermes_cli.local_runtime.endpoint import _state_endpoint
|
|
|
|
with _cross_process_boot_lock():
|
|
state = _state_endpoint()
|
|
if state is not None:
|
|
if not _presets_stale():
|
|
logger.info("managed llama-server already running (another process)")
|
|
return None
|
|
logger.info("running server's presets predate the staged models; "
|
|
"replacing it so every model launches with a policy")
|
|
_stop_state_server(state)
|
|
|
|
try:
|
|
from hermes_cli.local_runtime.binaries import installed_engine
|
|
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
|
|
|
|
engine = installed_engine(section.get("backend", "auto"))
|
|
if engine is None:
|
|
logger.info("local runtime enabled but no PM engine installed; use the Local Models pane")
|
|
return None
|
|
|
|
mdir = models_dir()
|
|
mdir.mkdir(parents=True, exist_ok=True)
|
|
preset_path = _generate_presets(mdir, runtimes_root() / "presets.ini")
|
|
|
|
sup = LlamaServerSupervisor(engine.binary, mdir, preset_path=preset_path,
|
|
models_max=_admitted_models_max(
|
|
mdir, int(section.get("models_max", 4))),
|
|
port=int(section.get("port", 0)) or None)
|
|
try:
|
|
sup.start()
|
|
except Exception:
|
|
# start() can fail after the router process exists (health timeout): leaving it
|
|
# running unsupervised strands its VRAM behind a port nothing will clean up.
|
|
with suppress(Exception):
|
|
sup.stop()
|
|
raise
|
|
_SUPERVISOR = sup
|
|
logger.info("managed llama-server up at %s (backend=%s tag=%s)", sup.base_url, engine.backend, engine.tag)
|
|
_start_idle_sweeper(sup)
|
|
return sup
|
|
except Exception as exc: # noqa: BLE001 — never break session start
|
|
logger.warning("managed local runtime unavailable: %s", exc)
|
|
return None
|
|
|
|
|
|
def shutdown_local_runtime() -> None:
|
|
global _SUPERVISOR
|
|
if _SUPERVISOR is not None:
|
|
_SUPERVISOR.stop()
|
|
_SUPERVISOR = None
|
|
|
|
|
|
def get_supervisor():
|
|
"""The process-local supervisor, or None (a server may still run under another process —
|
|
check the state file)."""
|
|
return _SUPERVISOR
|
|
|
|
|
|
_SWEEP_EVERY_S = 120
|
|
_REFIT_EVERY_S = 30
|
|
|
|
|
|
def _start_idle_sweeper(sup) -> None:
|
|
"""Idle-residency loop: every couple of minutes, unload models idle past the supervisor's
|
|
threshold; every 30 s with nothing loaded, re-plan launch windows against free GPU memory.
|
|
Daemon thread tied to the supervisor's lifetime — exits when the server stops."""
|
|
import threading
|
|
|
|
def _loop():
|
|
last_sweep = time.monotonic()
|
|
while sup.proc is not None and sup.proc.poll() is None:
|
|
time.sleep(_REFIT_EVERY_S)
|
|
if time.monotonic() - last_sweep >= _SWEEP_EVERY_S:
|
|
last_sweep = time.monotonic()
|
|
try:
|
|
sup.sweep_idle()
|
|
except Exception as exc: # noqa: BLE001
|
|
logger.debug("idle sweep skipped: %s", exc)
|
|
try:
|
|
refit_idle_presets(sup)
|
|
except Exception as exc: # noqa: BLE001
|
|
logger.debug("launch window re-plan skipped: %s", exc)
|
|
|
|
threading.Thread(target=_loop, daemon=True, name="local-runtime-idle-sweep").start()
|