Files
Teknium a5bd246865 Old pre-decomposition import paths are gone: plugin compat layer removed on schedule (#126164)
* refactor(plugins): remove the Sep 2026 decomposition compat layer on schedule

The PLUGIN-COMPAT layer (2776813df3 + d63e380324 + 0a5164cebe) kept pre-#102117 import paths
alive for external plugins until 2026-09-14. That window closed two weeks ago; since then the loader
has already been skipping plugins that use the old paths. This removes the layer itself:

- 328 appended `PLUGIN-COMPAT` blocks (lazy `__getattr__` pointer tables, re-exported third-party
  names, restored dead definitions) and the three re-export stub modules
  (gateway/startup_watchdog, hermes_cli/observability/relay_runtime, tools/environments/modal_utils)
- COMPAT_MANIFEST.md, compat_manifest.json, scripts/check_compat_pointers.py and its lint step
- the reporting surfaces: CLI banner notice, `hermes plugins compat`, the `hermes doctor` section,
  the post-update notice, the Desktop one-time dialog, the loader's pre-import skip and the
  `plugins.allow_deprecated_imports` escape hatch

An external plugin that still imports an old path now fails to load with its ImportError as the
reason in `hermes plugins list`, the same path as any broken plugin.

hermes_cli/plugin_compat.py stays as three inert stubs (compat_report, removal_in_effect,
summary_lines): an already-running pre-removal `hermes update` lazy-imports them after the checkout
swap (tests/compat/old_updater_surface.json).

In-tree fallout, both already dead: hermes_cli/setup.py::_check_espeak_ng (no callers; its
`shutil` came from a compat block) and gateway/config.py::SessionResetPolicy ("retained solely for
the scheduled plugin-compat window"). Two test_run_agent patches targeted the removed
`run_agent.handle_function_call` pointer; they now patch `model_tools.handle_function_call`, the
seam production reads, like every sibling test in that file.

* chore: retrigger CI (zero-job startup_failure phantom)

* test: drop resolution allowlist rows for the two deleted which() sites

hermes_cli/setup.py::_check_espeak_ng (dead) and tools/skillevaluator_scan.py::scanner_available
(a restored definition inside a PLUGIN-COMPAT block) no longer exist; the stale-row gate requires
their allowlist entries go with them.
2026-09-28 10:21:41 -07:00

495 lines
21 KiB
Python

"""Bootstrap for the managed runtime: config -> installed binaries -> running supervised server.
One public call, ``ensure_local_runtime(config)``, safe at any session start: disabled or
already-running -> no-op; enabled -> serve the installed build under a supervisor. Kept
import-light: callers gate on config before importing so disabled sessions never pay the import.
"""
from __future__ import annotations
from contextlib import contextmanager, suppress
import logging
import os
import signal
import subprocess
import time
from pathlib import Path
from hermes_cli.local_runtime.binaries import runtimes_root
from hermes_cli.local_runtime.gguf import SPLIT_PART_RE, model_id_from_stem
logger = logging.getLogger(__name__)
_SUPERVISOR = None # process-wide singleton; one router per Hermes process
def _detect_gpu_vendor() -> str | None:
"""Best-effort GPU vendor for backend selection. NVIDIA via nvidia-smi resolved by the hardware
probe's PATH-independent ladder (a stripped service PATH must not demote an NVIDIA box to
vulkan/cpu); anything else defers to select_backend's fallback ladder."""
from hermes_cli.local_runtime.hardware import _nvidia_smi_path
smi = _nvidia_smi_path()
if smi is None:
return None
with suppress(OSError, subprocess.TimeoutExpired):
out = subprocess.run(
[smi, "--query-gpu=name", "--format=csv,noheader"],
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=10)
if out.returncode == 0 and out.stdout.strip():
return "nvidia " + out.stdout.strip().splitlines()[0]
return None
def models_dir() -> Path:
"""Machine-scoped, deliberately NOT profile-scoped: a 20 GB GGUF is a machine asset, and every
profile shares the one managed server that serves it (same rule as runtimes_root())."""
from hermes_constants import get_default_hermes_root
return get_default_hermes_root() / "models"
def assets_dir() -> Path:
"""Non-model companion files (mmproj projectors, spec-decode drafts). A subdirectory so the
router's model listing — and staged_models() — never mistakes an asset for a servable model."""
return models_dir() / "assets"
def staged_in(models_dir: Path, *, require_complete: bool = True) -> "list[Path]":
"""Servable GGUFs in a directory: single files, plus split GGUFs once by their first part.
With ``require_complete`` a split counts only when EVERY part is on disk — a mid-download split
is not servable and must not surface anywhere as a model."""
files = sorted(models_dir.glob("*.gguf"))
names = {p.name for p in files}
out = []
for p in files:
m = SPLIT_PART_RE.search(p.name)
if m is None:
out.append(p)
continue
if m.group(1) != "00001":
continue
stem, total = p.name[: m.start()], int(m.group(2))
if not require_complete or all(f"{stem}-{i:05d}-of-{m.group(2)}.gguf" in names
for i in range(2, total + 1)):
out.append(p)
return out
def adopt_legacy_models() -> "list[Path]":
"""Move GGUFs left in the old per-profile ``<profile home>/models`` layout (and its assets/)
into the machine-scoped dirs, so everything downstream keeps reading one directory.
``os.rename`` only: within one filesystem it is instant even for a 20 GB model, while
``shutil.move`` silently degrades to a copy across devices. A cross-device profile dir is left
in place with a warning rather than copying tens of GB at session start. A name that already
exists in the destination is left alone (check-then-rename: the only window is two processes
adopting two profiles' same-named file at once, and a same name is the same catalog variant).
Two processes racing on one file are harmless: the loser's rename finds the source gone and
skips it. Returns the new paths of the moved files."""
from hermes_constants import get_default_hermes_root, named_profile_has_identity
profiles_root = get_default_hermes_root() / "profiles"
if not profiles_root.is_dir():
return []
moved: list[Path] = []
for home in sorted(profiles_root.iterdir()):
old = home / "models"
if home.name.startswith(".") or not old.is_dir() or not named_profile_has_identity(home):
continue
for src_dir, dest_dir in ((old, models_dir()), (old / "assets", assets_dir())):
for src in sorted(src_dir.glob("*.gguf")):
dest = dest_dir / src.name
if dest.exists():
logger.warning("legacy model %s not moved: %s already exists", src, dest)
continue
try:
dest_dir.mkdir(parents=True, exist_ok=True)
os.rename(src, dest)
except FileNotFoundError:
continue
except OSError as exc:
logger.warning("legacy model %s not moved to %s: %s", src, dest_dir, exc)
continue
moved.append(dest)
for emptied in (old / "assets", old):
with suppress(OSError):
emptied.rmdir()
if moved:
logger.info("moved %d legacy model file(s) into %s", len(moved), models_dir())
return moved
def staged_models() -> "list[Path]":
"""Servable staged models (continuation parts, incomplete splits and assets/ never count)."""
return staged_in(models_dir())
def staged_model_ids() -> "list[str]":
return [model_id_from_stem(p.stem) for p in staged_models()]
def _presets_stale() -> bool:
"""True when a staged model has no section in the preset INI — it would autoload with stock
fit instead of a policy decision."""
with suppress(Exception):
from hermes_cli.local_runtime.presets import read_preset_decisions
known = read_preset_decisions()
return any(mid not in known or (not known[mid].refusal and not (known[mid].keys or {}).get("model"))
for mid in staged_model_ids())
return False
def _stop_state_server(state: dict) -> None:
"""Best-effort stop of the server the state file points at (an incumbent this process doesn't
supervise). The state pid is ours by contract — the file only ever describes the managed
server."""
from hermes_cli.local_runtime.endpoint import _pid_alive
try:
pid = int(state.get("pid"))
if pid <= 0:
return
os.kill(pid, signal.SIGTERM)
except (TypeError, ValueError, OSError):
return
# Give it a moment to release the port and the GPU. Liveness via psutil — on Windows
# os.kill(pid, 0) TERMINATES the process, it is not a probe.
for _ in range(50):
if not _pid_alive(pid):
return
time.sleep(0.1)
def refresh_local_runtime() -> bool:
"""Restart the managed server so it rescans the models directory. The router's model list is
SPAWN-ONLY: a GGUF added after start is invisible to GET /models and 400s on completion, so
anything that changes the staged set while the server runs must bounce it."""
global _SUPERVISOR
try:
from hermes_cli.config import load_config
if _SUPERVISOR is None:
from hermes_cli.local_runtime.endpoint import _state_endpoint
state = _state_endpoint()
if state is None:
return False
logger.info("bouncing adopted llama-server (pid=%s) to rescan models", state.get("pid"))
_stop_state_server(state)
else:
shutdown_local_runtime()
return ensure_local_runtime(load_config(), force=True) is not None
except Exception as exc: # noqa: BLE001
logger.warning("local runtime refresh failed: %s", exc)
return False
def _admitted_models_max(mdir: Path, configured: int) -> int:
"""Residency cap to hand the router: derived from the hardware budget, ``models_max`` as a ceiling.
A cap of "four" on a card that holds one model is how a second child ends up paged (WDDM) and
silently slow — llama.cpp evicts its LRU before an incoming load only when the cap says the
card is full. A probe miss or an unpriceable model keeps the configured count: this must never
block a boot.
"""
try:
from hermes_cli.local_runtime.hardware import probe_budget
from hermes_cli.local_runtime.presets import admitted_residency_count
cap = admitted_residency_count(mdir, probe_budget(planning=True), configured)
except Exception as exc: # noqa: BLE001
logger.warning("residency cap probe failed (%s); using models_max=%s", exc, configured)
return configured
if cap != configured:
logger.info("residency cap: %s resident model(s) on this card (models_max=%s)",
cap, configured)
return cap
def _generate_presets(mdir: Path, preset_path: Path) -> Path | None:
"""Write the launch-policy INI for every staged model; returns the path to hand the router.
Priced against CAPACITY, not live free VRAM: this runs while the outgoing server instance may
still hold the card (restart, refresh after a download), and its memory is freed before the new
instance loads anything. Pricing against live-free once pinned a fitting model's weights to CPU.
The window is then narrowed to what fits beside other programs' GPU memory
(``hardware.launch_budget``), which never moves weights to the CPU.
Degradation ladder on failure: a STALE policy still beats no policy — stock fit (f16 KV at max
context, no placement) is the silent-busy-wait failure on Windows. Keep serving with the
previous INI when one exists; only a first boot with no INI at all falls to stock fit.
"""
from hermes_cli.local_runtime.hardware import probe_budget
from hermes_cli.local_runtime.presets import generate_presets
try:
capacity = probe_budget(planning=True)
for entry in generate_presets(mdir, capacity, preset_path, live=_launch_budget(capacity)):
if entry.refusal:
logger.warning("model refused by physics check: %s", entry.refusal)
return preset_path
except Exception as exc: # noqa: BLE001 — policy failure must not block serving
if preset_path.exists():
logger.error("preset generation failed (%s); serving with the "
"PREVIOUS launch policies — models staged since "
"the last successful generation run unpoliced "
"until this is fixed", exc)
return preset_path
logger.error("preset generation failed (%s) and no previous "
"policy file exists; router runs stock fit", exc)
return None
def _launch_budget(capacity, own_bytes: int = 0):
"""``hardware.launch_budget``, or None when the probe fails: a boot never waits on it."""
from hermes_cli.local_runtime.hardware import launch_budget
try:
return launch_budget(capacity, own_bytes=own_bytes)
except Exception as exc: # noqa: BLE001
logger.warning("free GPU memory probe failed (%s); launch windows use capacity", exc)
return None
# Router statuses that hold GPU memory. The preset file changes only while every model is out of
# them: the router unloads a loaded model whose launch flags change on reload.
_HOLDS_MEMORY = frozenset({"loaded", "loading", "sleeping", "ready"})
def refit_idle_presets(sup) -> bool:
"""Re-plan launch windows against free GPU memory while no model is loaded.
Boot plans every window once, but models load later (on demand, after an idle unload), and by
then other programs may hold more or less of the card. Returns True when the router was given
new presets. A request that starts a load between the last status check and the reload is
unloaded by the router and fails once; that gap is a few milliseconds, and only when a window
changes.
"""
from hermes_cli.local_runtime.hardware import probe_budget
from hermes_cli.local_runtime.presets import plan_presets, render_presets
path = sup.preset_path
if path is None or not path.exists() or _any_holds_memory(sup):
return False
capacity = probe_budget(planning=True)
live = _launch_budget(capacity)
if live is None:
return False
last = sup._refit_usable
if last is not None and abs(live.usable_vram_bytes - last) < _REFIT_STEP_BYTES:
return False
text = render_presets(plan_presets(models_dir(), capacity, live=live))
# Held across the write and the reload so a restart can't spawn between them; everything
# slow ran above.
with sup._lifecycle_lock:
if sup.proc is None or sup.proc.poll() is not None or _any_holds_memory(sup):
return False
sup._refit_usable = live.usable_vram_bytes
if text == path.read_text(encoding="utf-8-sig"):
return False
from utils import atomic_write_text
atomic_write_text(path, text, tmp_prefix=f".{path.name}_", mode=0o600)
sup.reload_presets()
logger.info("launch windows re-planned for %.1f GiB of free GPU memory",
live.usable_vram_bytes / (1 << 30))
return True
# Smaller changes in free memory don't move a window by a ladder rung; skipping them keeps the
# idle loop from re-reading every model header each pass.
_REFIT_STEP_BYTES = 256 << 20
def _any_holds_memory(sup) -> bool:
return any(status in _HOLDS_MEMORY for status in sup.models(timeout_s=5).values())
def _try_lock_boot_fd(fd: int) -> bool:
"""Non-blocking exclusive attempt; portable across fcntl/msvcrt."""
if os.name == "nt":
import msvcrt
if os.fstat(fd).st_size == 0:
os.write(fd, b"\0")
os.lseek(fd, 0, os.SEEK_SET)
try:
msvcrt.locking(fd, msvcrt.LK_NBLCK, 1)
return True
except OSError:
return False
else:
import fcntl
try:
fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
return True
except OSError:
return False
def _unlock_boot_fd(fd: int) -> None:
if os.name == "nt":
import msvcrt
os.lseek(fd, 0, os.SEEK_SET)
msvcrt.locking(fd, msvcrt.LK_UNLCK, 1)
else:
import fcntl
fcntl.flock(fd, fcntl.LOCK_UN)
@contextmanager
def _cross_process_boot_lock(timeout_s: float = 130.0):
"""Serialize the state-check-then-spawn sequence across every Hermes process on this
machine — the ``_SUPERVISOR`` singleton above only rules out a race within ONE process.
Two profiles booting in the same second each see no ``server.json`` yet and each spawn a
router on the stable port (#116682); an OS-held lock makes the second caller wait for the
first to publish its state file, so it re-checks and adopts instead of spawning a duplicate.
Bounded, not indefinite: never hang session start dead if the lock is somehow stuck, and
never raise into session start: an unwritable runtimes dir (or a foreign-owned lock file)
proceeds unlocked with a warning, like the contention timeout."""
path = runtimes_root() / "boot.lock"
try:
path.parent.mkdir(parents=True, exist_ok=True)
fd = os.open(path, os.O_CREAT | os.O_RDWR, 0o600)
except OSError as exc:
logger.warning("boot lock unavailable (%s); proceeding without it", exc)
yield
return
try:
deadline = time.monotonic() + timeout_s
while not _try_lock_boot_fd(fd):
if time.monotonic() >= deadline:
logger.warning("boot lock contended past %.0fs; proceeding without it", timeout_s)
break
time.sleep(0.2)
try:
yield
finally:
with suppress(OSError):
_unlock_boot_fd(fd)
finally:
os.close(fd)
def ensure_local_runtime(config: dict, force: bool = False) -> "object | None":
"""Idempotent boot of the managed runtime. Returns the supervisor (or None when
disabled/unavailable). Never raises into a session start — failures log and return None; chat
falls back to configured providers."""
global _SUPERVISOR
section = (config or {}).get("local_runtime") or {}
if not force and not section.get("enabled"):
return None
if _SUPERVISOR is not None:
return _SUPERVISOR
try:
adopt_legacy_models()
except OSError as exc: # an unreadable profiles dir must not block serving what's staged
logger.warning("legacy model adoption failed: %s", exc)
# Residency: no staged models means nothing to serve — don't boot an empty server (delete
# your last model and boots stop). force boots as ever.
if not force and not staged_models():
logger.info("local runtime enabled but no models staged; not booting")
return None
# Another Hermes process may already be supervising — reuse via state, but ONLY while its
# launch policy still covers every staged model. A server whose preset file predates a
# download serves the new model with no policy at all (--models-autoload + stock fit). A stale
# incumbent gets stopped and replaced by a fresh boot with regenerated presets; sessions ride
# through like any other supervised restart (stable port + persisted key).
#
# The state check and the spawn below run under a cross-process lock: two backends racing
# to boot (#116682) must not both find no state file and both spawn a router on the stable
# port — the loser waits here, then re-checks state and adopts the winner's server instead.
from hermes_cli.local_runtime.endpoint import _state_endpoint
with _cross_process_boot_lock():
state = _state_endpoint()
if state is not None:
if not _presets_stale():
logger.info("managed llama-server already running (another process)")
return None
logger.info("running server's presets predate the staged models; "
"replacing it so every model launches with a policy")
_stop_state_server(state)
try:
from hermes_cli.local_runtime.binaries import installed_engine
from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
engine = installed_engine(section.get("backend", "auto"))
if engine is None:
logger.info("local runtime enabled but no PM engine installed; use the Local Models pane")
return None
mdir = models_dir()
mdir.mkdir(parents=True, exist_ok=True)
preset_path = _generate_presets(mdir, runtimes_root() / "presets.ini")
sup = LlamaServerSupervisor(engine.binary, mdir, preset_path=preset_path,
models_max=_admitted_models_max(
mdir, int(section.get("models_max", 4))),
port=int(section.get("port", 0)) or None)
try:
sup.start()
except Exception:
# start() can fail after the router process exists (health timeout): leaving it
# running unsupervised strands its VRAM behind a port nothing will clean up.
with suppress(Exception):
sup.stop()
raise
_SUPERVISOR = sup
logger.info("managed llama-server up at %s (backend=%s tag=%s)", sup.base_url, engine.backend, engine.tag)
_start_idle_sweeper(sup)
return sup
except Exception as exc: # noqa: BLE001 — never break session start
logger.warning("managed local runtime unavailable: %s", exc)
return None
def shutdown_local_runtime() -> None:
global _SUPERVISOR
if _SUPERVISOR is not None:
_SUPERVISOR.stop()
_SUPERVISOR = None
def get_supervisor():
"""The process-local supervisor, or None (a server may still run under another process —
check the state file)."""
return _SUPERVISOR
_SWEEP_EVERY_S = 120
_REFIT_EVERY_S = 30
def _start_idle_sweeper(sup) -> None:
"""Idle-residency loop: every couple of minutes, unload models idle past the supervisor's
threshold; every 30 s with nothing loaded, re-plan launch windows against free GPU memory.
Daemon thread tied to the supervisor's lifetime — exits when the server stops."""
import threading
def _loop():
last_sweep = time.monotonic()
while sup.proc is not None and sup.proc.poll() is None:
time.sleep(_REFIT_EVERY_S)
if time.monotonic() - last_sweep >= _SWEEP_EVERY_S:
last_sweep = time.monotonic()
try:
sup.sweep_idle()
except Exception as exc: # noqa: BLE001
logger.debug("idle sweep skipped: %s", exc)
try:
refit_idle_presets(sup)
except Exception as exc: # noqa: BLE001
logger.debug("launch window re-plan skipped: %s", exc)
threading.Thread(target=_loop, daemon=True, name="local-runtime-idle-sweep").start()