Run models locally as a first-class provider. The CLI grows a managed llama.cpp runtime (engine install, model download, server supervision); the desktop app grows the full setup and management story on top of it. GUI surfaces ship behind the desktop --local launch flag (hermes desktop --local, or the flag on the packaged app); backend routes and the CLI are always live. Runtime (hermes_cli/local_runtime/): - curated GGUF catalog with per-machine variant selection: hardware probe (VRAM/RAM/UMA), fit planning with spill accounting, quant choice by context window - derived recommendation: quality-ranked picks gated by a predicted decode-speed floor, bandwidth-aware on unified memory; the decision table is pinned as a test (pick AND reason per memory class), and the Recommended badge explains its pick in a tooltip fed by the resolver's actual branch - engine install + model download with resumable split parts, cumulative plan-level progress, and staged-model integrity (a split GGUF counts only when every part is present) - server supervision: spawn/adopt/stop, router mode with per-model load progress relayed over SSE, abandoned-request cleanup Desktop: - Settings -> Providers -> Local models: one-click quickstart (install engine, download the recommended model, boot) plus per-model download/ activate/eject, fit-ranked catalog with context pills - model pickers (composer dropdown + Cmd+K) show staged local models, in-flight downloads as live progress rows, and load-into-memory bars - local-setup campaign tip for eligible hardware; System resources statusbar widget (GPU/VRAM/RAM); in-chat load progress during sends - friendly dead-server errors, and failed agent builds retry on the next send instead of wedging the session Co-developed with NVIDIA field feedback on RTX 5090 and DGX Spark.
83 lines
3.1 KiB
Python
83 lines
3.1 KiB
Python
"""The managed runtime is machine-scoped, not profile-scoped.
|
|
|
|
Engine binaries, models, presets, and server state are machine assets: a
|
|
second profile must reuse them, never re-download 20 GB of GGUFs or fight
|
|
the running server for its port. Profile-scoped decisions (default model,
|
|
enabled flag) stay in each profile's config.yaml."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import importlib
|
|
|
|
import pytest
|
|
|
|
|
|
@pytest.fixture
|
|
def profile_home(tmp_path, monkeypatch):
|
|
"""A NAMED-profile HERMES_HOME under <root>/profiles/<name>."""
|
|
root = tmp_path / ".hermes"
|
|
profile = root / "profiles" / "coder"
|
|
profile.mkdir(parents=True)
|
|
monkeypatch.setenv("HERMES_HOME", str(profile))
|
|
# hermes_constants memoizes root resolution per (native, env) pair;
|
|
# reload to make the new env authoritative for this test.
|
|
import hermes_constants
|
|
|
|
importlib.reload(hermes_constants)
|
|
yield root, profile
|
|
importlib.reload(hermes_constants)
|
|
|
|
|
|
def test_models_and_runtimes_resolve_to_the_shared_root(profile_home):
|
|
root, profile = profile_home
|
|
import hermes_cli.local_runtime.binaries as binaries
|
|
import hermes_cli.local_runtime.bootstrap as bootstrap
|
|
|
|
models = bootstrap.models_dir()
|
|
runtimes = binaries.runtimes_root()
|
|
|
|
assert models == root / "models", (
|
|
f"models dir leaked into the profile: {models}")
|
|
assert runtimes == root / "runtimes" / "llamacpp", (
|
|
f"runtimes dir leaked into the profile: {runtimes}")
|
|
assert "profiles" not in models.parts
|
|
assert "profiles" not in runtimes.parts
|
|
|
|
|
|
def test_all_runtime_state_follows_runtimes_root(profile_home):
|
|
"""Presets, window overrides, server state, and the api key all live
|
|
under runtimes_root() — one resolver, so profile-scoping bugs cannot
|
|
come back one file at a time."""
|
|
root, profile = profile_home
|
|
from hermes_cli.local_runtime.growth import window_overrides_path
|
|
from hermes_cli.local_runtime.presets import read_preset_decisions
|
|
import hermes_cli.local_runtime.binaries as binaries
|
|
|
|
shared = root / "runtimes" / "llamacpp"
|
|
assert window_overrides_path() == shared / "window_overrides.json"
|
|
# read_preset_decisions' default path must be the shared INI: write a
|
|
# section there and read it back through the default-path branch.
|
|
shared.mkdir(parents=True, exist_ok=True)
|
|
(shared / "presets.ini").write_text("[m1]\nctx-size = 65536\n",
|
|
encoding="utf-8")
|
|
assert "m1" in read_preset_decisions()
|
|
|
|
|
|
def test_default_profile_paths_unchanged(tmp_path, monkeypatch):
|
|
"""HERMES_HOME at the root itself (default profile) resolves exactly
|
|
as before the scoping change — no migration for existing installs."""
|
|
root = tmp_path / ".hermes"
|
|
root.mkdir()
|
|
monkeypatch.setenv("HERMES_HOME", str(root))
|
|
import hermes_constants
|
|
|
|
importlib.reload(hermes_constants)
|
|
try:
|
|
import hermes_cli.local_runtime.binaries as binaries
|
|
import hermes_cli.local_runtime.bootstrap as bootstrap
|
|
|
|
assert bootstrap.models_dir() == root / "models"
|
|
assert binaries.runtimes_root() == root / "runtimes" / "llamacpp"
|
|
finally:
|
|
importlib.reload(hermes_constants)
|