Run models locally as a first-class provider. The CLI grows a managed llama.cpp runtime (engine install, model download, server supervision); the desktop app grows the full setup and management story on top of it. GUI surfaces ship behind the desktop --local launch flag (hermes desktop --local, or the flag on the packaged app); backend routes and the CLI are always live. Runtime (hermes_cli/local_runtime/): - curated GGUF catalog with per-machine variant selection: hardware probe (VRAM/RAM/UMA), fit planning with spill accounting, quant choice by context window - derived recommendation: quality-ranked picks gated by a predicted decode-speed floor, bandwidth-aware on unified memory; the decision table is pinned as a test (pick AND reason per memory class), and the Recommended badge explains its pick in a tooltip fed by the resolver's actual branch - engine install + model download with resumable split parts, cumulative plan-level progress, and staged-model integrity (a split GGUF counts only when every part is present) - server supervision: spawn/adopt/stop, router mode with per-model load progress relayed over SSE, abandoned-request cleanup Desktop: - Settings -> Providers -> Local models: one-click quickstart (install engine, download the recommended model, boot) plus per-model download/ activate/eject, fit-ranked catalog with context pills - model pickers (composer dropdown + Cmd+K) show staged local models, in-flight downloads as live progress rows, and load-into-memory bars - local-setup campaign tip for eligible hardware; System resources statusbar widget (GPU/VRAM/RAM); in-chat load progress during sends - friendly dead-server errors, and failed agent builds retry on the next send instead of wedging the session Co-developed with NVIDIA field feedback on RTX 5090 and DGX Spark.
119 lines
4.0 KiB
Python
119 lines
4.0 KiB
Python
"""The pulled catalog: packaged JSON is the offline truth, a GitHub fetch
|
|
swaps entries in memory only, and min_engine gates day-0 models.
|
|
|
|
Nothing here touches disk beyond the packaged file — the design constraint
|
|
is that a git checkout must never see a dirty tracked catalog.json."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import dataclasses
|
|
import io
|
|
import json
|
|
import urllib.request
|
|
|
|
import pytest
|
|
|
|
import hermes_cli.local_runtime.catalog as cat
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _reset_refresh_state(monkeypatch):
|
|
"""Each test starts outside the TTL window with the packaged catalog."""
|
|
monkeypatch.setattr(cat, "_last_refresh_attempt", 0.0)
|
|
packaged = cat._packaged_catalog()
|
|
monkeypatch.setattr(cat, "CATALOG", packaged)
|
|
yield
|
|
|
|
|
|
def _doc_from(entries):
|
|
"""A fetchable catalog document built by mutating the packaged JSON."""
|
|
from importlib.resources import files
|
|
|
|
doc = json.loads(files("hermes_cli.local_runtime")
|
|
.joinpath("catalog.json").read_text(encoding="utf-8"))
|
|
doc["models"] = entries(doc["models"])
|
|
return doc
|
|
|
|
|
|
def _fetch_returns(monkeypatch, doc):
|
|
body = json.dumps(doc).encode()
|
|
|
|
class R(io.BytesIO):
|
|
def __enter__(self):
|
|
return self
|
|
|
|
def __exit__(self, *a):
|
|
return False
|
|
|
|
monkeypatch.setattr(urllib.request, "urlopen",
|
|
lambda *a, **k: R(body))
|
|
|
|
|
|
def test_packaged_json_round_trips_the_catalog():
|
|
"""The packaged JSON must produce a complete, selection-ready catalog:
|
|
every entry carries estimator inputs and at least one variant, and the
|
|
known invariants (best-first ordering, Q4 floor) hold — the same
|
|
contract the literals obeyed."""
|
|
assert len(cat.CATALOG) >= 4
|
|
for e in cat.CATALOG:
|
|
assert e.variants and e.n_ctx_train > 0 and e.per_layer_f16 >= 0
|
|
sizes = [v.size_bytes for v in e.variants]
|
|
assert sizes == sorted(sizes, reverse=True), f"{e.id} not best-first"
|
|
|
|
|
|
def test_refresh_swaps_in_memory_only(monkeypatch, tmp_path):
|
|
"""A fetched catalog replaces CATALOG in memory; the packaged file on
|
|
disk is untouched (checkout stays clean)."""
|
|
from importlib.resources import files
|
|
|
|
packaged_path = files("hermes_cli.local_runtime").joinpath("catalog.json")
|
|
before = packaged_path.read_text(encoding="utf-8")
|
|
|
|
def add_day0(models):
|
|
day0 = dict(models[0])
|
|
day0.update(id="day0-model", display_name="Day 0",
|
|
description="new", min_engine="b99999")
|
|
return models + [day0]
|
|
|
|
_fetch_returns(monkeypatch, _doc_from(add_day0))
|
|
assert cat.refresh_catalog(force=True) is True
|
|
assert "day0-model" in {e.id for e in cat.CATALOG}
|
|
assert cat.catalog_by_id()["day0-model"].min_engine == "b99999"
|
|
assert packaged_path.read_text(encoding="utf-8") == before
|
|
|
|
|
|
def test_refresh_failure_keeps_current_catalog(monkeypatch):
|
|
def boom(*a, **k):
|
|
raise OSError("offline")
|
|
|
|
monkeypatch.setattr(urllib.request, "urlopen", boom)
|
|
ids_before = [e.id for e in cat.CATALOG]
|
|
assert cat.refresh_catalog(force=True) is False
|
|
assert [e.id for e in cat.CATALOG] == ids_before
|
|
|
|
|
|
def test_refresh_rejects_wrong_schema(monkeypatch):
|
|
doc = _doc_from(lambda m: m)
|
|
doc["schema_version"] = 2
|
|
_fetch_returns(monkeypatch, doc)
|
|
ids_before = [e.id for e in cat.CATALOG]
|
|
assert cat.refresh_catalog(force=True) is False
|
|
assert [e.id for e in cat.CATALOG] == ids_before
|
|
|
|
|
|
def test_loader_ignores_unknown_fields():
|
|
doc = _doc_from(lambda m: m)
|
|
doc["models"][0]["future_field"] = {"anything": True}
|
|
entries = cat._load_catalog(doc)
|
|
assert entries[0].id == doc["models"][0]["id"]
|
|
|
|
|
|
def test_min_engine_gate(monkeypatch):
|
|
from hermes_cli.web_routers.local_models import _engine_too_old
|
|
|
|
monkeypatch.setattr("hermes_cli.local_runtime.binaries.installed_tags",
|
|
lambda: ["b10362"])
|
|
assert _engine_too_old("") is False, "no requirement, no gate"
|
|
assert _engine_too_old("b10000") is False, "installed engine suffices"
|
|
assert _engine_too_old("b10363") is True, "newer requirement gates"
|