fix(local-runtime): unify memory accounting and effective-window MTP
Price weights, context, runtime, projector and batch overhead consistently across catalog admission, initial launch, growth and restored windows. Keep MTP and the larger window when lean batches avoid unnecessary spill. Admit optional external drafts only when their complete footprint fits. Use preset-only model discovery so refused files cannot autoload, and preserve refusal/spill decisions atomically for desktop status read-back. Add regression coverage for complete-footprint boundaries, MTP restarts, growth admission, draft budgets and placement status transitions. Builds on the overhead-accounting contribution in #102993 and the restored-window MTP contribution in #106897. Does not adopt the 40% host-RAM reserve or resolve the remaining requests in #102865/#106895. Co-authored-by: infinitycrew39 <infinitycrew39@gmail.com> Co-authored-by: KoNit-K <124019182+KoNit-K@users.noreply.github.com>
This commit is contained in:
@@ -91,8 +91,9 @@ def _presets_stale() -> bool:
|
||||
with suppress(Exception):
|
||||
from hermes_cli.local_runtime.presets import read_preset_decisions
|
||||
|
||||
known = set(read_preset_decisions())
|
||||
return any(mid not in known for mid in staged_model_ids())
|
||||
known = read_preset_decisions()
|
||||
return any(mid not in known or (not known[mid].refusal and not (known[mid].keys or {}).get("model"))
|
||||
for mid in staged_model_ids())
|
||||
return False
|
||||
|
||||
|
||||
|
||||
@@ -17,8 +17,8 @@ from dataclasses import dataclass, field
|
||||
from pathlib import PurePosixPath
|
||||
|
||||
from hermes_cli.local_runtime.context_policy import (
|
||||
FLOOR, RUNTIME_OVERHEAD_BYTES, TARGET_WINDOW, ub_logits_bytes)
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile, ctx_bytes
|
||||
FLOOR, RUNTIME_OVERHEAD_BYTES, TARGET_WINDOW, LaunchPlan, plan_launch)
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile, PhysicsRefusal
|
||||
from hermes_cli.local_runtime.gguf import model_id_from_stem
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -115,6 +115,12 @@ class CatalogEntry:
|
||||
n_ctx_train=self.n_ctx_train, layers=layers, swa_window=self.swa_window, moe=self.moe,
|
||||
n_vocab=self.n_vocab, kv_scale=1.2 if self.mtp else 1.0)
|
||||
|
||||
def launch_plan(self, variant: QuantVariant, budget: HardwareBudget) -> LaunchPlan:
|
||||
# Optional external drafts may use spare memory after download, never reduce this grant.
|
||||
return plan_launch(self.profile(variant), budget, mtp_capable=self.mtp,
|
||||
fixed_overhead=RUNTIME_OVERHEAD_BYTES
|
||||
+ (self.mmproj.size_bytes if self.mmproj else 0))
|
||||
|
||||
def download_files(self, variant: QuantVariant) -> tuple:
|
||||
"""Everything a download job fetches for this variant, in order."""
|
||||
extras = tuple(a for a in (self.mmproj, self.draft) if a is not None)
|
||||
@@ -141,22 +147,14 @@ def select_variant(entry: CatalogEntry, budget: HardwareBudget) -> VariantChoice
|
||||
"best-large-window": zero-spill at TARGET_WINDOW; "best-fits": zero-spill at the 64K floor;
|
||||
"smallest-fits-spilled": weights spill to host RAM, priced honestly; None: physics refuses.
|
||||
"""
|
||||
overhead = (RUNTIME_OVERHEAD_BYTES
|
||||
+ (entry.mmproj.size_bytes if entry.mmproj else 0)
|
||||
+ ub_logits_bytes(entry.n_vocab, mtp_capable=entry.mtp))
|
||||
native = entry.n_ctx_train or FLOOR
|
||||
variant = entry.variants[-1]
|
||||
profile = entry.profile(variant)
|
||||
need = variant.weights_bytes + overhead
|
||||
vram = budget.usable_vram_bytes
|
||||
if need + ctx_bytes(profile, min(TARGET_WINDOW, native)) <= vram:
|
||||
return VariantChoice(variant, zero_spill=True, reason_key="best-large-window")
|
||||
floor_kv = ctx_bytes(profile, min(FLOOR, native))
|
||||
if need + floor_kv <= vram:
|
||||
return VariantChoice(variant, zero_spill=True, reason_key="best-fits")
|
||||
if need + floor_kv <= vram + budget.ram_available_bytes:
|
||||
decision = entry.launch_plan(variant, budget).decision
|
||||
if isinstance(decision, PhysicsRefusal):
|
||||
return None
|
||||
if decision.spilled:
|
||||
return VariantChoice(variant, zero_spill=False, reason_key="smallest-fits-spilled")
|
||||
return None
|
||||
reason = "best-large-window" if decision.window >= min(TARGET_WINDOW, entry.n_ctx_train or FLOOR) else "best-fits"
|
||||
return VariantChoice(variant, zero_spill=True, reason_key=reason)
|
||||
|
||||
|
||||
# ── recommendation: best quality that fits and isn't miserably slow ──
|
||||
|
||||
@@ -7,10 +7,10 @@ behavior measured on real hardware (llama.cpp, discrete NVIDIA on Windows/WDDM,
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from dataclasses import dataclass, field, replace
|
||||
|
||||
from hermes_cli.local_runtime.estimator import (
|
||||
HardwareBudget, ModelProfile, PhysicsRefusal, ctx_bytes, physics_check)
|
||||
HardwareBudget, ModelProfile, PhysicsRefusal, ctx_bytes, footprint_bytes, physics_check)
|
||||
|
||||
FLOOR = 64 * 1024 # = target; one internal constant
|
||||
_LADDER_GROWTH = 1.5
|
||||
@@ -64,7 +64,8 @@ def initial_window(profile: ModelProfile, budget: HardwareBudget, *, flash_atten
|
||||
everywhere, capped at native. ``overhead_bytes`` is runtime cost beyond weights+KV; zero keeps
|
||||
this pure physics for decision-table tests, production callers pass it.
|
||||
"""
|
||||
refusal = physics_check(profile, budget, FLOOR, flash_attention=flash_attention)
|
||||
refusal = physics_check(profile, budget, FLOOR, flash_attention=flash_attention,
|
||||
overhead_bytes=overhead_bytes)
|
||||
if refusal:
|
||||
return refusal
|
||||
|
||||
@@ -74,9 +75,13 @@ def initial_window(profile: ModelProfile, budget: HardwareBudget, *, flash_atten
|
||||
def kv(rung: int) -> int:
|
||||
return ctx_bytes(profile, rung, flash_attention=flash_attention)
|
||||
|
||||
def need(rung: int) -> int:
|
||||
return footprint_bytes(profile, rung, flash_attention=flash_attention,
|
||||
overhead_bytes=overhead_bytes)
|
||||
|
||||
best_zero_spill: int | None = None
|
||||
for rung in rungs:
|
||||
if profile.weights_bytes + overhead_bytes + kv(rung) > budget.usable_vram_bytes:
|
||||
if need(rung) > budget.usable_vram_bytes:
|
||||
break
|
||||
best_zero_spill = rung
|
||||
|
||||
@@ -91,15 +96,74 @@ def initial_window(profile: ModelProfile, budget: HardwareBudget, *, flash_atten
|
||||
for rung in rungs:
|
||||
if rung < window:
|
||||
continue
|
||||
if kv(rung) > cap:
|
||||
if (kv(rung) > cap
|
||||
or need(rung) > budget.usable_vram_bytes + budget.ram_available_bytes):
|
||||
break
|
||||
window = rung
|
||||
reason = f"floor held at {window // 1024}K; weights spill (deliberate price of the guarantee)"
|
||||
|
||||
kv_bytes = kv(window)
|
||||
return WindowDecision(window=window, reasons=[reason],
|
||||
spill_bytes=max(0, profile.weights_bytes + kv_bytes - budget.usable_vram_bytes),
|
||||
kv_on_gpu=kv_bytes <= budget.usable_vram_bytes)
|
||||
spill_bytes=max(0, need(window) - budget.usable_vram_bytes),
|
||||
kv_on_gpu=kv_bytes + overhead_bytes <= budget.usable_vram_bytes)
|
||||
|
||||
|
||||
@dataclass
|
||||
class LaunchPlan:
|
||||
decision: WindowDecision | PhysicsRefusal
|
||||
mtp_prefill: bool
|
||||
overhead_bytes: int
|
||||
|
||||
|
||||
def plan_launch(profile: ModelProfile, budget: HardwareBudget, *, mtp_capable: bool = False,
|
||||
fixed_overhead: int = RUNTIME_OVERHEAD_BYTES,
|
||||
requested_window: int | None = None) -> LaunchPlan:
|
||||
"""Window first, then prefill; price both postures at the effective window.
|
||||
|
||||
A restored window may fit only under lean MTP. Evaluate it before discarding it because
|
||||
stacked exceeds memory, and keep deliberate spill when neither posture is resident.
|
||||
"""
|
||||
if mtp_capable and profile.kv_scale == 1.0:
|
||||
profile = replace(profile, kv_scale=1.2)
|
||||
|
||||
initial: dict[bool, WindowDecision | PhysicsRefusal] = {}
|
||||
|
||||
def candidate(stacked: bool) -> LaunchPlan:
|
||||
overhead = fixed_overhead + ub_logits_bytes(
|
||||
profile.n_vocab, mtp_capable=mtp_capable, mtp_prefill=stacked)
|
||||
decision = initial_window(profile, budget, overhead_bytes=overhead)
|
||||
initial[stacked] = decision
|
||||
if isinstance(decision, WindowDecision) and requested_window:
|
||||
target = min(requested_window, profile.n_ctx_train or requested_window)
|
||||
if target > decision.window and physics_check(
|
||||
profile, budget, target, overhead_bytes=overhead) is None:
|
||||
need = footprint_bytes(profile, target, overhead_bytes=overhead)
|
||||
decision = WindowDecision(
|
||||
window=target, spill_bytes=max(0, need - budget.usable_vram_bytes),
|
||||
kv_on_gpu=ctx_bytes(profile, target) + overhead <= budget.usable_vram_bytes,
|
||||
reasons=[f"grown window restored ({target // 1024}K)"])
|
||||
return LaunchPlan(decision, stacked, overhead)
|
||||
|
||||
lean = candidate(False)
|
||||
if not mtp_capable:
|
||||
return lean
|
||||
stacked = candidate(True)
|
||||
if isinstance(stacked.decision, PhysicsRefusal):
|
||||
return lean
|
||||
if isinstance(lean.decision, PhysicsRefusal):
|
||||
return stacked
|
||||
if stacked.decision.window < lean.decision.window:
|
||||
return lean
|
||||
if not stacked.decision.spilled:
|
||||
return stacked
|
||||
# A previously granted window keeps its spill policy unless lean can make it resident.
|
||||
stacked_initial, lean_initial = initial[True], initial[False]
|
||||
if (requested_window and isinstance(stacked_initial, WindowDecision)
|
||||
and isinstance(lean_initial, WindowDecision) and not stacked_initial.spilled
|
||||
and stacked_initial.window >= lean_initial.window
|
||||
and stacked.decision.window > stacked_initial.window and lean.decision.spilled):
|
||||
return stacked
|
||||
return lean
|
||||
|
||||
|
||||
@dataclass
|
||||
|
||||
@@ -131,11 +131,18 @@ class PhysicsRefusal:
|
||||
message: str
|
||||
|
||||
|
||||
def footprint_bytes(profile: ModelProfile, window: int, *, flash_attention: bool = True,
|
||||
overhead_bytes: int = 0) -> int:
|
||||
"""Complete estimated footprint; the hardware budget already excludes its reserve."""
|
||||
return (profile.weights_bytes + ctx_bytes(profile, window, flash_attention=flash_attention)
|
||||
+ max(0, overhead_bytes))
|
||||
|
||||
|
||||
def physics_check(profile: ModelProfile, budget: HardwareBudget,
|
||||
floor: int, *, flash_attention: bool = True) -> PhysicsRefusal | None:
|
||||
needed = (profile.weights_bytes
|
||||
+ ctx_bytes(profile, min(floor, profile.n_ctx_train or floor),
|
||||
flash_attention=flash_attention))
|
||||
floor: int, *, flash_attention: bool = True,
|
||||
overhead_bytes: int = 0) -> PhysicsRefusal | None:
|
||||
needed = footprint_bytes(profile, min(floor, profile.n_ctx_train or floor),
|
||||
flash_attention=flash_attention, overhead_bytes=overhead_bytes)
|
||||
available = budget.usable_vram_bytes + budget.ram_available_bytes
|
||||
if needed <= available:
|
||||
return None
|
||||
@@ -144,4 +151,4 @@ def physics_check(profile: ModelProfile, budget: HardwareBudget,
|
||||
needed_bytes=needed, available_bytes=available,
|
||||
message=(f"{profile.name}: needs ~{needed / gib:.1f} GiB at the "
|
||||
f"{floor // 1024}K floor but only ~{available / gib:.1f} GiB "
|
||||
"of VRAM+RAM exist — try a smaller quant (UD-Q3/Q2)"))
|
||||
"of VRAM+RAM are available — try a smaller model or a supported smaller quant"))
|
||||
|
||||
@@ -73,14 +73,15 @@ def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
|
||||
get_supervisor, refresh_local_runtime, staged_models)
|
||||
from hermes_cli.local_runtime.context_policy import growth_decision
|
||||
from hermes_cli.local_runtime.estimator import profile_from_gguf
|
||||
from hermes_cli.local_runtime.gguf import read_gguf_header
|
||||
from hermes_cli.local_runtime.gguf import model_id_from_stem, read_gguf_header
|
||||
from hermes_cli.local_runtime.hardware import probe_budget
|
||||
from hermes_cli.local_runtime.presets import preset_for_model, read_preset_decisions
|
||||
|
||||
sup = get_supervisor()
|
||||
if sup is None or not is_managed_endpoint(base_url):
|
||||
return None
|
||||
|
||||
gguf = next((p for p in staged_models() if p.stem.startswith(model_id) or model_id in p.stem), None)
|
||||
gguf = next((p for p in staged_models() if model_id_from_stem(p.stem) == model_id), None)
|
||||
if gguf is None:
|
||||
return None
|
||||
|
||||
@@ -95,11 +96,12 @@ def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
|
||||
except Exception: # noqa: BLE001
|
||||
server_idle = False
|
||||
|
||||
budget = probe_budget(planning=True)
|
||||
decision = growth_decision(
|
||||
# Capacity budget, not live-free: growth executes via a server bounce, so the grown
|
||||
# instance loads onto a freed card. Live-free is distorted by the very model being grown
|
||||
# — it reads its own residency as unavailable and vetoes rungs that fit.
|
||||
profile, probe_budget(planning=True),
|
||||
profile, budget,
|
||||
current_window=current_window,
|
||||
session_tokens=session_tokens,
|
||||
measured_decode_tok_s=measured_decode_tok_s,
|
||||
@@ -113,6 +115,11 @@ def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
|
||||
logger.debug("growth %s: %s (%s)", model_id, decision.action, decision.reason)
|
||||
return None
|
||||
|
||||
plan = preset_for_model(gguf, budget, set(), requested_window=decision.next_window)
|
||||
if plan is None or plan.refusal or plan.window < decision.next_window:
|
||||
logger.debug("growth %s: complete launch footprint does not admit the next rung", model_id)
|
||||
return None
|
||||
|
||||
logger.info("context growth %s: %s", model_id, decision.reason)
|
||||
save_window_override(model_id, decision.next_window)
|
||||
if not refresh_local_runtime():
|
||||
@@ -120,4 +127,8 @@ def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
|
||||
# compresses instead of overflowing a stale window.
|
||||
logger.warning("growth %s: server refresh failed; compression proceeds", model_id)
|
||||
return None
|
||||
return decision.next_window
|
||||
materialized = read_preset_decisions().get(model_id)
|
||||
if materialized is None or materialized.window < decision.next_window:
|
||||
logger.warning("growth %s: refreshed preset did not grant the requested window", model_id)
|
||||
return None
|
||||
return materialized.window
|
||||
|
||||
@@ -4,14 +4,15 @@ launch decisions.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
from dataclasses import dataclass, replace
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
from hermes_cli.local_runtime.context_policy import (
|
||||
RUNTIME_OVERHEAD_BYTES, WindowDecision, initial_window, launch_args, ub_logits_bytes)
|
||||
RUNTIME_OVERHEAD_BYTES, launch_args, plan_launch, ub_logits_bytes)
|
||||
from hermes_cli.local_runtime.estimator import (
|
||||
HardwareBudget, ModelProfile, PhysicsRefusal, ctx_bytes, profile_from_gguf)
|
||||
HardwareBudget, PhysicsRefusal, ctx_bytes, footprint_bytes, profile_from_gguf)
|
||||
from hermes_cli.local_runtime.gguf import model_id_from_stem, read_gguf_header
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -56,53 +57,30 @@ def _asset_path(asset) -> "Path | None":
|
||||
return path if path.exists() else None
|
||||
|
||||
|
||||
def _choose_mtp_posture(profile: ModelProfile, budget: HardwareBudget,
|
||||
fixed_overhead: int) -> tuple[bool, int]:
|
||||
"""(mtp_prefill, logits_bytes) for an MTP model — window first, prefill second.
|
||||
def _draft_fits(path: Path, profile, budget: HardwareBudget, window: int, overhead: int) -> bool:
|
||||
"""Optional draft never shrinks the advertised window or displaces its GPU buffers.
|
||||
|
||||
Price the launch under both postures and keep whichever grants the larger window: the stacked
|
||||
posture's bigger compute buffer buys ~3x short-prompt prefill but costs ~2 GiB that would
|
||||
otherwise be window (measured at 256K the ub512 posture still prefills at 2.7K tok/s), so
|
||||
never trade context away for prefill. Same window -> stacked.
|
||||
The catalog does not know the draft's context layout. Admit it only after reading the file;
|
||||
draft KV defaults to f16, independently of the target's q8 cache.
|
||||
"""
|
||||
plain_logits = ub_logits_bytes(profile.n_vocab, mtp_capable=True)
|
||||
stacked_logits = ub_logits_bytes(profile.n_vocab, mtp_capable=True, mtp_prefill=True)
|
||||
stacked = initial_window(profile, budget, overhead_bytes=fixed_overhead + stacked_logits)
|
||||
plain = initial_window(profile, budget, overhead_bytes=fixed_overhead + plain_logits)
|
||||
if (not isinstance(stacked, PhysicsRefusal) and not stacked.spilled
|
||||
and (isinstance(plain, PhysicsRefusal) or stacked.window >= plain.window)):
|
||||
return True, stacked_logits
|
||||
return False, plain_logits
|
||||
|
||||
|
||||
def _restore_grown_window(model_id: str, profile: ModelProfile, budget: HardwareBudget,
|
||||
decision: WindowDecision, overhead: int) -> WindowDecision:
|
||||
"""Session growth (growth.py): a persisted override lifts the launch window to where the ladder
|
||||
last grew it — capped at native, and only when physics still clears the bigger window on THIS
|
||||
boot's budget (a smaller-VRAM day re-fits honestly back down)."""
|
||||
try:
|
||||
from hermes_cli.local_runtime.growth import load_window_overrides
|
||||
|
||||
override = load_window_overrides().get(model_id)
|
||||
native = profile.n_ctx_train or decision.window
|
||||
if override and override > decision.window:
|
||||
target = min(int(override), native)
|
||||
kv = ctx_bytes(profile, target)
|
||||
need = profile.weights_bytes + kv + overhead
|
||||
if need <= budget.usable_vram_bytes + budget.ram_available_bytes:
|
||||
return WindowDecision(
|
||||
window=target, spill_bytes=max(0, need - budget.usable_vram_bytes),
|
||||
kv_on_gpu=kv <= budget.usable_vram_bytes,
|
||||
reasons=[f"grown window restored ({target // 1024}K)"])
|
||||
except Exception as exc: # noqa: BLE001 — overrides are advisory
|
||||
logger.debug("window override skipped for %s: %s", model_id, exc)
|
||||
return decision
|
||||
draft = profile_from_gguf(read_gguf_header(path))
|
||||
draft_need = footprint_bytes(
|
||||
draft, window, flash_attention=False,
|
||||
overhead_bytes=RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(draft.n_vocab, mtp_capable=False))
|
||||
except (ValueError, OSError) as exc:
|
||||
logger.warning("draft omitted %s: %s", path.name, exc)
|
||||
return False
|
||||
return (footprint_bytes(profile, window, overhead_bytes=overhead) + draft_need
|
||||
<= budget.usable_vram_bytes + budget.ram_available_bytes
|
||||
and ctx_bytes(profile, window) + overhead + draft_need <= budget.usable_vram_bytes)
|
||||
|
||||
|
||||
def _preset_for(gguf: Path, budget: HardwareBudget,
|
||||
mtp_capable: set[str]) -> PresetEntry | None:
|
||||
def preset_for_model(gguf: Path, budget: HardwareBudget,
|
||||
mtp_capable: set[str], *, requested_window: int | None = None) -> PresetEntry | None:
|
||||
"""The launch decision for one staged model, or None when its header is unreadable."""
|
||||
from hermes_cli.local_runtime.catalog import entry_for_model
|
||||
from hermes_cli.local_runtime.growth import load_window_overrides
|
||||
|
||||
model_id = model_id_from_stem(gguf.stem)
|
||||
try:
|
||||
@@ -113,30 +91,22 @@ def _preset_for(gguf: Path, budget: HardwareBudget,
|
||||
return None
|
||||
entry = entry_for_model(model_id)
|
||||
is_mtp = entry.mtp if entry is not None else model_id in mtp_capable
|
||||
if is_mtp and profile.kv_scale == 1.0:
|
||||
# Header-derived profiles don't know about MTP's draft context; apply the calibrated KV
|
||||
# multiplier so the launch fit prices what the server will actually allocate.
|
||||
profile = replace(profile, kv_scale=1.2)
|
||||
|
||||
mmproj_path = _asset_path(entry.mmproj) if entry is not None else None
|
||||
# Overhead beyond weights+KV: runtime buffers, the vision projector when present, and the
|
||||
# logits buffers of whichever microbatch/MTP posture launch_args will choose — flag and price
|
||||
# decided together, from the same facts.
|
||||
fixed_overhead = RUNTIME_OVERHEAD_BYTES + (
|
||||
entry.mmproj.size_bytes if entry is not None and mmproj_path is not None else 0)
|
||||
if is_mtp:
|
||||
mtp_prefill, logits_bytes = _choose_mtp_posture(profile, budget, fixed_overhead)
|
||||
else:
|
||||
mtp_prefill, logits_bytes = False, ub_logits_bytes(profile.n_vocab, mtp_capable=False)
|
||||
overhead = fixed_overhead + logits_bytes
|
||||
decision = initial_window(profile, budget, overhead_bytes=overhead)
|
||||
plan = plan_launch(profile, budget, mtp_capable=is_mtp, fixed_overhead=fixed_overhead,
|
||||
requested_window=(load_window_overrides().get(model_id)
|
||||
if requested_window is None else requested_window))
|
||||
decision = plan.decision
|
||||
if isinstance(decision, PhysicsRefusal):
|
||||
return PresetEntry(model_id=model_id, window=0, spilled=False, refusal=decision.message)
|
||||
decision = _restore_grown_window(model_id, profile, budget, decision, overhead)
|
||||
|
||||
# The launch flags MUST match the pricing above (same entry/is_mtp/posture).
|
||||
# Router discovery is preset-only: refused files must never autoload with stock fit.
|
||||
keys = _args_to_keys(launch_args(
|
||||
profile, decision, mtp_capable=is_mtp, uma=budget.uma, mtp_prefill=mtp_prefill,
|
||||
profile, decision, mtp_capable=is_mtp, uma=budget.uma, mtp_prefill=plan.mtp_prefill,
|
||||
mtp_draft_depth=entry.mtp_draft_depth if entry is not None else 3))
|
||||
keys["model"] = str(gguf)
|
||||
if entry is not None and is_mtp:
|
||||
# Integrated-MTP targets sample on the backend, and so does the draft (pairing validated
|
||||
# against the vendor's published llama.cpp recipes).
|
||||
@@ -155,7 +125,7 @@ def _preset_for(gguf: Path, budget: HardwareBudget,
|
||||
if mmproj_path is not None:
|
||||
keys["mmproj"] = str(mmproj_path)
|
||||
draft_path = _asset_path(entry.draft) if decision.spilled else None
|
||||
if draft_path is not None:
|
||||
if draft_path is not None and _draft_fits(draft_path, profile, budget, decision.window, plan.overhead_bytes):
|
||||
keys["model-draft"] = str(draft_path)
|
||||
keys["spec-type"] = "draft-dspark"
|
||||
# Unsloth's measured cliff: acceptance 83% at 2-3 drafts, collapses at 4.
|
||||
@@ -172,18 +142,31 @@ def generate_presets(models_dir: Path, budget: HardwareBudget, preset_path: Path
|
||||
|
||||
entries: list[PresetEntry] = []
|
||||
sections: list[str] = []
|
||||
for gguf in staged_in(models_dir, require_complete=False):
|
||||
entry = _preset_for(gguf, budget, mtp_capable or set())
|
||||
for gguf in staged_in(models_dir):
|
||||
entry = preset_for_model(gguf, budget, mtp_capable or set())
|
||||
if entry is None:
|
||||
continue
|
||||
entries.append(entry)
|
||||
# INI comments preserve non-flag facts atomically with the launch policy.
|
||||
sections.append("# hermes-decision: " + json.dumps({
|
||||
"model_id": entry.model_id, "window": entry.window,
|
||||
"spilled": entry.spilled, "refusal": entry.refusal}) + "\n")
|
||||
if entry.keys is not None:
|
||||
body = "\n".join(f"{k} = {v}" for k, v in entry.keys.items())
|
||||
sections.append(f"[{entry.model_id}]\n{body}\n")
|
||||
|
||||
preset_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
preset_path.write_text("\n".join(sections), encoding="utf-8")
|
||||
logger.info("wrote %d preset sections to %s", len(sections), preset_path)
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
fd, tmp = tempfile.mkstemp(prefix=preset_path.name, suffix=".tmp", dir=preset_path.parent)
|
||||
try:
|
||||
with os.fdopen(fd, "w", encoding="utf-8") as stream:
|
||||
stream.write("\n".join(sections))
|
||||
os.replace(tmp, preset_path)
|
||||
finally:
|
||||
Path(tmp).unlink(missing_ok=True)
|
||||
logger.info("wrote %d preset sections to %s", sum(e.keys is not None for e in entries), preset_path)
|
||||
return entries
|
||||
|
||||
|
||||
@@ -198,12 +181,21 @@ def read_preset_decisions(preset_path: Path | None = None) -> dict[str, PresetEn
|
||||
preset_path = runtimes_root() / "presets.ini"
|
||||
out: dict[str, PresetEntry] = {}
|
||||
try:
|
||||
parser = configparser.ConfigParser()
|
||||
parser.read(preset_path, encoding="utf-8")
|
||||
parser = configparser.ConfigParser(interpolation=None)
|
||||
text = preset_path.read_text(encoding="utf-8")
|
||||
parser.read_string(text)
|
||||
recorded = {}
|
||||
for line in text.splitlines():
|
||||
if line.startswith("# hermes-decision: "):
|
||||
fact = json.loads(line.removeprefix("# hermes-decision: "))
|
||||
recorded[fact["model_id"]] = fact
|
||||
if fact.get("refusal"):
|
||||
out[fact["model_id"]] = PresetEntry(**fact)
|
||||
for section in parser.sections():
|
||||
out[section] = PresetEntry(
|
||||
model_id=section, window=parser.getint(section, "ctx-size", fallback=0),
|
||||
spilled=parser.has_option(section, "override-tensor"))
|
||||
spilled=recorded.get(section, {}).get("spilled", parser.has_option(section, "override-tensor")),
|
||||
keys=dict(parser[section]))
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.debug("preset read-back failed: %s", exc)
|
||||
return out
|
||||
|
||||
@@ -152,7 +152,6 @@ class LlamaServerSupervisor:
|
||||
"--host", "127.0.0.1",
|
||||
"--port", str(self.port),
|
||||
"--api-key", self.api_key,
|
||||
"--models-dir", str(self.models_dir),
|
||||
"--models-max", str(self.models_max),
|
||||
# Residency contract: a chat request to a staged-but-unloaded model loads it (slow
|
||||
# first token) instead of a bare 400/404 after an eject.
|
||||
@@ -167,6 +166,8 @@ class LlamaServerSupervisor:
|
||||
]
|
||||
if self.preset_path and self.preset_path.exists():
|
||||
cmd += ["--models-preset", str(self.preset_path)]
|
||||
else:
|
||||
cmd += ["--models-dir", str(self.models_dir)]
|
||||
cmd += self.extra_args
|
||||
self.log_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
if self._log_handle is not None:
|
||||
|
||||
@@ -552,12 +552,7 @@ def _catalog_row(entry, budget, recommended, recommended_reason, staged_ids) ->
|
||||
return row
|
||||
|
||||
variant = choice.variant
|
||||
# Same overhead the launch decision prices (runtime buffers + vision projector + microbatch/MTP
|
||||
# logits): the row must advertise the window the model will actually get, not a paper number.
|
||||
overhead = (context_policy.RUNTIME_OVERHEAD_BYTES
|
||||
+ (entry.mmproj.size_bytes if entry.mmproj else 0)
|
||||
+ context_policy.ub_logits_bytes(entry.n_vocab, mtp_capable=entry.mtp))
|
||||
decision = context_policy.initial_window(entry.profile(variant), budget, overhead_bytes=overhead)
|
||||
decision = entry.launch_plan(variant, budget).decision
|
||||
download_total = entry.download_bytes(variant)
|
||||
row.update({
|
||||
"fits": True, "model_id": variant.model_id, "quant": variant.quant,
|
||||
|
||||
@@ -28,7 +28,7 @@ def _stage(home, name):
|
||||
def _write_presets(home, *model_ids):
|
||||
pdir = home / "runtimes" / "llamacpp"
|
||||
pdir.mkdir(parents=True, exist_ok=True)
|
||||
body = "\n".join(f"[{m}]\nctx-size = 65536\n" for m in model_ids)
|
||||
body = "\n".join(f"[{m}]\nmodel = {home / 'models' / (m + '.gguf')}\nctx-size = 65536\n" for m in model_ids)
|
||||
(pdir / "presets.ini").write_text(body, encoding="utf-8")
|
||||
|
||||
|
||||
@@ -49,6 +49,16 @@ def test_presets_current_when_every_staged_model_is_covered(hermes_home):
|
||||
assert _presets_stale() is False
|
||||
|
||||
|
||||
def test_legacy_presets_without_model_paths_are_regenerated(hermes_home):
|
||||
from hermes_cli.local_runtime.bootstrap import _presets_stale
|
||||
|
||||
_stage(hermes_home, "model-a")
|
||||
_write_presets(hermes_home, "model-a")
|
||||
ini = hermes_home / "runtimes/llamacpp/presets.ini"
|
||||
ini.write_text("[model-a]\nctx-size = 65536\n")
|
||||
assert _presets_stale()
|
||||
|
||||
|
||||
def test_no_models_is_never_stale(hermes_home):
|
||||
from hermes_cli.local_runtime.bootstrap import _presets_stale
|
||||
|
||||
|
||||
@@ -151,6 +151,45 @@ def test_find_entry_for_model_resolves_split_ids():
|
||||
assert variant.quant == "UD-Q4_K_XL"
|
||||
|
||||
|
||||
def test_catalog_and_preset_agree_on_identical_model_facts(tmp_path, monkeypatch):
|
||||
from types import SimpleNamespace
|
||||
|
||||
from hermes_cli.local_runtime import bootstrap, catalog, presets
|
||||
from hermes_cli.local_runtime.context_policy import RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
|
||||
from hermes_cli.local_runtime.estimator import ctx_bytes
|
||||
from hermes_cli.web_routers.local_models import _catalog_row
|
||||
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
monkeypatch.setattr("hermes_cli.web_routers.local_models._engine_too_old", lambda tag: False)
|
||||
for entry in catalog.CATALOG:
|
||||
variant = entry.variants[0]
|
||||
profile = entry.profile(variant)
|
||||
path = tmp_path / f"{variant.model_id}.gguf"
|
||||
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(sampling_defaults={}))
|
||||
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: profile)
|
||||
if entry.mmproj:
|
||||
asset = bootstrap.assets_dir() / entry.mmproj.local_name
|
||||
asset.parent.mkdir(parents=True, exist_ok=True)
|
||||
asset.touch()
|
||||
for vram in (16, 24, 32, 48):
|
||||
for uma in (False, True):
|
||||
machine = HardwareBudget(int(vram * GIB * 0.8), vram * GIB,
|
||||
0 if uma else 32 * GIB, uma)
|
||||
row = _catalog_row(entry, machine, None, None, set())
|
||||
preset = presets.preset_for_model(path, machine, set())
|
||||
assert row["fits"] == (preset.refusal is None)
|
||||
if preset.refusal:
|
||||
continue
|
||||
assert row["start_window"] == preset.window
|
||||
assert row["spilled"] == preset.spilled
|
||||
overhead = (RUNTIME_OVERHEAD_BYTES + (entry.mmproj.size_bytes if entry.mmproj else 0)
|
||||
+ ub_logits_bytes(profile.n_vocab, mtp_capable=entry.mtp,
|
||||
mtp_prefill=preset.keys.get("ubatch-size") == "2048" and entry.mtp))
|
||||
need = profile.weights_bytes + ctx_bytes(profile, preset.window) + overhead
|
||||
assert preset.spilled == (need > machine.usable_vram_bytes)
|
||||
assert need <= machine.usable_vram_bytes + machine.ram_available_bytes
|
||||
|
||||
|
||||
def test_hybrid_long_context_stays_cheap():
|
||||
"""The reason Nemotron/Qwen3.6 headline the catalog: their priced
|
||||
64K-floor KV must be a small fraction of a dense model's."""
|
||||
|
||||
@@ -177,6 +177,38 @@ def test_physics_check_prices_at_floor_not_native():
|
||||
assert physics_check(p, card(24, ram_gib=8), FLOOR) is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("uma", [False, True])
|
||||
def test_initial_window_accounts_for_overhead_in_every_verdict(uma):
|
||||
from dataclasses import replace
|
||||
|
||||
profile = hybrid(weights_gib=8, native=FLOOR)
|
||||
base = profile.weights_bytes + ctx_bytes(profile, FLOOR)
|
||||
overhead = 2 * GIB
|
||||
budget = HardwareBudget(base + GIB, base + GIB, 0 if uma else 4 * GIB, uma)
|
||||
decision = initial_window(profile, budget, overhead_bytes=overhead)
|
||||
if uma:
|
||||
assert isinstance(decision, PhysicsRefusal)
|
||||
assert decision.needed_bytes == base + overhead
|
||||
else:
|
||||
assert isinstance(decision, WindowDecision)
|
||||
assert decision.spill_bytes == GIB
|
||||
|
||||
exact = replace(budget, usable_vram_bytes=base + overhead, ram_available_bytes=0)
|
||||
assert not initial_window(profile, exact, overhead_bytes=overhead).spilled
|
||||
short = replace(exact, usable_vram_bytes=exact.usable_vram_bytes - 1)
|
||||
assert isinstance(initial_window(profile, short, overhead_bytes=overhead), PhysicsRefusal)
|
||||
|
||||
# A cheap-KV model may grow on the spill path, but only into memory that exists.
|
||||
growing = hybrid(weights_gib=20, full_layers=4, recurrent_layers=0,
|
||||
per_token_f16=1024, native=1024 * KIB)
|
||||
floor_need = growing.weights_bytes + ctx_bytes(growing, FLOOR) + overhead
|
||||
limited = HardwareBudget(8 * GIB, 8 * GIB, floor_need - 8 * GIB)
|
||||
decision = initial_window(growing, limited, overhead_bytes=overhead)
|
||||
assert isinstance(decision, WindowDecision)
|
||||
assert decision.window == FLOOR
|
||||
assert decision.spill_bytes == limited.ram_available_bytes
|
||||
|
||||
|
||||
# ── ladder + initial window ──────────────────────────────────
|
||||
|
||||
|
||||
|
||||
@@ -230,6 +230,121 @@ def test_preset_restores_grown_window_midladder(hermes_home, tmp_path, monkeypat
|
||||
assert restored.window >= grown, "override must lift the launch window"
|
||||
|
||||
|
||||
def test_mtp_plan_matches_cost_at_initial_and_restored_windows(hermes_home, tmp_path, monkeypatch):
|
||||
from dataclasses import replace
|
||||
from types import SimpleNamespace
|
||||
|
||||
from hermes_cli.local_runtime import presets
|
||||
from hermes_cli.local_runtime.context_policy import FLOOR, RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile, ctx_bytes
|
||||
from hermes_cli.local_runtime.growth import save_window_override
|
||||
|
||||
gib = 1 << 30
|
||||
profile = ModelProfile(name="mtp-fit", weights_bytes=16 * gib, embd_table_bytes=0,
|
||||
n_ctx_train=262144, layers=[(LayerKind.FULL, 4096)] * 32,
|
||||
moe=True, n_vocab=151936)
|
||||
priced = replace(profile, kv_scale=1.2)
|
||||
lean = RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(profile.n_vocab, mtp_capable=True)
|
||||
stacked = RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(profile.n_vocab, mtp_capable=True, mtp_prefill=True)
|
||||
mdir = tmp_path / "models"
|
||||
_stage_fake_gguf(mdir, profile.name)
|
||||
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(sampling_defaults={}))
|
||||
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: profile)
|
||||
|
||||
def generate(device, ram, override=0):
|
||||
save_window_override(profile.name, override)
|
||||
budget = HardwareBudget(device, device, ram)
|
||||
return presets.generate_presets(mdir, budget, tmp_path / "presets.ini", {profile.name})[0]
|
||||
|
||||
floor_need = profile.weights_bytes + ctx_bytes(priced, FLOOR)
|
||||
initial = generate(floor_need + lean, 8 * gib)
|
||||
assert initial.window == FLOOR
|
||||
assert not initial.spilled
|
||||
assert "ubatch-size" not in initial.keys
|
||||
assert initial.keys["spec-type"] == "draft-mtp"
|
||||
# Persisting a floor grant must not turn a lean spilled boot into stacked prefill.
|
||||
for override in (0, FLOOR, 73728):
|
||||
spilled_boot = generate(16 * gib, 64 * gib, override)
|
||||
assert spilled_boot.spilled and "ubatch-size" not in spilled_boot.keys
|
||||
|
||||
device = floor_need + stacked
|
||||
control = generate(device, 8 * gib)
|
||||
assert control.keys["ubatch-size"] == "2048"
|
||||
grown_window = 73728
|
||||
for ram in (8 * gib, 0):
|
||||
# Also preserve a grown window when stacked exceeds total memory, not just VRAM.
|
||||
grown = generate(device, ram, grown_window)
|
||||
assert grown.window == grown_window
|
||||
assert not grown.spilled
|
||||
assert "ubatch-size" not in grown.keys
|
||||
assert "override-tensor" not in grown.keys
|
||||
assert grown.keys["spec-type"] == "draft-mtp"
|
||||
assert profile.weights_bytes + ctx_bytes(priced, grown.window) + lean <= device
|
||||
|
||||
both_spill = generate(device, 64 * gib, 147456)
|
||||
assert both_spill.window == 147456 and both_spill.spilled
|
||||
assert both_spill.keys["ubatch-size"] == "2048"
|
||||
assert "override-tensor" in both_spill.keys
|
||||
smaller_boot = generate(floor_need + lean, 0, grown_window)
|
||||
assert smaller_boot.window == FLOOR and not smaller_boot.spilled
|
||||
assert profile.weights_bytes + ctx_bytes(priced, control.window) + stacked <= device
|
||||
|
||||
|
||||
def test_growth_requires_an_admissible_materialized_preset(hermes_home, tmp_path, monkeypatch):
|
||||
from dataclasses import replace
|
||||
from types import SimpleNamespace
|
||||
|
||||
from hermes_cli.local_runtime import bootstrap, catalog, growth, hardware, presets
|
||||
from hermes_cli.local_runtime.context_policy import FLOOR, RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget, ctx_bytes
|
||||
|
||||
entry = next(e for e in catalog.CATALOG if e.mtp and e.mmproj)
|
||||
model_id = entry.variants[-1].model_id
|
||||
mdir = tmp_path / "models"
|
||||
_stage_fake_gguf(mdir, model_id)
|
||||
profile = replace(entry.profile(entry.variants[-1]), kv_scale=1.0)
|
||||
monkeypatch.setattr(bootstrap, "staged_models", lambda: list(mdir.glob("*.gguf")))
|
||||
monkeypatch.setattr(bootstrap, "get_supervisor", lambda: SimpleNamespace(is_idle=lambda m: True))
|
||||
monkeypatch.setattr(growth, "is_managed_endpoint", lambda url: True)
|
||||
from hermes_cli.local_runtime import gguf, estimator
|
||||
monkeypatch.setattr(gguf, "read_gguf_header", lambda p: _header_stub())
|
||||
monkeypatch.setattr(estimator, "profile_from_gguf", lambda h: profile)
|
||||
monkeypatch.setattr(presets, "read_gguf_header", lambda p: _header_stub())
|
||||
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: profile)
|
||||
asset = bootstrap.assets_dir() / entry.mmproj.local_name
|
||||
asset.parent.mkdir(parents=True, exist_ok=True)
|
||||
asset.touch()
|
||||
overhead = RUNTIME_OVERHEAD_BYTES + entry.mmproj.size_bytes + ub_logits_bytes(profile.n_vocab, mtp_capable=True)
|
||||
priced = replace(profile, kv_scale=1.2)
|
||||
next_window = FLOOR * 3 // 2
|
||||
floor_need = profile.weights_bytes + ctx_bytes(priced, FLOOR) + overhead
|
||||
next_need = profile.weights_bytes + ctx_bytes(priced, next_window) + overhead
|
||||
budget = HardwareBudget(floor_need, floor_need, 0, True)
|
||||
monkeypatch.setattr(hardware, "probe_budget", lambda **kw: budget)
|
||||
from hermes_cli.local_runtime.binaries import runtimes_root
|
||||
preset_path = runtimes_root() / "presets.ini"
|
||||
calls = []
|
||||
|
||||
def refresh():
|
||||
calls.append(True)
|
||||
presets.generate_presets(mdir, budget, preset_path)
|
||||
return True
|
||||
|
||||
monkeypatch.setattr(bootstrap, "refresh_local_runtime", refresh)
|
||||
args = dict(base_url="http://127.0.0.1:1/v1", session_tokens=FLOOR, current_window=FLOOR)
|
||||
assert growth.maybe_grow_window(model_id, **args) is None
|
||||
assert not calls and not growth.load_window_overrides()
|
||||
budget = replace(budget, usable_vram_bytes=next_need, total_device_bytes=next_need)
|
||||
assert growth.maybe_grow_window(model_id, **args) == next_window
|
||||
assert presets.read_preset_decisions(preset_path)[model_id].window == next_window
|
||||
assert growth.load_window_overrides()[model_id] == next_window
|
||||
|
||||
# A restart that claims success but does not materialize the grant must not tell the agent it grew.
|
||||
monkeypatch.setattr(bootstrap, "refresh_local_runtime", lambda: True)
|
||||
budget = replace(budget, usable_vram_bytes=64 << 30, total_device_bytes=64 << 30)
|
||||
assert growth.maybe_grow_window(model_id, **{**args, "current_window": next_window}) is None
|
||||
|
||||
|
||||
def test_sampling_ladder_file_beats_catalog_beats_nothing(hermes_home, tmp_path, monkeypatch):
|
||||
"""The sampling deference ladder: the GGUF's own general.sampling.*
|
||||
wins per key, catalog fills only what the file left silent, and a
|
||||
|
||||
@@ -66,6 +66,61 @@ def test_status_lists_staged_models_with_labels(client, tmp_path):
|
||||
assert row["size_label"].endswith("GB")
|
||||
|
||||
|
||||
def test_status_tracks_preset_spill_and_restored_window(client, tmp_path, monkeypatch):
|
||||
from dataclasses import replace
|
||||
from types import SimpleNamespace
|
||||
|
||||
from hermes_cli.local_runtime import bootstrap, presets
|
||||
from hermes_cli.local_runtime.binaries import runtimes_root
|
||||
from hermes_cli.local_runtime.context_policy import FLOOR, RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile, ctx_bytes
|
||||
from hermes_cli.local_runtime.growth import save_window_override
|
||||
from hermes_cli.web_routers import local_models
|
||||
|
||||
# Dense spill has no override-tensor flag: status must use the recorded decision.
|
||||
profile = ModelProfile("status-mtp", 16 << 30, 0, 262144,
|
||||
[(LayerKind.FULL, 4096)] * 32, n_vocab=151936)
|
||||
model_id = profile.name
|
||||
_write_fake_gguf(bootstrap.models_dir() / f"{model_id}.gguf")
|
||||
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(sampling_defaults={}))
|
||||
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: profile)
|
||||
monkeypatch.setattr(local_models, "_state_endpoint", lambda: {"base_url": "http://127.0.0.1:1/v1"})
|
||||
|
||||
server_window = FLOOR
|
||||
|
||||
def router_response(running, route, **kwargs):
|
||||
if route == "/models":
|
||||
return {"data": [{"id": model_id, "status": {"value": "loaded"}}]}
|
||||
assert route == f"/props?model={model_id}"
|
||||
return {"default_generation_settings": {"n_ctx": server_window}}
|
||||
|
||||
monkeypatch.setattr(local_models, "_router_request", router_response)
|
||||
floor_need = profile.weights_bytes + ctx_bytes(replace(profile, kv_scale=1.2), FLOOR)
|
||||
lean = RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(profile.n_vocab, mtp_capable=True)
|
||||
stacked = RUNTIME_OVERHEAD_BYTES + ub_logits_bytes(profile.n_vocab, mtp_capable=True, mtp_prefill=True)
|
||||
ini = runtimes_root() / "presets.ini"
|
||||
grown = 73728
|
||||
for device, override, spilled in ((floor_need + lean - 1, FLOOR, True),
|
||||
(floor_need + stacked, grown, False)):
|
||||
save_window_override(model_id, override)
|
||||
preset = presets.generate_presets(bootstrap.models_dir(),
|
||||
HardwareBudget(device, device, 8 << 30), ini, {model_id})[0]
|
||||
assert preset.window == override and preset.spilled is spilled
|
||||
assert preset.keys["spec-type"] == "draft-mtp"
|
||||
assert "ubatch-size" not in preset.keys and "override-tensor" not in preset.keys
|
||||
# Deliberately differ from the plan to prove the server remains the grant authority.
|
||||
server_window = preset.window - 1024
|
||||
response = client.get("/api/local-models/status")
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
assert data["loaded_models"][model_id] == "loaded"
|
||||
placement = data["placement"][model_id]
|
||||
assert placement["spilled"] is spilled
|
||||
assert placement["window"] == preset.window
|
||||
assert placement["granted_window"] == server_window
|
||||
assert placement["granted_window_label"] == local_models._k_label(server_window)
|
||||
|
||||
|
||||
# ── hardware ─────────────────────────────────────────────────
|
||||
|
||||
|
||||
|
||||
86
tests/hermes_cli/test_local_preset_admission.py
Normal file
86
tests/hermes_cli/test_local_preset_admission.py
Normal file
@@ -0,0 +1,86 @@
|
||||
"""The router must serve only admitted presets, retaining refusal and spill facts on read-back."""
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
from hermes_cli.local_runtime import presets, supervisor
|
||||
from hermes_cli.local_runtime.estimator import HardwareBudget, ModelProfile
|
||||
|
||||
|
||||
def test_preset_roundtrip_keeps_refusals_and_dense_spill(tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
mdir = tmp_path / "models"
|
||||
mdir.mkdir()
|
||||
for name in ("allowed", "refused"):
|
||||
(mdir / f"{name}.gguf").touch()
|
||||
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(path=p, sampling_defaults={}))
|
||||
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: ModelProfile(
|
||||
name=h.path.stem, weights_bytes=(4 if h.path.stem == "allowed" else 40) << 30,
|
||||
embd_table_bytes=0, n_ctx_train=65536, layers=[]))
|
||||
ini = tmp_path / "presets.ini"
|
||||
generated = presets.generate_presets(mdir, HardwareBudget(2 << 30, 2 << 30, 8 << 30), ini)
|
||||
reread = presets.read_preset_decisions(ini)
|
||||
assert set(reread) == {p.model_id for p in generated}
|
||||
assert reread["refused"].refusal
|
||||
assert reread["allowed"].spilled
|
||||
assert reread["allowed"].keys["model"] == str(mdir / "allowed.gguf")
|
||||
assert "override-tensor" not in reread["allowed"].keys # Dense spill has no tensor-pattern override.
|
||||
|
||||
|
||||
def test_optional_draft_is_enabled_only_with_room_at_the_selected_window(tmp_path, monkeypatch):
|
||||
from dataclasses import replace
|
||||
from hermes_cli.local_runtime import bootstrap, catalog
|
||||
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
entry = next(e for e in catalog.CATALOG if e.draft)
|
||||
main = tmp_path / f"{entry.variants[0].model_id}.gguf"
|
||||
draft = bootstrap.assets_dir() / entry.draft.local_name
|
||||
draft.parent.mkdir(parents=True, exist_ok=True)
|
||||
draft.touch()
|
||||
main_profile = ModelProfile("main", 12 << 30, 0, 65536, [], moe=True)
|
||||
draft_profile = ModelProfile("draft", 1 << 30, 0, 65536, [])
|
||||
monkeypatch.setattr(presets, "read_gguf_header", lambda p: SimpleNamespace(path=p, sampling_defaults={}))
|
||||
monkeypatch.setattr(presets, "profile_from_gguf", lambda h: draft_profile if h.path == draft else main_profile)
|
||||
tight = HardwareBudget(8 << 30, 8 << 30, 6 << 30)
|
||||
result = presets.preset_for_model(main, tight, set())
|
||||
assert result.window == 65536 and result.spilled
|
||||
assert "model-draft" not in result.keys
|
||||
roomy = replace(tight, ram_available_bytes=16 << 30)
|
||||
with_draft = presets.preset_for_model(main, roomy, set())
|
||||
assert with_draft.window == result.window
|
||||
assert with_draft.keys["model-draft"] == str(draft)
|
||||
assert with_draft.keys["spec-type"] == "draft-dspark"
|
||||
# Full target-window f16 state and logits count even above the draft's native window.
|
||||
from hermes_cli.local_runtime.context_policy import RUNTIME_OVERHEAD_BYTES, ub_logits_bytes
|
||||
from hermes_cli.local_runtime.estimator import LayerKind, ctx_bytes
|
||||
|
||||
draft_profile = replace(draft_profile, n_ctx_train=32768,
|
||||
layers=[(LayerKind.FULL, 4096)] * 4, n_vocab=32768)
|
||||
draft_cost = (draft_profile.weights_bytes
|
||||
+ ctx_bytes(draft_profile, result.window, flash_attention=False)
|
||||
+ RUNTIME_OVERHEAD_BYTES
|
||||
+ ub_logits_bytes(draft_profile.n_vocab, mtp_capable=False))
|
||||
device_boundary = RUNTIME_OVERHEAD_BYTES + draft_cost
|
||||
exact = replace(roomy, usable_vram_bytes=device_boundary, total_device_bytes=device_boundary)
|
||||
assert "model-draft" in presets.preset_for_model(main, exact, set()).keys
|
||||
below = replace(exact, usable_vram_bytes=device_boundary - 1)
|
||||
assert "model-draft" not in presets.preset_for_model(main, below, set()).keys
|
||||
|
||||
# A draft too large for GPU memory is optional, not permission to move its buffers to RAM.
|
||||
draft_profile = replace(draft_profile, weights_bytes=9 << 30)
|
||||
assert "model-draft" not in presets.preset_for_model(main, roomy, set()).keys
|
||||
|
||||
|
||||
def test_supervisor_with_presets_does_not_scan_unadmitted_files(tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
ini = tmp_path / "presets.ini"
|
||||
ini.write_text("[allowed]\nmodel = allowed.gguf\nctx-size = 65536\n")
|
||||
calls = []
|
||||
monkeypatch.setattr(supervisor, "server_binary", lambda p: Path("llama-server"))
|
||||
monkeypatch.setattr(supervisor.subprocess, "Popen", lambda cmd, **kw: calls.append(cmd) or SimpleNamespace(pid=123))
|
||||
sup = supervisor.LlamaServerSupervisor(tmp_path, tmp_path, port=1234, preset_path=ini)
|
||||
try:
|
||||
sup._spawn()
|
||||
finally:
|
||||
sup._log_handle.close()
|
||||
assert "--models-preset" in calls[0]
|
||||
assert "--models-dir" not in calls[0]
|
||||
@@ -64,8 +64,14 @@ end-to-end and exposes no knobs:
|
||||
overflow in system RAM in the order that hurts least (expert weights
|
||||
first, never the attention cache), trading some speed to protect the
|
||||
context guarantee.
|
||||
- **Conversation compression only kicks in at the model's maximum
|
||||
window** — growth always comes first.
|
||||
- **Memory fit includes the launch configuration**, not just the model file:
|
||||
context state, runtime buffers, the vision projector, and MTP buffers all
|
||||
count. For multi-token prediction (MTP), Hermes uses smaller batches when
|
||||
larger batches would spill at the same context window. MTP stays enabled.
|
||||
The same calculation runs when a grown window is restored after restart.
|
||||
- **Conversation compression follows a growth check.** If a larger window
|
||||
cannot fit, generation is too slow, or the native maximum is reached,
|
||||
Hermes compresses instead of claiming a window the server did not receive.
|
||||
- Idle models are unloaded after 15 minutes to free GPU memory; they
|
||||
reload automatically on the next message.
|
||||
|
||||
|
||||
Reference in New Issue
Block a user