diff --git a/hermes_cli/anon_auth.py b/hermes_cli/anon_auth.py index 5460034dcc..ffc413f11a 100644 --- a/hermes_cli/anon_auth.py +++ b/hermes_cli/anon_auth.py @@ -284,7 +284,28 @@ def _mint_locked( # Per-process memo: one failed mint is enough for a process (a 429 or a closed gate must not be hit # twice); ``clear_dead_guest`` resets it because a retired credential is a reason to mint again. +# The bool is the unscoped (launch profile) slot; routed multiplex profiles each get their own entry +# in the set — profile A's 429 must not stop profile B from ever getting an identity. _mint_failed = False +_mint_failed_homes: set[str] = set() + + +def _mint_failed_for_profile() -> bool: + from hermes_constants import get_hermes_home_override, hermes_home_key + if get_hermes_home_override() is None: + return _mint_failed + return hermes_home_key() in _mint_failed_homes + + +def _set_mint_failed(failed: bool) -> None: + global _mint_failed + from hermes_constants import get_hermes_home_override, hermes_home_key + if get_hermes_home_override() is None: + _mint_failed = failed + elif failed: + _mint_failed_homes.add(hermes_home_key()) + else: + _mint_failed_homes.discard(hermes_home_key()) def _reconcile_and_provision(*, timeout_seconds: float, carries_inference: bool = True) -> Optional[Dict[str, Any]]: @@ -346,16 +367,15 @@ def ensure_portal_identity( """ if not explicit: raise ValueError("ensure_portal_identity: only explicit creators may call this (explicit=True)") - global _mint_failed if not guest_enabled(): return None - if _mint_failed and not current_nous_state(): - return None # this process already tried and failed; do not hammer the portal + if _mint_failed_for_profile() and not current_nous_state(): + return None # this profile already tried and failed in this process; do not hammer the portal try: return _reconcile_and_provision( timeout_seconds=timeout_seconds, carries_inference=carries_inference) except Exception: - _mint_failed = True + _set_mint_failed(True) raise @@ -401,8 +421,7 @@ def clear_dead_guest(reason: str, *, dead_token: Optional[str] = None) -> None: shared = _read_shared_nous_state() if token and is_guest_state(shared) and shared.get("anon_token") == token: _clear_shared_nous_state(reason) - global _mint_failed - _mint_failed = False + _set_mint_failed(False) logger.info("Nous free-tier identity retired (%s); a new one is set up on next use", reason) diff --git a/hermes_cli/banner.py b/hermes_cli/banner.py index 19d12da311..b21f7f0cbb 100644 --- a/hermes_cli/banner.py +++ b/hermes_cli/banner.py @@ -92,7 +92,13 @@ _UNCACHED = object() # compute() result that must not be memoized def _memo(cache_name: str, compute): - """Return the cached value under module global ``cache_name``, computing (and storing) it once.""" + """Return the cached value under module global ``cache_name``, computing (and storing) it once. + + Not consulted under a routed profile (HERMES_HOME override): every memo here is derived from the + launch home (its skills tree, its checkout), and the TUI gateway calls these per profile.""" + from hermes_constants import get_hermes_home_override + if get_hermes_home_override() is not None: + return compute() cached = globals()[cache_name] if cached is not None: return cached[0] diff --git a/hermes_cli/model_catalog.py b/hermes_cli/model_catalog.py index 2cb569ec4b..d938789212 100644 --- a/hermes_cli/model_catalog.py +++ b/hermes_cli/model_catalog.py @@ -37,9 +37,12 @@ SUPPORTED_SCHEMA_VERSION = 1 _HERMES_USER_AGENT = f"hermes-cli/{_HERMES_VERSION}" -# In-process cache, invalidated against the disk file's mtime and TTL. +# In-process cache, invalidated against the disk file's path + mtime and TTL. The path matters: +# under a multiplexed gateway each profile has its own ``/cache/model_catalog.json``, and +# mtime alone cannot tell two profiles' files apart. _catalog_cache: dict[str, Any] | None = None _catalog_cache_source_mtime: float = 0.0 +_catalog_cache_source_path: str = "" def _load_catalog_config() -> dict[str, Any]: @@ -193,11 +196,19 @@ def _spawn_catalog_swr_refresh(url: str) -> None: def _remember(data: dict[str, Any], mtime: float) -> dict[str, Any]: - global _catalog_cache, _catalog_cache_source_mtime + global _catalog_cache, _catalog_cache_source_mtime, _catalog_cache_source_path _catalog_cache, _catalog_cache_source_mtime = data, mtime + _catalog_cache_source_path = str(_cache_path()) return data +def _in_process_catalog() -> dict[str, Any] | None: + """The in-process copy when it mirrors the ACTIVE profile's cache file, else None.""" + if _catalog_cache is not None and _catalog_cache_source_path == str(_cache_path()): + return _catalog_cache + return None + + def get_catalog(*, force_refresh: bool = False) -> dict[str, Any]: """Parsed model catalog manifest, or ``{}`` on failure — never raises, so the CLI works offline (callers treat a missing provider/model as "use the in-repo fallback").""" @@ -210,8 +221,9 @@ def get_catalog(*, force_refresh: bool = False) -> dict[str, Any]: disk_fresh = disk_data is not None and (now - disk_mtime) < ttl_seconds if not force_refresh and disk_data is not None: - if disk_fresh and _catalog_cache is not None and disk_mtime == _catalog_cache_source_mtime: - return _catalog_cache + cached = _in_process_catalog() + if disk_fresh and cached is not None and disk_mtime == _catalog_cache_source_mtime: + return cached if not disk_fresh: # Stale-while-revalidate: serve the expired disk copy now and refresh off-thread so the # /model picker (which calls this on every open) never blocks on the manifest fetch. @@ -303,7 +315,8 @@ def _default_model_from_block(block: dict[str, Any] | None) -> str | None: def get_default_model_from_cache(provider: str) -> str | None: """The manifest's labeled default for ``provider`` (the model Hermes silently lands on when the user never picked one) — in-process then disk cache only, never a fetch.""" - found = _default_model_from_block(_block_of(_catalog_cache, provider)) if _catalog_cache is not None else None + cached = _in_process_catalog() + found = _default_model_from_block(_block_of(cached, provider)) if cached is not None else None if found: return found disk_data, _mtime = _read_disk_cache() @@ -331,6 +344,7 @@ def seed_cache_from_checkout(project_root: "Path | str") -> bool: def reset_cache() -> None: """Clear the in-process cache. Used by tests and ``hermes model --refresh``.""" - global _catalog_cache, _catalog_cache_source_mtime + global _catalog_cache, _catalog_cache_source_mtime, _catalog_cache_source_path _catalog_cache = None _catalog_cache_source_mtime = 0.0 + _catalog_cache_source_path = "" diff --git a/hermes_cli/models.py b/hermes_cli/models.py index 57cf83d1e9..d70dc09fcd 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -8,11 +8,13 @@ Origin module; cohesive clusters live in siblings and are re-imported here so from __future__ import annotations +import contextvars import copy import json import logging import os import re +import sys import threading import urllib.parse import urllib.request @@ -542,17 +544,21 @@ def _fetch_live_catalog_index(url: str, timeout: float, opener) -> Optional[tupl def fetch_openrouter_models( timeout: float = 8.0, *, force_refresh: bool = False) -> list[tuple[str, str]]: """Return the curated OpenRouter picker list, refreshed from the live catalog when possible.""" - global _openrouter_catalog_cache + # The curated list is filtered from this profile's manifest (``model_catalog.*`` config, its + # ``/cache`` copy), so a routed profile keeps its own slot instead of the module one. + from hermes_cli.models_profile_cache import profile_slot_get, profile_slot_set + _me = sys.modules[__name__] + cached = profile_slot_get(_me, "_openrouter_catalog_cache") - if _openrouter_catalog_cache is not None and not force_refresh: - return list(_openrouter_catalog_cache) + if cached is not None and not force_refresh: + return list(cached) # Cold process: serve from the persisted disk cache when fresh so the # picker doesn't re-download the full ~686KB catalog on every open. if not force_refresh: disk = _read_openrouter_catalog_disk() if disk: - _openrouter_catalog_cache = disk + profile_slot_set(_me, "_openrouter_catalog_cache", disk) return list(disk) # Remote catalog manifest first, in-repo snapshot when unreachable; the live /v1/models filter @@ -566,7 +572,7 @@ def fetch_openrouter_models( live = _fetch_live_catalog_index(_OPENROUTER_CATALOG_URL, timeout, _urlopen_model_catalog_request) if live is None: - return list(_openrouter_catalog_cache or fallback) + return list(cached or fallback) live_items, live_by_id = live # Free warm-up for the reasoning-capability cache: same payload the caps fetch would pull. @@ -592,10 +598,10 @@ def fetch_openrouter_models( curated.append((preferred_id, desc)) if not curated: - return list(_openrouter_catalog_cache or fallback) + return list(cached or fallback) if not curated[0][1]: curated[0] = (curated[0][0], "recommended") - _openrouter_catalog_cache = curated + profile_slot_set(_me, "_openrouter_catalog_cache", curated) _write_openrouter_catalog_disk(curated) return list(curated) @@ -1479,10 +1485,15 @@ def _spawn_swr_refresh(cache_key: str, refresh_fn=None) -> None: Failures are swallowed — the stale entry stays served until a later refresh succeeds. ``refresh_fn`` (no-args → fresh entry dict or None) lets ``custom:`` keys from :func:`cached_fetch_api_models` reuse the same inflight-dedupe scaffolding.""" + # Under a routed profile the inflight key includes the home: the same provider slug names a + # different disk cache and credential set per profile, so one profile's refresh must not + # suppress another's. Unscoped keeps the bare key (tests inspect the set by slug). + from hermes_constants import get_hermes_home_override, hermes_home_key + inflight_key = cache_key if get_hermes_home_override() is None else (hermes_home_key(), cache_key) with _swr_refresh_lock: - if cache_key in _swr_refresh_inflight: + if inflight_key in _swr_refresh_inflight: return - _swr_refresh_inflight.add(cache_key) + _swr_refresh_inflight.add(inflight_key) def _default_refresh(): live = provider_model_ids(cache_key, force_refresh=True) @@ -1499,9 +1510,11 @@ def _spawn_swr_refresh(cache_key: str, refresh_fn=None) -> None: logger.debug("SWR refresh failed for %s", cache_key, exc_info=True) finally: with _swr_refresh_lock: - _swr_refresh_inflight.discard(cache_key) + _swr_refresh_inflight.discard(inflight_key) - threading.Thread(target=_refresh, daemon=True, name=f"model-cache-swr-{cache_key}").start() + # copy_context: the refresh must read the calling profile's credentials and write ITS disk cache. + ctx = contextvars.copy_context() + threading.Thread(target=lambda: ctx.run(_refresh), daemon=True, name=f"model-cache-swr-{cache_key}").start() def _provider_models_cache_path() -> Path: @@ -1893,15 +1906,21 @@ def fetch_github_model_catalog( # Module-level cache: {model_id: max_prompt_tokens} _copilot_context_cache: dict[str, int] = {} _copilot_context_cache_time: float = 0.0 +_copilot_context_cache_key: Optional[str] = None # fingerprint of the api_key the entry was fetched with _COPILOT_CONTEXT_CACHE_TTL = 3600 # 1 hour def get_copilot_model_context(model_id: str, api_key: Optional[str] = None) -> Optional[int]: """``max_prompt_tokens`` for a Copilot model from the live /models API (cached in-process 1h; a miss on a fresh cache does not re-fetch), or None.""" - global _copilot_context_cache, _copilot_context_cache_time + global _copilot_context_cache, _copilot_context_cache_time, _copilot_context_cache_key - if _copilot_context_cache and (time.time() - _copilot_context_cache_time < _COPILOT_CONTEXT_CACHE_TTL): + # Keyed on the credential like fetch_github_model_catalog: the catalog (and its limits) is + # per-account, so another profile's token must not be served this entry. + from agent.credential_persistence import fingerprint_secret_value + key_fp = fingerprint_secret_value(api_key) + if (_copilot_context_cache and _copilot_context_cache_key == key_fp + and (time.time() - _copilot_context_cache_time < _COPILOT_CONTEXT_CACHE_TTL)): return _copilot_context_cache.get(model_id) catalog = fetch_github_model_catalog(api_key=api_key) @@ -1915,6 +1934,7 @@ def get_copilot_model_context(model_id: str, api_key: Optional[str] = None) -> O cache[mid] = max_prompt _copilot_context_cache = cache _copilot_context_cache_time = time.time() + _copilot_context_cache_key = key_fp return cache.get(model_id) @@ -2355,11 +2375,20 @@ _deepinfra_catalog_neg_cache: dict[str, float] = {} _DEEPINFRA_CATALOG_NEG_TTL = 60.0 # seconds +def _deepinfra_env(key: str) -> str: + """Profile-scoped ``.env``/environ read: under a multiplexed turn the launch env is not this profile's.""" + from hermes_cli.config import get_env_value_prefer_dotenv + return (get_env_value_prefer_dotenv(key) or "").strip() + + def _deepinfra_catalog_url() -> tuple[str, str]: - """Return ``(cache_key, full_url)`` for the DeepInfra catalog endpoint.""" - base = os.getenv("DEEPINFRA_BASE_URL", "").strip() or _DEEPINFRA_DEFAULT_BASE_URL - cache_key = base.rstrip("/") - return cache_key, f"{cache_key}/models?{_DEEPINFRA_MODELS_QUERY}" + """Return ``(cache_key, full_url)`` for the DeepInfra catalog endpoint. The key carries the + api-key fingerprint: the catalog is user-scoped (private fine-tunes), so two profiles with + different keys must not share an entry.""" + base = (_deepinfra_env("DEEPINFRA_BASE_URL") or _DEEPINFRA_DEFAULT_BASE_URL).rstrip("/") + from agent.credential_persistence import fingerprint_secret_value + fp = fingerprint_secret_value(_deepinfra_env("DEEPINFRA_API_KEY")) or "anon" + return f"{base}#{fp}", f"{base}/models?{_DEEPINFRA_MODELS_QUERY}" def _fetch_deepinfra_catalog( @@ -2375,7 +2404,7 @@ def _fetch_deepinfra_catalog( return None headers: dict[str, str] = {"User-Agent": _HERMES_USER_AGENT} - api_key = os.getenv("DEEPINFRA_API_KEY", "").strip() + api_key = _deepinfra_env("DEEPINFRA_API_KEY") if api_key: headers["Authorization"] = f"Bearer {api_key}" try: @@ -2435,7 +2464,7 @@ def deepinfra_model_ids(tag: str, *, force_refresh: bool = False) -> list[str]: def deepinfra_base_url(section: Optional[dict] = None) -> str: """DeepInfra base URL: config-section ``base_url`` → ``DEEPINFRA_BASE_URL`` env → default; stripped.""" candidate = section.get("base_url") if isinstance(section, dict) else None - value = candidate or os.getenv("DEEPINFRA_BASE_URL") or _DEEPINFRA_DEFAULT_BASE_URL + value = candidate or _deepinfra_env("DEEPINFRA_BASE_URL") or _DEEPINFRA_DEFAULT_BASE_URL return str(value).strip().rstrip("/") diff --git a/hermes_cli/models_profile_cache.py b/hermes_cli/models_profile_cache.py new file mode 100644 index 0000000000..7b6e47f511 --- /dev/null +++ b/hermes_cli/models_profile_cache.py @@ -0,0 +1,32 @@ +"""Per-profile view of ``hermes_cli.models``' module-level catalog slots. + +Several caches on the facade (curated OpenRouter list, reasoning-capability catalogs and their +once-per-process guards) hold values derived from ONE profile's config, ``.env`` and ``/cache`` +files. In a multiplexed gateway every turn runs under a HERMES_HOME override, so a single module slot +would hand the launch profile's value to every other profile. Under an override the slot is read and +written per home key (routed profiles start cold, never from the launch profile's warmed value); +without one the module attribute stays the slot, so single-profile behaviour and the tests that reset +``models._X = None`` are untouched. Same shape as ``tools.approval._permanent_set``. +""" + +from __future__ import annotations + +from typing import Any + +from hermes_constants import get_hermes_home_override, hermes_home_key + +_SLOTS_BY_HOME: dict[tuple[str, str], Any] = {} + + +def profile_slot_get(module: Any, attr: str, default: Any = None) -> Any: + """``module.`` for the active profile; ``default`` is a routed profile's cold value.""" + if get_hermes_home_override() is None: + return getattr(module, attr) + return _SLOTS_BY_HOME.get((hermes_home_key(), attr), default) + + +def profile_slot_set(module: Any, attr: str, value: Any) -> None: + if get_hermes_home_override() is None: + setattr(module, attr, value) + else: + _SLOTS_BY_HOME[(hermes_home_key(), attr)] = value diff --git a/hermes_cli/models_reasoning_caps.py b/hermes_cli/models_reasoning_caps.py index 459be653c5..7072de677e 100644 --- a/hermes_cli/models_reasoning_caps.py +++ b/hermes_cli/models_reasoning_caps.py @@ -14,6 +14,7 @@ malformed). from __future__ import annotations +import contextvars import json import logging import os @@ -109,7 +110,8 @@ def _warm_reasoning_caps_async(refresh) -> None: Callers own the once-per-process guard; the fetch keeps its own failure TTL.""" if os.environ.get("PYTEST_CURRENT_TEST"): return - threading.Thread(target=refresh, name="reasoning-caps-warm", daemon=True).start() + # copy_context: the Portal URL and disk mirror are the calling profile's, not the launch home's. + threading.Thread(target=contextvars.copy_context().run, args=(refresh,), name="reasoning-caps-warm", daemon=True).start() def _hydrate_reasoning_caps_from_disk(url: str, refresh) -> Optional[Caps]: @@ -170,12 +172,23 @@ class _CapsSource: disk_checked: str warm_started: str url: Callable[[], str] + # True when the URL (hence the catalog) follows the active profile's credentials/.env: under a + # routed profile the slots then live per home, and the once-per-process guards must not let the + # launch profile's disk hydrate or warm count as another profile's. + per_profile: bool = False def get(self, slot: str): - return getattr(_origin(), getattr(self, slot)) + if not self.per_profile: + return getattr(_origin(), getattr(self, slot)) + from hermes_cli.models_profile_cache import profile_slot_get + return profile_slot_get(_origin(), getattr(self, slot), False if slot in ("disk_checked", "warm_started") else None) def set(self, slot: str, value) -> None: - setattr(_origin(), getattr(self, slot), value) + if not self.per_profile: + setattr(_origin(), getattr(self, slot), value) + return + from hermes_cli.models_profile_cache import profile_slot_set + profile_slot_set(_origin(), getattr(self, slot), value) def _fetch_caps(src: _CapsSource, timeout: float = 6.0, *, force: bool = False) -> Optional[Caps]: @@ -250,7 +263,7 @@ _OPENROUTER_CAPS = _CapsSource( _NOUS_CAPS = _CapsSource( "_nous_reasoning_caps_cache", "_nous_reasoning_caps_failed_at", "_nous_caps_disk_checked", "_nous_caps_warm_started", - lambda: nous_catalog_url(), + lambda: nous_catalog_url(), per_profile=True, ) diff --git a/hermes_cli/skin_engine.py b/hermes_cli/skin_engine.py index ca4515616b..4ee2576c71 100644 --- a/hermes_cli/skin_engine.py +++ b/hermes_cli/skin_engine.py @@ -342,6 +342,23 @@ _BUILTIN_SKINS: Dict[str, Dict[str, Any]] = { _active_skin: Optional[SkinConfig] = None _active_skin_name: str = "default" +# Routed multiplex profiles: (name, skin) per home key. ``display.skin`` and ``/skins/*.yaml`` +# are per profile, and the relay display name / TUI skin payload are read under each profile's +# override — one module slot would be last-writer-wins across profiles. Unscoped keeps the module slot. +_active_skin_by_home: Dict[str, Tuple[str, SkinConfig]] = {} + + +def _routed_home_key() -> Optional[str]: + from hermes_constants import get_hermes_home_override, hermes_home_key + return None if get_hermes_home_override() is None else hermes_home_key() + + +def _profile_config() -> dict: + try: + from hermes_cli.config import load_config_readonly + return load_config_readonly() or {} + except Exception: + return {} def _skins_dir() -> Path: @@ -414,6 +431,14 @@ def load_skin(name: str) -> SkinConfig: def get_active_skin() -> SkinConfig: """Currently active skin config (cached).""" global _active_skin + home_key = _routed_home_key() + if home_key is not None: + entry = _active_skin_by_home.get(home_key) + if entry is None: + # Cold routed profile: its own ``display.skin`` (nobody ran init_skin_from_config for it). + init_skin_from_config(_profile_config()) + entry = _active_skin_by_home[home_key] + return entry[1] if _active_skin is None: _active_skin = load_skin(_active_skin_name) return _active_skin @@ -422,12 +447,21 @@ def get_active_skin() -> SkinConfig: def set_active_skin(name: str) -> SkinConfig: """Switch the active skin. Returns the new SkinConfig.""" global _active_skin, _active_skin_name + skin = load_skin(name) + home_key = _routed_home_key() + if home_key is not None: + _active_skin_by_home[home_key] = (name, skin) + return skin _active_skin_name = name - _active_skin = load_skin(name) + _active_skin = skin return _active_skin def get_active_skin_name() -> str: + home_key = _routed_home_key() + if home_key is not None: + entry = _active_skin_by_home.get(home_key) + return entry[0] if entry else "default" return _active_skin_name diff --git a/tests/hermes_cli/test_multiplex_cli_cache_scope.py b/tests/hermes_cli/test_multiplex_cli_cache_scope.py new file mode 100644 index 0000000000..dc5a358e8e --- /dev/null +++ b/tests/hermes_cli/test_multiplex_cli_cache_scope.py @@ -0,0 +1,258 @@ +"""Multiplexed gateway: hermes_cli's process-wide caches must not hand profile A's value to profile B. + +Every test builds two real profile homes (config.yaml / .env / cache files that differ), warms a +cache under ``set_hermes_home_override(A)`` and reads under B. Only the HTTP transport is canned — +its payload depends on the Authorization header or URL so a leaked entry is observable. +""" + +from __future__ import annotations + +import io +import json +import os +import threading +import time + +import httpx +import pytest + +from agent.secret_scope import build_profile_secret_scope, reset_secret_scope, set_secret_scope +from hermes_constants import get_hermes_home, reset_hermes_home_override, set_hermes_home_override + + +class _Resp(io.BytesIO): + def __enter__(self): + return self + + def __exit__(self, *a): + return False + + +def _json_resp(payload) -> _Resp: + return _Resp(json.dumps(payload).encode()) + + +class _Scoped: + """Run a block as one profile's multiplexed turn (home override + its .env secret scope).""" + + def __init__(self, home): + self.home = home + + def __enter__(self): + self._t = set_hermes_home_override(self.home) + self._s = set_secret_scope(build_profile_secret_scope(self.home)) + return self + + def __exit__(self, *a): + reset_secret_scope(self._s) + reset_hermes_home_override(self._t) + + +@pytest.fixture +def homes(tmp_path, monkeypatch): + a = tmp_path / "hermes" + b = a / "profiles" / "B" + for home in (a, b): + (home / "cache").mkdir(parents=True) + monkeypatch.setenv("HERMES_HOME", str(a)) + for var in ("DEEPINFRA_API_KEY", "DEEPINFRA_BASE_URL", "NOUS_INFERENCE_BASE_URL"): + monkeypatch.delenv(var, raising=False) + return a, b + + +def test_deepinfra_catalog_is_fetched_with_each_profiles_key(homes, monkeypatch): + a, b = homes + (a / ".env").write_text("DEEPINFRA_API_KEY=key-A\n", encoding="utf-8") + (b / ".env").write_text("DEEPINFRA_API_KEY=key-B\n", encoding="utf-8") + import hermes_cli.models as models + + monkeypatch.setattr(models, "_deepinfra_catalog_cache", {}) + monkeypatch.setattr(models, "_deepinfra_catalog_neg_cache", {}) + + def transport(req, *, timeout, **kw): + who = req.headers.get("Authorization", "").rsplit("-", 1)[-1] or "anon" + return _json_resp({"data": [{"id": f"di/model-{who}", "metadata": {"tags": ["chat"]}}]}) + + monkeypatch.setattr(models, "_urlopen_model_catalog_request", transport) + with _Scoped(a): + assert models._fetch_deepinfra_models() == ["di/model-A"] + with _Scoped(b): + assert models._fetch_deepinfra_models() == ["di/model-B"] + + +def test_copilot_context_cache_hit_requires_same_api_key(homes, monkeypatch): + import hermes_cli.models as models + + monkeypatch.setattr(models, "_copilot_context_cache", {}) + monkeypatch.setattr(models, "_copilot_context_cache_time", 0.0) + monkeypatch.setattr(models, "_github_model_catalog_cache", None) + + def transport(req, *, timeout, **kw): + limit = 111 if req.headers.get("Authorization", "").endswith("copilot-A") else 222 + return _json_resp({"data": [{"id": "gpt-x", "model_picker_enabled": True, + "supported_endpoints": ["/chat/completions"], + "capabilities": {"type": "chat", "limits": {"max_prompt_tokens": limit}}}]}) + + monkeypatch.setattr(models, "_urlopen_model_catalog_request", transport) + assert models.get_copilot_model_context("gpt-x", api_key="copilot-A") == 111 + assert models.get_copilot_model_context("gpt-x", api_key="copilot-B") == 222 + assert models.get_copilot_model_context("gpt-x", api_key="copilot-A") == 111 + + +def test_nous_reasoning_caps_follow_each_profiles_portal(homes, monkeypatch): + a, b = homes + (a / ".env").write_text("NOUS_INFERENCE_BASE_URL=https://portal-a.example/v1\n", encoding="utf-8") + (b / ".env").write_text("NOUS_INFERENCE_BASE_URL=https://portal-b.example/v1\n", encoding="utf-8") + import hermes_cli.models as models + import hermes_cli.models_reasoning_caps as caps + + for attr, value in (("_nous_reasoning_caps_cache", None), ("_nous_reasoning_caps_failed_at", None), + ("_nous_caps_disk_checked", False), ("_nous_caps_warm_started", False)): + monkeypatch.setattr(models, attr, value) + + def transport(req, *, timeout, **kw): + effort = "low" if "portal-a" in req.full_url else "high" + return _json_resp({"data": [{"id": "nous/m", "supported_parameters": ["reasoning"], + "reasoning": {"supported_efforts": [effort]}}]}) + + monkeypatch.setattr(models, "_urlopen_model_catalog_request", transport) + with _Scoped(a): + assert caps.nous_model_reasoning_capabilities("nous/m", allow_fetch=True)["supported_efforts"] == ["low"] + with _Scoped(b): + assert caps.nous_model_reasoning_capabilities("nous/m", allow_fetch=True)["supported_efforts"] == ["high"] + + +def test_swr_refresh_runs_as_the_profile_that_spawned_it(homes): + a, b = homes + import hermes_cli.models as models + + seen: dict[str, str] = {} + done = threading.Event() + + def refresh(): + seen["home"] = str(get_hermes_home()) + done.set() + return {"fp": "fp", "at": time.time(), "models": ["m"]} + + with _Scoped(b): + models._spawn_swr_refresh("custom:https://gw.example/v1#fp", refresh) + assert done.wait(5) + assert seen["home"] == str(b) + deadline = time.monotonic() + 5 + while not (b / "provider_models_cache.json").exists() and time.monotonic() < deadline: + time.sleep(0.05) + assert (b / "provider_models_cache.json").exists() + assert not (a / "provider_models_cache.json").exists() + + +def _write_manifest(home, model_id: str, mtime: float) -> None: + path = home / "cache" / "model_catalog.json" + path.write_text(json.dumps({"version": 1, "providers": {"openrouter": {"models": [ + {"id": model_id, "description": "x", "default": True}]}}}), encoding="utf-8") + os.utime(path, (mtime, mtime)) + + +def test_model_catalog_in_process_copy_is_bound_to_its_cache_file(homes, monkeypatch): + a, b = homes + import hermes_cli.model_catalog as mc + + for home in (a, b): + (home / "config.yaml").write_text("model_catalog:\n ttl_minutes: 600\n", encoding="utf-8") + same_mtime = time.time() - 5 # identical mtimes: only the path can tell the two files apart + _write_manifest(a, "vendor/a-model", same_mtime) + _write_manifest(b, "vendor/b-model", same_mtime) + mc.reset_cache() + with _Scoped(a): + assert [m["id"] for m in mc.get_catalog()["providers"]["openrouter"]["models"]] == ["vendor/a-model"] + with _Scoped(b): + assert [m["id"] for m in mc.get_catalog()["providers"]["openrouter"]["models"]] == ["vendor/b-model"] + assert mc.get_default_model_from_cache("openrouter") == "vendor/b-model" + + +def test_openrouter_curated_list_is_per_profile(homes, monkeypatch): + a, b = homes + import hermes_cli.models as models + + for home in (a, b): + (home / "config.yaml").write_text("model_catalog:\n ttl_minutes: 600\n", encoding="utf-8") + now = time.time() + for home, mid in ((a, "vendor/a-model"), (b, "vendor/b-model")): + (home / "cache" / "openrouter_curated_catalog.json").write_text( + json.dumps({"fetched_at": now, "curated": [[mid, "free"]]}), encoding="utf-8") + monkeypatch.setattr(models, "_openrouter_catalog_cache", None) + with _Scoped(a): + assert [m for m, _ in models.fetch_openrouter_models()] == ["vendor/a-model"] + with _Scoped(b): + assert [m for m, _ in models.fetch_openrouter_models()] == ["vendor/b-model"] + + +def test_banner_skills_are_the_routed_profiles(homes): + a, b = homes + import hermes_cli.banner as banner + + for home, tag in ((a, "a"), (b, "b")): + skill = home / "skills" / f"skill_{tag}" + skill.mkdir(parents=True) + (skill / "SKILL.md").write_text(f"---\nname: skill_{tag}\ndescription: {tag}\n---\nbody\n", encoding="utf-8") + banner._available_skills_cache = None + try: + with _Scoped(a): + assert sorted(sum(banner.get_available_skills().values(), [])) == ["skill_a"] + with _Scoped(b): + assert sorted(sum(banner.get_available_skills().values(), [])) == ["skill_b"] + finally: + banner._available_skills_cache = None + + +def test_failed_guest_mint_only_suppresses_that_profile(homes, monkeypatch, tmp_path): + a, b = homes + monkeypatch.setenv("HERMES_GUEST_ONBOARDING", "1") + monkeypatch.setenv("HERMES_SHARED_AUTH_DIR", str(tmp_path / "shared")) + import hermes_cli.anon_auth as anon + import hermes_cli.auth_nous as auth_nous + + monkeypatch.setattr(anon, "_mint_failed", False) + monkeypatch.setattr(anon, "_mint_failed_homes", set(), raising=False) + status = {"code": 429} + attempts: list[str] = [] + + def client(timeout_seconds, verify): + def handler(request): + attempts.append(str(get_hermes_home())) + if status["code"] == 429: + return httpx.Response(429, json={"error": "rate"}) + return httpx.Response(201, json={"token": "anon_b", "user_id": "u", "org_id": "o"}) + return httpx.Client(transport=httpx.MockTransport(handler)) + + monkeypatch.setattr(auth_nous, "_nous_http_client", client) + with _Scoped(a), pytest.raises(Exception): + anon.ensure_portal_identity(explicit=True, timeout_seconds=1) + assert attempts == [str(a)] + status["code"] = 201 + with _Scoped(b): + assert anon.ensure_portal_identity(explicit=True, timeout_seconds=1) is not None + assert attempts[-1] == str(b) + + +def test_active_skin_is_per_profile_and_leaves_launch_slot_alone(homes): + a, b = homes + from hermes_cli import skin_engine + + (a / "config.yaml").write_text("display:\n skin: ares\n", encoding="utf-8") + (b / "config.yaml").write_text("display:\n skin: mono\n", encoding="utf-8") + skin_engine._active_skin = None + skin_engine._active_skin_name = "default" + getattr(skin_engine, "_active_skin_by_home", {}).clear() + try: + with _Scoped(a): + skin_engine.init_skin_from_config({"display": {"skin": "ares"}}) + assert skin_engine.get_active_skin().name == "ares" + with _Scoped(b): + assert skin_engine.get_active_skin().name == "mono" # B's own display.skin, never A's + with _Scoped(a): + assert skin_engine.get_active_skin().name == "ares" + assert skin_engine.get_active_skin_name() == "default" # routed turns never touch the launch slot + finally: + skin_engine._active_skin = None + skin_engine._active_skin_name = "default" + getattr(skin_engine, "_active_skin_by_home", {}).clear()