949 lines
38 KiB
Python
949 lines
38 KiB
Python
"""Provider/model inventory context — shared substrate for the dashboard ``/api/model/options``, the
|
|
TUI ``model.options``/``model.save_key`` JSON-RPC handlers, and the interactive picker.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from dataclasses import dataclass, replace
|
|
from typing import Any, Optional
|
|
|
|
|
|
# ─── Public types ───────────────────────────────────────────────────────
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ConfigContext:
|
|
"""Snapshot of the model + provider config every inventory caller needs. Built once via
|
|
``load_picker_context()``; the TUI overlays live agent state via ``with_overrides()`` before
|
|
passing through.
|
|
"""
|
|
|
|
current_provider: str
|
|
current_model: str
|
|
current_base_url: str
|
|
user_providers: dict
|
|
custom_providers: list
|
|
excluded_providers: list = None
|
|
|
|
def with_overrides(
|
|
self,
|
|
*,
|
|
current_provider: Optional[str] = None,
|
|
current_model: Optional[str] = None,
|
|
current_base_url: Optional[str] = None,
|
|
) -> "ConfigContext":
|
|
"""Return a copy with truthy overrides applied.
|
|
|
|
Truthy-only because the TUI reads agent attributes that may be empty strings before an agent
|
|
is spawned — empties must NOT clobber the disk-config values.
|
|
"""
|
|
kw = {
|
|
k: v
|
|
for k, v in (
|
|
("current_provider", current_provider),
|
|
("current_model", current_model),
|
|
("current_base_url", current_base_url),
|
|
)
|
|
if v
|
|
}
|
|
return replace(self, **kw) if kw else self
|
|
|
|
|
|
def load_picker_context() -> ConfigContext:
|
|
"""Load the disk-config snapshot every consumer needs."""
|
|
from hermes_cli.config import (
|
|
coerce_provider_id,
|
|
get_compatible_custom_providers,
|
|
load_config,
|
|
stringify_provider_map,
|
|
)
|
|
|
|
cfg = load_config()
|
|
model_cfg = cfg.get("model", {})
|
|
if isinstance(model_cfg, dict):
|
|
# PyYAML parses unquoted scalars as int (`provider: 2070`). Keep these
|
|
# as strings so picker/options paths never call `.strip()` on an int.
|
|
current_model = str(model_cfg.get("default", model_cfg.get("name", "")) or "")
|
|
current_provider = coerce_provider_id(model_cfg.get("provider", ""))
|
|
current_base_url = str(model_cfg.get("base_url", "") or "")
|
|
else:
|
|
# config.model can be a bare string in older configs.
|
|
current_model = str(model_cfg) if model_cfg else ""
|
|
current_provider = ""
|
|
current_base_url = ""
|
|
excluded = cfg.get("model_catalog", {}).get("excluded_providers") or []
|
|
return ConfigContext(
|
|
current_provider=current_provider,
|
|
current_model=current_model,
|
|
current_base_url=current_base_url,
|
|
user_providers=stringify_provider_map(cfg.get("providers")),
|
|
custom_providers=get_compatible_custom_providers(cfg),
|
|
excluded_providers=excluded if isinstance(excluded, list) else [],
|
|
)
|
|
|
|
|
|
def _slug(row: dict) -> str:
|
|
return str(row.get("slug") or "").strip().lower()
|
|
|
|
|
|
def _without_slug(rows: list[dict], slug: str) -> list[dict]:
|
|
return [r for r in rows if _slug(r) != slug]
|
|
|
|
|
|
# ─── Public: payload builder ────────────────────────────────────────────
|
|
|
|
|
|
def build_models_payload(
|
|
ctx: ConfigContext,
|
|
*,
|
|
explicit_only: bool = False,
|
|
include_unconfigured: bool = False,
|
|
picker_hints: bool = False,
|
|
canonical_order: bool = False,
|
|
pricing: bool = False,
|
|
capabilities: bool = False,
|
|
featured: bool = False,
|
|
force_fresh_nous_tier: bool = False,
|
|
refresh: bool = False,
|
|
probe_custom_providers: bool = True,
|
|
probe_current_custom_provider: bool = False,
|
|
for_picker: bool = False,
|
|
max_models: int | None = None,
|
|
) -> dict:
|
|
"""Build the ``{providers, model, provider}`` shape every consumer needs from a single substrate call.
|
|
|
|
Flags: - ``explicit_only``: keep only providers the user explicitly configured (current
|
|
provider, providers from config, or providers backed by provider-specific env vars). This hides
|
|
ambient / auto-seeded credentials from desktop chat pickers.
|
|
"""
|
|
from hermes_cli.model_switch import list_authenticated_providers
|
|
|
|
rows = list_authenticated_providers(
|
|
current_provider=ctx.current_provider,
|
|
current_base_url=ctx.current_base_url,
|
|
current_model=ctx.current_model,
|
|
user_providers=ctx.user_providers,
|
|
custom_providers=ctx.custom_providers,
|
|
force_fresh_nous_tier=force_fresh_nous_tier,
|
|
max_models=max_models,
|
|
refresh=refresh,
|
|
probe_custom_providers=probe_custom_providers,
|
|
probe_current_custom_provider=probe_current_custom_provider,
|
|
for_picker=for_picker,
|
|
excluded_providers=ctx.excluded_providers or [],
|
|
)
|
|
|
|
# Managed local runtime: staged GGUFs are selectable like any provider's
|
|
# models. list_authenticated_providers can't know about them (no
|
|
# credential, no custom_providers entry — the credential is
|
|
# reachability), so inject the row here where every picker surface
|
|
# inherits it. Present whenever models are staged; picking one routes
|
|
# through the llamacpp alias -> managed/detected server resolution.
|
|
local_row = _local_runtime_row(ctx)
|
|
if local_row is not None:
|
|
rows = _without_slug(rows, "llamacpp") + [local_row]
|
|
# A live session on the managed server reports provider "custom"
|
|
# (the resolution seam's generic label for a raw base_url), which
|
|
# would otherwise materialize a duplicate "Custom endpoint" row
|
|
# carrying the same staged models and stealing the checkmark. The
|
|
# Local row owns the managed server's identity — drop custom rows
|
|
# that point at the managed endpoint.
|
|
if local_row.get("is_current"):
|
|
staged = set(local_row["models"])
|
|
|
|
def _is_managed_custom(row: dict) -> bool:
|
|
models = {str(m) for m in (row.get("models") or [])}
|
|
return _slug(row) == "custom" and bool(models) and models <= staged
|
|
|
|
rows = [r for r in rows if not _is_managed_custom(r)]
|
|
|
|
moa_row = _moa_provider_row(ctx.current_provider)
|
|
if moa_row is not None:
|
|
rows = [moa_row] + _without_slug(rows, "moa")
|
|
|
|
if explicit_only:
|
|
rows = _filter_explicit_provider_rows(rows, ctx)
|
|
# Desktop chat pickers request the explicit subset without the full
|
|
# unconfigured provider universe. If the configured current provider
|
|
# has lost its credential, list_authenticated_providers() omits it;
|
|
# keep that one row visible so the UI can show the saved selection and
|
|
# a re-auth affordance instead of appearing to jump to another provider.
|
|
# Exception: a "custom" current whose endpoint is the managed local
|
|
# server is already represented (with the checkmark) by the Local row
|
|
# — the skeleton would resurrect the duplicate the dedup above removed.
|
|
_local_owns_current = bool(local_row and local_row.get("is_current")
|
|
and (ctx.current_provider or "").lower() == "custom")
|
|
if not _local_owns_current:
|
|
rows = list(rows) + _append_unconfigured_rows(rows, ctx, current_only=True)
|
|
|
|
# --- Deduplicate: remove models from aggregators that overlap with
|
|
# user-defined providers. When a local proxy (e.g. litellm-proxy)
|
|
# serves a model whose name also appears in an aggregator's curated
|
|
# catalog, the picker would show the model under both providers.
|
|
# Selecting it from the aggregator row sets model.provider to the
|
|
# aggregator (e.g. openrouter) instead of the user's proxy — silently
|
|
# breaking the call. Filtering at the payload level keeps the
|
|
# aggregator rows honest: they only show models the user can't get
|
|
# from a more-specific provider. (#45954)
|
|
_strip_aggregator_overlaps(rows)
|
|
|
|
if include_unconfigured:
|
|
rows = list(rows) + _without_slug(_append_unconfigured_rows(rows, ctx), "moa")
|
|
if picker_hints:
|
|
_apply_picker_hints(rows)
|
|
if canonical_order:
|
|
rows = _reorder_canonical(rows)
|
|
if pricing:
|
|
_apply_pricing(rows, force_fresh_nous_tier=force_fresh_nous_tier)
|
|
if capabilities:
|
|
_apply_capabilities(rows)
|
|
if featured:
|
|
_apply_featured(rows)
|
|
_apply_custom_aliases(rows)
|
|
|
|
return {
|
|
"providers": rows,
|
|
"model": ctx.current_model,
|
|
"provider": ctx.current_provider,
|
|
}
|
|
|
|
|
|
def _strip_aggregator_overlaps(rows: list[dict]) -> None:
|
|
"""Drop models from TRUE routing aggregators (OpenRouter, custom:* proxies) that a user-defined
|
|
provider also serves, so the picker never lists them under both (#45954).
|
|
|
|
A user's own configured provider is never an "aggregator duplicate" of itself: user_models is
|
|
built from these very rows, and is_routing_aggregator() reports True for every custom:* slug —
|
|
without that guard the dedup would empty a user-defined custom provider's row. Flat-namespace
|
|
resellers (opencode-go / opencode-zen) serve every listed model first-party and must keep models
|
|
that a user's proxy happens to share a name with (#47077).
|
|
"""
|
|
try:
|
|
from hermes_cli.providers import is_routing_aggregator
|
|
except Exception:
|
|
return
|
|
|
|
user_models: set[str] = set()
|
|
for row in rows:
|
|
if row.get("is_user_defined"):
|
|
user_models.update(m.lower() for m in (row.get("models") or []))
|
|
if not user_models:
|
|
return
|
|
for row in rows:
|
|
if row.get("is_user_defined") or not is_routing_aggregator(row.get("slug", "")):
|
|
continue
|
|
original = row.get("models") or []
|
|
filtered = [m for m in original if m.lower() not in user_models]
|
|
if len(filtered) < len(original):
|
|
row["models"] = filtered
|
|
row["total_models"] = len(filtered)
|
|
|
|
|
|
def build_model_options_payload(
|
|
ctx: ConfigContext,
|
|
*,
|
|
explicit_only: bool = False,
|
|
include_unconfigured: bool = False,
|
|
refresh: bool = False,
|
|
) -> dict:
|
|
"""Build the shared API-server/dashboard/TUI model-options payload.
|
|
|
|
This wraps ``build_models_payload`` with the stable picker shape and the safe custom-provider
|
|
probe policy used for normal GUI/TUI opens:
|
|
|
|
- normal open: probe only the current custom provider so offline saved endpoints do not block
|
|
the picker - explicit refresh: probe every custom provider while busting the model cache so live
|
|
catalogs repopulate fully
|
|
"""
|
|
refresh = bool(refresh)
|
|
return build_models_payload(
|
|
ctx,
|
|
explicit_only=bool(explicit_only),
|
|
include_unconfigured=bool(include_unconfigured),
|
|
picker_hints=True,
|
|
canonical_order=True,
|
|
pricing=True,
|
|
capabilities=True,
|
|
featured=True,
|
|
refresh=refresh,
|
|
probe_custom_providers=refresh,
|
|
probe_current_custom_provider=not refresh,
|
|
)
|
|
|
|
|
|
# ─── Public: auxiliary-task pickers ─────────────────────────────────────
|
|
|
|
|
|
def build_aux_picker_rows(
|
|
*,
|
|
current_provider: str = "",
|
|
current_model: str = "",
|
|
current_base_url: str = "",
|
|
max_models: int | None = None,
|
|
) -> list[dict]:
|
|
"""Provider rows for any auxiliary-task picker (vision, compression, …).
|
|
|
|
- user-defined ``providers:`` and saved ``custom_providers:`` entries -
|
|
``model_catalog.excluded_providers`` honoured, matching ``/model`` - exhausted-credential-pool
|
|
providers stay visible (``for_picker``) - the active custom endpoint is probed, offline saved
|
|
ones are not, so the picker never blocks on a dead local server
|
|
|
|
The virtual ``moa`` row is excluded: auxiliary tasks must not run the MoA reference fan-out, and
|
|
``auxiliary_client`` unwraps a ``moa`` provider to its aggregator slot anyway (see
|
|
``_resolve_auto``), so offering it here would be a choice silently rewritten behind the user's
|
|
back.
|
|
"""
|
|
ctx = load_picker_context().with_overrides(
|
|
current_provider=current_provider,
|
|
current_model=current_model,
|
|
current_base_url=current_base_url,
|
|
)
|
|
rows = build_models_payload(
|
|
ctx,
|
|
for_picker=True,
|
|
probe_custom_providers=False,
|
|
probe_current_custom_provider=True,
|
|
max_models=max_models,
|
|
)["providers"]
|
|
return _without_slug(rows, "moa")
|
|
|
|
|
|
def format_aux_picker_entries(
|
|
rows: list[dict],
|
|
*,
|
|
current_provider: str = "",
|
|
current_base_url: str = "",
|
|
) -> list[tuple[str, str, list[str]]]:
|
|
"""Render aux-picker rows as ``(slug, label, models)`` menu entries.
|
|
|
|
A custom endpoint set via a raw ``base_url`` is "current" only through that URL — never through
|
|
a provider slug — so when ``current_base_url`` is set no provider row is marked, matching the
|
|
pre-existing behaviour of both call sites.
|
|
"""
|
|
entries: list[tuple[str, str, list[str]]] = []
|
|
current_slug = str(current_provider or "").strip().lower()
|
|
has_base_url = bool(str(current_base_url or "").strip())
|
|
for row in rows:
|
|
slug = str(row.get("slug") or "")
|
|
name = row.get("name") or slug
|
|
total = row.get("total_models") or len(row.get("models") or [])
|
|
model_hint = f" — {total} models" if total else ""
|
|
marker = (
|
|
" ← current"
|
|
if slug.lower() == current_slug and current_slug and not has_base_url
|
|
else ""
|
|
)
|
|
entries.append((slug, f"{name}{model_hint}{marker}", list(row.get("models") or [])))
|
|
return entries
|
|
|
|
|
|
def _reasoning_catalog_reader(slug: str):
|
|
"""Per-model reasoning-capability reader for aggregators that publish one.
|
|
|
|
Cache-only — building the picker payload must never block on HTTP. A cold cache warms in the
|
|
background so the next open is accurate; until then the model reports no restriction and the UI
|
|
offers the full scale.
|
|
"""
|
|
try:
|
|
from hermes_cli.models import (
|
|
nous_model_reasoning_capabilities,
|
|
openrouter_model_reasoning_capabilities,
|
|
warm_nous_reasoning_caps_async,
|
|
warm_openrouter_reasoning_caps_async,
|
|
)
|
|
except Exception:
|
|
return None
|
|
|
|
readers = {
|
|
"nous": (warm_nous_reasoning_caps_async, nous_model_reasoning_capabilities),
|
|
"openrouter": (warm_openrouter_reasoning_caps_async, openrouter_model_reasoning_capabilities),
|
|
}
|
|
if slug not in readers:
|
|
return None
|
|
warm, read = readers[slug]
|
|
warm()
|
|
return read
|
|
|
|
|
|
def _apply_capabilities(rows: list[dict]) -> None:
|
|
"""Attach a ``{model: {fast, reasoning, ...}}`` map to each provider row.
|
|
|
|
``fast`` mirrors the runtime gate. ``reasoning`` defaults True when the catalog is silent:
|
|
the dial is a no-op on models that ignore it, and hiding it from a capable model is worse.
|
|
A serving aggregator's per-model detail overrides models.dev (adds ``can_disable_reasoning``).
|
|
``supported_efforts`` is deliberately NOT forwarded -- it under-reports levels that work.
|
|
"""
|
|
from hermes_cli.models import model_supports_fast_mode
|
|
|
|
try:
|
|
from agent.models_dev import get_model_capabilities
|
|
except Exception:
|
|
get_model_capabilities = None # type: ignore[assignment]
|
|
|
|
for row in rows:
|
|
slug = row.get("slug") or ""
|
|
caps: dict[str, dict[str, Any]] = {}
|
|
read_reasoning_catalog = _reasoning_catalog_reader(slug.lower())
|
|
|
|
for model in row.get("models") or []:
|
|
reasoning = True
|
|
if get_model_capabilities is not None and slug:
|
|
try:
|
|
meta = get_model_capabilities(slug, model)
|
|
if meta is not None:
|
|
reasoning = bool(meta.supports_reasoning)
|
|
except Exception:
|
|
reasoning = True
|
|
|
|
entry: dict[str, Any] = {
|
|
"fast": bool(model_supports_fast_mode(model)),
|
|
"reasoning": reasoning,
|
|
}
|
|
|
|
if reasoning and read_reasoning_catalog is not None:
|
|
try:
|
|
detail = read_reasoning_catalog(model)
|
|
except Exception:
|
|
detail = None
|
|
if detail and not detail.get("supports_reasoning"):
|
|
# For a route it serves, the aggregator's own catalog beats
|
|
# models.dev: no reasoning parameter means no reasoning
|
|
# controls, so there is no disable to describe either.
|
|
entry["reasoning"] = False
|
|
elif detail:
|
|
entry["can_disable_reasoning"] = not detail.get("mandatory")
|
|
|
|
caps[model] = entry
|
|
|
|
row["capabilities"] = caps
|
|
|
|
|
|
# How many models per lab the picker features by default. Aggregator rows keep
|
|
# the newest N of each lab (by models.dev release_date) and hide the older tail
|
|
# behind search / show-all. 5 keeps a lab's current headliners without letting a
|
|
# prolific vendor (OpenAI's gpt-5.6-* family) flood the default view.
|
|
_FEATURED_PER_LAB = 5
|
|
|
|
|
|
def _apply_featured(rows: list[dict]) -> None:
|
|
"""Attach a ``featured_models`` shortlist to each aggregator provider row.
|
|
|
|
Aggregators serve many labs, so a flat top-N would drop whole labs; instead surface the
|
|
newest ``_FEATURED_PER_LAB`` per vendor, ranked by models.dev ``release_date`` within the
|
|
row's own models (never vs. today, so the choice is stable). Ties fall back to the curated
|
|
flagship-first order. Non-aggregators get an empty list and keep top-N behaviour.
|
|
"""
|
|
try:
|
|
from agent.models_dev import get_model_info
|
|
except Exception:
|
|
get_model_info = None # type: ignore[assignment]
|
|
|
|
for row in rows:
|
|
slug = str(row.get("slug") or "").strip().lower()
|
|
models = row.get("models") or []
|
|
|
|
# Group models by lab; only multi-lab aggregators get a shortlist.
|
|
by_lab: dict[str, list[tuple[int, str, str]]] = {}
|
|
for pos, model in enumerate(models):
|
|
lab = model.split("/", 1)[0] if "/" in model else ""
|
|
if not lab:
|
|
# No vendor prefix → single-namespace provider, not an
|
|
# aggregator. Bail on the whole row (see below).
|
|
by_lab = {}
|
|
break
|
|
date = ""
|
|
if get_model_info is not None:
|
|
info = get_model_info(slug, model) or get_model_info("openrouter", model)
|
|
date = getattr(info, "release_date", "") if info else ""
|
|
by_lab.setdefault(lab, []).append((pos, date, model))
|
|
|
|
# A shortlist only makes sense when the row spans several labs.
|
|
if len(by_lab) < 2:
|
|
row["featured_models"] = []
|
|
continue
|
|
|
|
featured: list[str] = []
|
|
for entries in by_lab.values():
|
|
# Newest release_date first; earlier list position breaks ties and
|
|
# is the sole key when a lab has no dated models (all ""). Keep the
|
|
# newest _FEATURED_PER_LAB of each lab.
|
|
ranked = sorted(entries, key=lambda e: (e[1], -e[0]), reverse=True)
|
|
featured.extend(model for _pos, _date, model in ranked[:_FEATURED_PER_LAB])
|
|
# Preserve the row's model order for stable rendering.
|
|
order = {m: i for i, m in enumerate(models)}
|
|
row["featured_models"] = sorted(featured, key=lambda m: order[m])
|
|
|
|
|
|
def _apply_custom_aliases(rows: list[dict]) -> None:
|
|
"""Attach the accepted identity set to each user-defined provider row.
|
|
|
|
A session's ``model.options`` reports the canonical ``custom:<key>`` identity (via
|
|
``canonical_custom_identity``), while catalog rows carry the bare config key as ``slug``. GUI
|
|
pickers compare the two to decide which row is active; exact equality never matches for custom
|
|
providers (#87035).
|
|
"""
|
|
from hermes_cli.providers import custom_provider_aliases
|
|
|
|
for row in rows:
|
|
if not row.get("is_user_defined"):
|
|
continue
|
|
try:
|
|
row["aliases"] = sorted(
|
|
custom_provider_aliases(
|
|
str(row.get("name", "")), str(row.get("slug", ""))
|
|
)
|
|
)
|
|
except Exception:
|
|
continue
|
|
|
|
|
|
# ─── Internal: row post-processing ──────────────────────────────────────
|
|
|
|
|
|
def _provider_auth_hint(slug: str) -> tuple[str, str]:
|
|
"""``(auth_type, key_env)`` for a canonical provider (``("api_key", "")`` when unregistered)."""
|
|
from hermes_cli.auth import PROVIDER_REGISTRY
|
|
|
|
cfg = PROVIDER_REGISTRY.get(slug)
|
|
auth_type = cfg.auth_type if cfg else "api_key"
|
|
key_env = cfg.api_key_env_vars[0] if (cfg and cfg.api_key_env_vars) else ""
|
|
return auth_type, key_env
|
|
|
|
|
|
def _canonical_row(entry, cur: str, **extra: Any) -> dict:
|
|
from hermes_cli.models import _PROVIDER_LABELS
|
|
|
|
return {
|
|
"slug": entry.slug,
|
|
"name": _PROVIDER_LABELS.get(entry.slug, entry.label),
|
|
"is_current": entry.slug.lower() == cur,
|
|
"is_user_defined": False,
|
|
**extra,
|
|
}
|
|
|
|
|
|
def _append_unconfigured_rows(
|
|
rows: list[dict],
|
|
ctx: ConfigContext,
|
|
*,
|
|
current_only: bool = False,
|
|
) -> list[dict]:
|
|
"""Build fallback rows for canonical providers missing from ``rows``.
|
|
|
|
Most missing canonical providers become empty setup skeletons. The one exception is the
|
|
*current* configured provider: if config.yaml still points at it but credentials are presently
|
|
unavailable, keep a visible row carrying the saved model so GUI pickers don't silently snap to
|
|
some other provider.
|
|
"""
|
|
from hermes_cli.models import CANONICAL_PROVIDERS
|
|
|
|
seen = {r["slug"].lower() for r in rows}
|
|
cur = (ctx.current_provider or "").lower()
|
|
cur_model = str(ctx.current_model or "").strip()
|
|
extras: list[dict] = []
|
|
for entry in CANONICAL_PROVIDERS:
|
|
if entry.slug.lower() in seen:
|
|
continue
|
|
if current_only and entry.slug.lower() != cur:
|
|
continue
|
|
if entry.slug.lower() == cur:
|
|
auth_type, key_env = _provider_auth_hint(entry.slug)
|
|
warning = (
|
|
f"Configured provider missing usable credentials; paste {key_env} to reactivate. "
|
|
"Showing the saved model only."
|
|
if auth_type == "api_key" and key_env
|
|
else "Configured provider is not authenticated; run `hermes model` to reactivate. "
|
|
"Showing the saved model only."
|
|
)
|
|
extras.append(_canonical_row(
|
|
entry, cur,
|
|
models=[cur_model] if cur_model else [],
|
|
total_models=1 if cur_model else 0,
|
|
source="configured-current",
|
|
authenticated=False,
|
|
auth_type=auth_type,
|
|
key_env=key_env,
|
|
warning=warning,
|
|
))
|
|
continue
|
|
extras.append(_canonical_row(entry, cur, models=[], total_models=0, source="canonical"))
|
|
return extras
|
|
|
|
|
|
def _anthropic_oauth_credentials_present() -> bool:
|
|
"""True when the user explicitly authenticated Anthropic via OAuth.
|
|
|
|
Two deliberate flows leave no trace in active_provider / model.provider / API-key env vars:
|
|
Hermes' own Anthropic device flow (token in auth.json) and a Claude Code login
|
|
(~/.claude/.credentials.json).
|
|
"""
|
|
try:
|
|
from agent.anthropic_adapter import (
|
|
read_claude_code_credentials,
|
|
read_hermes_oauth_credentials,
|
|
)
|
|
|
|
if any(
|
|
(read() or {}).get("accessToken")
|
|
for read in (read_hermes_oauth_credentials, read_claude_code_credentials)
|
|
):
|
|
return True
|
|
except Exception:
|
|
return False
|
|
# Pool-only OAuth entries (auth.json credential_pool.anthropic) are the
|
|
# canonical location for wired tokens and equally deliberate — the
|
|
# discovery side accepts them via pool.has_credentials(), so the filter
|
|
# must too or those rows are built and then silently dropped. Read-only
|
|
# dict access (no load_pool) so a picker open never mutates auth.json.
|
|
try:
|
|
from agent.credential_pool import AUTH_TYPE_OAUTH
|
|
from hermes_cli.auth import read_credential_pool
|
|
|
|
for entry in read_credential_pool("anthropic"):
|
|
if (
|
|
isinstance(entry, dict)
|
|
and entry.get("auth_type") == AUTH_TYPE_OAUTH
|
|
and str(entry.get("access_token") or "").strip()
|
|
):
|
|
return True
|
|
except Exception:
|
|
pass
|
|
return False
|
|
|
|
|
|
def _filter_explicit_provider_rows(rows: list[dict], ctx: ConfigContext) -> list[dict]:
|
|
"""Keep only rows backed by explicit user configuration.
|
|
|
|
``list_authenticated_providers`` also discovers ambient credentials (e.g. GitHub CLI ->
|
|
Copilot); Desktop chat pickers want only what the user configured for Hermes.
|
|
"""
|
|
from hermes_cli.auth import is_provider_explicitly_configured
|
|
|
|
current_slug = str(ctx.current_provider or "").strip().lower()
|
|
|
|
def _is_explicit(row: dict, slug: str) -> bool:
|
|
if (
|
|
row.get("is_user_defined")
|
|
or (current_slug and slug == current_slug)
|
|
# Managed local models are explicit configuration by existence:
|
|
# the user downloaded gigabytes into the machine-scoped models
|
|
# dir. There is deliberately no config credential to find
|
|
# (credential is reachability), so without this clause the row
|
|
# only survives on the profile where Use was last clicked —
|
|
# every other profile loses local models from its picker.
|
|
or row.get("source") == "local-runtime"
|
|
):
|
|
return True
|
|
if slug == "moa":
|
|
# MoA is a virtual routing mode, not an independently configured
|
|
# provider. Hide it from explicit-only pickers unless it is the
|
|
# current provider (handled above) or the user explicitly wrote an
|
|
# enabled MoA preset into config.yaml. Use raw config so the
|
|
# DEFAULT_CONFIG preset does not make every desktop picker show MoA.
|
|
return _raw_config_has_enabled_moa_preset()
|
|
return (
|
|
# Keyless providers (opencode-free) require no configuration at
|
|
# all — there is nothing to "explicitly configure", and hiding
|
|
# them would defeat their purpose (zero-setup discoverability).
|
|
_provider_is_keyless(slug)
|
|
# Anthropic OAuth logins (Hermes device flow / Claude Code) are
|
|
# deliberate sign-ins that leave no trace in active_provider,
|
|
# model.provider, or API-key env vars. The strict gate below
|
|
# would drop the row even though list_authenticated_providers
|
|
# just accepted those same credentials when building it.
|
|
or (slug == "anthropic" and _anthropic_oauth_credentials_present())
|
|
# External-process providers (copilot-acp) authenticate through their
|
|
# own CLI (`copilot login`) — same class as Anthropic OAuth above:
|
|
# verified CLI credentials are a deliberate sign-in with no trace in
|
|
# active_provider/model.provider/env, so keep the row the
|
|
# picker-discovery side just accepted.
|
|
or _external_process_signed_in(slug)
|
|
or is_provider_explicitly_configured(slug)
|
|
)
|
|
|
|
kept: list[dict] = []
|
|
for row in rows:
|
|
slug = str(row.get("slug", "")).strip().lower()
|
|
if slug and _is_explicit(row, slug):
|
|
kept.append(row)
|
|
return kept
|
|
|
|
|
|
def _external_process_signed_in(slug: str) -> bool:
|
|
"""True when an external-process provider has verified CLI credentials."""
|
|
try:
|
|
from hermes_cli.auth import (
|
|
PROVIDER_REGISTRY,
|
|
get_external_process_provider_status,
|
|
)
|
|
pconfig = PROVIDER_REGISTRY.get(slug)
|
|
if not pconfig or pconfig.auth_type != "external_process":
|
|
return False
|
|
return bool(get_external_process_provider_status(slug).get("auth_verified"))
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
def _provider_is_keyless(slug: str) -> bool:
|
|
"""True when the provider's Hermes overlay declares it keyless."""
|
|
try:
|
|
from hermes_cli.providers import HERMES_OVERLAYS
|
|
overlay = HERMES_OVERLAYS.get(slug)
|
|
return bool(overlay is not None and getattr(overlay, "keyless", False))
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
def _raw_config_has_enabled_moa_preset() -> bool:
|
|
"""Return True when the user's raw config explicitly enables MoA.
|
|
|
|
``load_config()`` includes ``DEFAULT_CONFIG["moa"].presets.default`` for everyone. Explicit-only
|
|
model pickers must not treat that default as a user choice, but they should keep MoA visible
|
|
once the user has saved at least one enabled preset (or an older flat MoA config) in their own
|
|
config.yaml.
|
|
"""
|
|
try:
|
|
from hermes_cli.config import read_raw_config
|
|
|
|
raw = read_raw_config()
|
|
except Exception:
|
|
return False
|
|
|
|
if not isinstance(raw, dict):
|
|
return False
|
|
moa = raw.get("moa")
|
|
if not isinstance(moa, dict):
|
|
return False
|
|
|
|
presets = moa.get("presets")
|
|
if isinstance(presets, dict):
|
|
return any(
|
|
not isinstance(preset, dict) or preset.get("enabled", True)
|
|
for name, preset in presets.items()
|
|
if str(name or "").strip()
|
|
)
|
|
|
|
legacy_keys = {
|
|
"reference_models",
|
|
"aggregator",
|
|
"reference_temperature",
|
|
"aggregator_temperature",
|
|
"max_tokens",
|
|
"reference_max_tokens",
|
|
"fanout",
|
|
}
|
|
return any(key in moa for key in legacy_keys) and bool(moa.get("enabled", True))
|
|
|
|
|
|
def _apply_picker_hints(rows: list[dict]) -> None:
|
|
"""Add ``authenticated``/``auth_type``/``key_env``/``warning`` per row."""
|
|
for row in rows:
|
|
if "authenticated" in row:
|
|
continue
|
|
# Distinguish authenticated rows (returned by
|
|
# list_authenticated_providers) from skeleton rows (from
|
|
# _append_unconfigured_rows). The skeleton rows have empty
|
|
# `models` AND source="canonical"; authenticated rows have
|
|
# populated `models` OR a non-canonical source.
|
|
is_skeleton = row.get("source") == "canonical" and not row.get("models")
|
|
row["authenticated"] = not is_skeleton
|
|
if not is_skeleton or row.get("is_user_defined"):
|
|
continue
|
|
auth_type, key_env = _provider_auth_hint(row["slug"])
|
|
row["auth_type"] = auth_type
|
|
row["key_env"] = key_env
|
|
row["warning"] = (
|
|
f"paste {key_env} to activate"
|
|
if auth_type == "api_key" and key_env
|
|
else f"run `hermes model` to configure ({auth_type})"
|
|
)
|
|
|
|
|
|
def _reorder_canonical(rows: list[dict]) -> list[dict]:
|
|
"""Canonical slugs in ``CANONICAL_PROVIDERS`` declaration order; truly-custom rows last.
|
|
|
|
Keys on slug membership, NOT ``is_user_defined``: rows from the ``providers:`` config dict
|
|
carry that flag even for canonical slugs, so keying on it would demote canonical providers
|
|
configured via the keyed schema.
|
|
"""
|
|
from hermes_cli.models import CANONICAL_PROVIDERS
|
|
|
|
order = {e.slug: i for i, e in enumerate(CANONICAL_PROVIDERS)}
|
|
canon = sorted(
|
|
(r for r in rows if r["slug"] in order),
|
|
key=lambda r: order[r["slug"]],
|
|
)
|
|
extras = [r for r in rows if r["slug"] not in order]
|
|
return canon + extras
|
|
|
|
|
|
def _apply_pricing(
|
|
rows: list[dict],
|
|
*,
|
|
force_fresh_nous_tier: bool = False,
|
|
) -> None:
|
|
"""Enrich each provider row with per-model pricing + Nous tier gating.
|
|
|
|
row["pricing"] = {model_id: {"input": "$3.00", "output": "$15.00", "cache": "$0.30" | None,
|
|
"free": bool}}
|
|
|
|
row["free_tier"] = bool # current account is free-tier row["unavailable_models"] = [...] # paid
|
|
models a free user can't pick
|
|
"""
|
|
from hermes_cli.models import (
|
|
_format_price_per_mtok,
|
|
check_nous_free_tier,
|
|
compute_sale_discount,
|
|
get_pricing_for_provider,
|
|
partition_nous_models_by_tier,
|
|
)
|
|
|
|
# Resolve Nous free-tier once (cached in models.py for the TTL window).
|
|
nous_free_tier: Optional[bool] = None
|
|
|
|
for row in rows:
|
|
slug = str(row.get("slug", "")).lower()
|
|
models = row.get("models") or []
|
|
if not models:
|
|
continue
|
|
try:
|
|
raw_pricing = get_pricing_for_provider(slug) or {}
|
|
except Exception:
|
|
raw_pricing = {}
|
|
if not raw_pricing:
|
|
continue
|
|
|
|
formatted: dict[str, dict] = {}
|
|
for mid in models:
|
|
p = raw_pricing.get(mid)
|
|
if not p:
|
|
continue
|
|
inp_raw = p.get("prompt", "")
|
|
out_raw = p.get("completion", "")
|
|
cache_raw = p.get("input_cache_read", "")
|
|
inp = _format_price_per_mtok(inp_raw) if inp_raw != "" else ""
|
|
out = _format_price_per_mtok(out_raw) if out_raw != "" else ""
|
|
entry: dict = {
|
|
"input": inp,
|
|
"output": out,
|
|
"cache": _format_price_per_mtok(cache_raw) if cache_raw else None,
|
|
# A model is "free" when both input and output cost nothing.
|
|
"free": inp == "free" and out in ("free", ""),
|
|
}
|
|
# Sale chrome is Nous Portal-only. Other providers (OpenRouter,
|
|
# Novita, …) never get discount_percent / was_* even if a nested
|
|
# pricing.original somehow appeared in their catalog. Free / $0
|
|
# models get flat -100% chrome (was_* only when the gateway
|
|
# served an original).
|
|
if slug == "nous":
|
|
sale = compute_sale_discount(
|
|
inp_raw, out_raw, p.get("original")
|
|
)
|
|
if sale is not None:
|
|
discount_percent, was_prompt_raw, was_out_raw = sale
|
|
entry["discount_percent"] = discount_percent
|
|
for key, was_raw in (("was_input", was_prompt_raw), ("was_output", was_out_raw)):
|
|
if was_raw != "":
|
|
entry[key] = _format_price_per_mtok(was_raw)
|
|
formatted[mid] = entry
|
|
|
|
if formatted:
|
|
row["pricing"] = formatted
|
|
|
|
if slug == "nous":
|
|
try:
|
|
if nous_free_tier is None:
|
|
nous_free_tier = check_nous_free_tier(
|
|
force_fresh=force_fresh_nous_tier
|
|
)
|
|
row["free_tier"] = bool(nous_free_tier)
|
|
row["unavailable_models"] = (
|
|
partition_nous_models_by_tier(list(models), raw_pricing, free_tier=True)[1]
|
|
if nous_free_tier
|
|
else []
|
|
)
|
|
except Exception:
|
|
# Tier detection failed — fail open (no gating) so the user
|
|
# is never blocked from picking a model.
|
|
row["free_tier"] = False
|
|
row["unavailable_models"] = []
|
|
|
|
|
|
def _local_runtime_row(ctx: "ConfigContext") -> dict | None:
|
|
"""Build the ``llamacpp`` provider row from staged local models.
|
|
|
|
Present whenever GGUFs are staged in the managed models directory — downloaded models must be
|
|
selectable even before the server is running (selection starts it via the runtime_provider seam
|
|
/ activate flow). Returns ``None`` when nothing is staged.
|
|
"""
|
|
try:
|
|
from hermes_cli.local_runtime.bootstrap import staged_model_ids
|
|
|
|
staged = staged_model_ids()
|
|
if not staged:
|
|
return None
|
|
current = (ctx.current_provider or "").strip().lower() in (
|
|
"llamacpp", "llama.cpp", "llama-cpp")
|
|
if not current:
|
|
# A LIVE session on the managed server reports provider "custom"
|
|
# (the resolution seam's label) with the managed base_url. Match
|
|
# on the endpoint so the picker still marks this row current —
|
|
# otherwise the session the user is chatting in shows no
|
|
# selection.
|
|
try:
|
|
from hermes_cli.local_runtime.endpoint import _state_endpoint
|
|
|
|
managed = _state_endpoint()
|
|
current = bool(
|
|
managed
|
|
and (ctx.current_base_url or "").strip().rstrip("/")
|
|
== managed["base_url"].rstrip("/"))
|
|
except Exception:
|
|
current = False
|
|
return {
|
|
"slug": "llamacpp",
|
|
# Bare "Local" everywhere user-facing: the engine name is an
|
|
# implementation detail (the pane brands this "Local models").
|
|
"name": "Local",
|
|
"is_current": current,
|
|
"is_user_defined": False,
|
|
"models": staged,
|
|
"total_models": len(staged),
|
|
"source": "local-runtime",
|
|
"authenticated": True, # the credential is reachability
|
|
"auth_type": "local",
|
|
"warning": None,
|
|
}
|
|
except Exception:
|
|
return None
|
|
|
|
|
|
def _moa_provider_row(current_provider: str = "") -> dict | None:
|
|
"""Build the virtual ``moa`` provider row for model pickers.
|
|
|
|
Shared by the CLI inventory and the gateway picker path so the row shape lives in one place.
|
|
Returns ``None`` when no MoA presets exist.
|
|
"""
|
|
try:
|
|
from hermes_cli.config import load_config
|
|
from hermes_cli.moa_config import normalize_moa_config
|
|
|
|
cfg = normalize_moa_config(load_config().get("moa") or {})
|
|
models = list(cfg.get("presets", {}).keys())
|
|
if not models:
|
|
return None
|
|
return {
|
|
"slug": "moa",
|
|
"name": "Mixture of Agents",
|
|
"is_current": (current_provider or "").lower() == "moa",
|
|
"is_user_defined": False,
|
|
"models": models,
|
|
"total_models": len(models),
|
|
"source": "virtual",
|
|
"authenticated": True,
|
|
"auth_type": "virtual",
|
|
"warning": "Aggregator acts as the selected model; references provide analysis before each call.",
|
|
}
|
|
except Exception:
|
|
return None
|