766 lines
31 KiB
Python
766 lines
31 KiB
Python
"""Single source of truth for provider identity in Hermes Agent."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
from dataclasses import dataclass
|
|
from typing import Any, Dict, List, Optional, Tuple
|
|
|
|
from utils import base_url_host_matches, base_url_hostname
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
# -- Hermes overlay ----------------------------------------------------------
|
|
# Hermes-specific metadata that models.dev doesn't provide.
|
|
|
|
@dataclass(frozen=True)
|
|
class HermesOverlay:
|
|
"""Hermes-specific provider metadata layered on top of models.dev."""
|
|
|
|
transport: str = "openai_chat" # openai_chat | anthropic_messages | codex_responses
|
|
is_aggregator: bool = False
|
|
auth_type: str = "api_key" # api_key | oauth_device_code | oauth_external | external_process
|
|
extra_env_vars: Tuple[str, ...] = () # env vars models.dev doesn't list
|
|
base_url_override: str = "" # override if models.dev URL is wrong/missing
|
|
base_url_env_var: str = "" # env var for user-custom base URL
|
|
keyless: bool = False # served anonymously — no credential exists to configure
|
|
|
|
|
|
HERMES_OVERLAYS: Dict[str, HermesOverlay] = {
|
|
"moa": HermesOverlay(auth_type="virtual", base_url_override="moa://local"),
|
|
"openrouter": HermesOverlay(is_aggregator=True, base_url_env_var="OPENROUTER_BASE_URL"),
|
|
"nous": HermesOverlay(
|
|
auth_type="oauth_device_code",
|
|
base_url_override="https://inference-api.nousresearch.com/v1",
|
|
),
|
|
"openai-codex": HermesOverlay(
|
|
transport="codex_responses",
|
|
auth_type="oauth_external",
|
|
base_url_override="https://chatgpt.com/backend-api/codex",
|
|
),
|
|
"openai-api": HermesOverlay(
|
|
transport="codex_responses",
|
|
base_url_override="https://api.openai.com/v1",
|
|
base_url_env_var="OPENAI_BASE_URL",
|
|
),
|
|
"xai-oauth": HermesOverlay(
|
|
transport="codex_responses",
|
|
auth_type="oauth_external",
|
|
base_url_override="https://api.x.ai/v1",
|
|
base_url_env_var="XAI_BASE_URL",
|
|
),
|
|
"qwen-oauth": HermesOverlay(
|
|
auth_type="oauth_external",
|
|
base_url_override="https://portal.qwen.ai/v1",
|
|
base_url_env_var="HERMES_QWEN_BASE_URL",
|
|
),
|
|
"lmstudio": HermesOverlay(
|
|
extra_env_vars=("LM_API_KEY",),
|
|
base_url_override="http://127.0.0.1:1234/v1",
|
|
base_url_env_var="LM_BASE_URL",
|
|
),
|
|
"copilot-acp": HermesOverlay(
|
|
transport="codex_responses",
|
|
auth_type="external_process",
|
|
base_url_override="acp://copilot",
|
|
base_url_env_var="COPILOT_ACP_BASE_URL",
|
|
),
|
|
"github-copilot": HermesOverlay(extra_env_vars=("COPILOT_GITHUB_TOKEN", "GH_TOKEN")),
|
|
"anthropic": HermesOverlay(
|
|
transport="anthropic_messages",
|
|
extra_env_vars=("ANTHROPIC_TOKEN", "CLAUDE_CODE_OAUTH_TOKEN"),
|
|
),
|
|
"zai": HermesOverlay(
|
|
extra_env_vars=("GLM_API_KEY", "ZAI_API_KEY", "Z_AI_API_KEY"),
|
|
base_url_env_var="GLM_BASE_URL",
|
|
),
|
|
"kimi-for-coding": HermesOverlay(base_url_env_var="KIMI_BASE_URL"),
|
|
"stepfun": HermesOverlay(
|
|
extra_env_vars=("STEPFUN_API_KEY",),
|
|
base_url_override="https://api.stepfun.ai/step_plan/v1",
|
|
base_url_env_var="STEPFUN_BASE_URL",
|
|
),
|
|
"minimax": HermesOverlay(transport="anthropic_messages", base_url_env_var="MINIMAX_BASE_URL"),
|
|
"minimax-oauth": HermesOverlay(
|
|
transport="anthropic_messages",
|
|
auth_type="oauth_external",
|
|
base_url_override="https://api.minimax.io/anthropic",
|
|
),
|
|
"minimax-cn": HermesOverlay(
|
|
transport="anthropic_messages",
|
|
base_url_env_var="MINIMAX_CN_BASE_URL",
|
|
),
|
|
"deepseek": HermesOverlay(base_url_env_var="DEEPSEEK_BASE_URL"),
|
|
"alibaba": HermesOverlay(base_url_env_var="DASHSCOPE_BASE_URL"),
|
|
"alibaba-coding-plan": HermesOverlay(base_url_env_var="ALIBABA_CODING_PLAN_BASE_URL"),
|
|
"vercel": HermesOverlay(is_aggregator=True),
|
|
"opencode": HermesOverlay(is_aggregator=True, base_url_env_var="OPENCODE_ZEN_BASE_URL"),
|
|
"opencode-go": HermesOverlay(is_aggregator=True, base_url_env_var="OPENCODE_GO_BASE_URL"),
|
|
"opencode-free": HermesOverlay(
|
|
is_aggregator=True,
|
|
base_url_override="https://opencode.ai/zen/v1",
|
|
keyless=True,
|
|
),
|
|
"kilo": HermesOverlay(is_aggregator=True, base_url_env_var="KILOCODE_BASE_URL"),
|
|
"huggingface": HermesOverlay(is_aggregator=True, base_url_env_var="HF_BASE_URL"),
|
|
"novita": HermesOverlay(is_aggregator=True, base_url_env_var="NOVITA_BASE_URL"),
|
|
"xai": HermesOverlay(
|
|
transport="codex_responses",
|
|
base_url_override="https://api.x.ai/v1",
|
|
base_url_env_var="XAI_BASE_URL",
|
|
),
|
|
"nvidia": HermesOverlay(
|
|
base_url_override="https://integrate.api.nvidia.com/v1",
|
|
base_url_env_var="NVIDIA_BASE_URL",
|
|
),
|
|
"xiaomi": HermesOverlay(base_url_env_var="XIAOMI_BASE_URL"),
|
|
"tencent-tokenhub": HermesOverlay(base_url_env_var="TOKENHUB_BASE_URL"),
|
|
"tencent-tokenplan": HermesOverlay(
|
|
transport="anthropic_messages",
|
|
base_url_override="https://api.lkeap.cloud.tencent.com/plan/anthropic",
|
|
base_url_env_var="TOKENPLAN_BASE_URL",
|
|
),
|
|
"arcee": HermesOverlay(
|
|
base_url_override="https://api.arcee.ai/api/v1",
|
|
base_url_env_var="ARCEE_BASE_URL",
|
|
),
|
|
"gmi": HermesOverlay(
|
|
extra_env_vars=("GMI_API_KEY",),
|
|
base_url_override="https://api.gmi-serving.com/v1",
|
|
base_url_env_var="GMI_BASE_URL",
|
|
),
|
|
"fireworks": HermesOverlay(
|
|
extra_env_vars=("FIREWORKS_API_KEY",),
|
|
base_url_override="https://api.fireworks.ai/inference/v1",
|
|
),
|
|
"actual": HermesOverlay(
|
|
transport="codex_responses",
|
|
extra_env_vars=("ACTUAL_API_KEY", "ACTUAL_BASE_URL"),
|
|
base_url_override="https://api.actual.inc/v1",
|
|
base_url_env_var="ACTUAL_BASE_URL",
|
|
),
|
|
"upstage": HermesOverlay(
|
|
extra_env_vars=("UPSTAGE_API_KEY",),
|
|
base_url_override="https://api.upstage.ai/v1",
|
|
base_url_env_var="UPSTAGE_BASE_URL",
|
|
),
|
|
"nebius-token-factory": HermesOverlay(
|
|
extra_env_vars=("NEBIUS_API_KEY", "NEBIUS_TOKEN_FACTORY_API_KEY"),
|
|
base_url_override="https://api.tokenfactory.nebius.com/v1",
|
|
base_url_env_var="NEBIUS_BASE_URL",
|
|
),
|
|
"ollama-cloud": HermesOverlay(
|
|
base_url_override="https://ollama.com/v1",
|
|
base_url_env_var="OLLAMA_BASE_URL",
|
|
),
|
|
# Azure Foundry: supports both OpenAI-style and Anthropic-style endpoints.
|
|
# The transport is determined at runtime from config.yaml model.api_mode.
|
|
"azure-foundry": HermesOverlay(base_url_env_var="AZURE_FOUNDRY_BASE_URL"), # openai_chat default; api_mode overrides
|
|
"bedrock": HermesOverlay(transport="bedrock_converse", auth_type="aws_sdk"),
|
|
# Vertex authenticates via OAuth2 (service-account JSON / ADC), not a
|
|
# static API key or models.dev entry — resolved specially by
|
|
# agent/vertex_adapter.py, like bedrock's aws_sdk. Without an overlay
|
|
# entry get_provider("vertex") returns None, which makes
|
|
# _preserve_provider_with_base_url() in agent/auxiliary_client.py treat
|
|
# a Vertex MoA slot's resolved (base_url, api_key) pair as an unknown
|
|
# custom endpoint instead of "vertex" — losing the provider identity
|
|
# that _refresh_provider_credentials() needs to re-mint an expired
|
|
# OAuth2 token on a 401.
|
|
"vertex": HermesOverlay(auth_type="vertex"),
|
|
}
|
|
|
|
|
|
# -- Resolved provider -------------------------------------------------------
|
|
# The merged result of models.dev + overlay + user config.
|
|
|
|
@dataclass
|
|
class ProviderDef:
|
|
"""Complete provider definition — merged from all sources."""
|
|
|
|
id: str
|
|
name: str
|
|
transport: str # openai_chat | anthropic_messages | codex_responses
|
|
api_key_env_vars: Tuple[str, ...] # all env vars to check for API key
|
|
base_url: str = ""
|
|
base_url_env_var: str = ""
|
|
is_aggregator: bool = False
|
|
auth_type: str = "api_key"
|
|
doc: str = ""
|
|
source: str = "" # "models.dev", "hermes", "user-config"
|
|
|
|
|
|
# -- Aliases ------------------------------------------------------------------
|
|
# Maps human-friendly / legacy names to canonical provider IDs.
|
|
# Uses models.dev IDs where possible.
|
|
|
|
# Aliases grouped by canonical provider id; ``ALIASES`` is the inverted lookup table.
|
|
_ALIAS_GROUPS: Dict[str, Tuple[str, ...]] = {
|
|
"openrouter": ("openai",),
|
|
"zai": ("glm", "z-ai", "z.ai", "zhipu"),
|
|
"xai": ("x-ai", "x.ai", "grok"),
|
|
"xai-oauth": ("grok-oauth", "xai-oauth", "x-ai-oauth", "xai-grok-oauth"),
|
|
"nvidia": ("nim", "nvidia-nim", "build-nvidia", "nemotron"),
|
|
"kimi-for-coding": ("kimi", "kimi-coding", "kimi-coding-cn", "moonshot"),
|
|
"stepfun": ("step", "stepfun-coding-plan"),
|
|
"minimax-cn": ("minimax-china", "minimax_cn"),
|
|
"anthropic": ("claude", "claude-code"),
|
|
"github-copilot": ("copilot", "github"),
|
|
"copilot-acp": ("github-copilot-acp",),
|
|
"vercel": ("ai-gateway", "aigateway", "vercel-ai-gateway"),
|
|
"opencode": ("opencode-zen", "zen"),
|
|
"opencode-go": ("go", "opencode-go-sub"),
|
|
"opencode-free": ("free", "opencode_free"),
|
|
"kilo": ("kilocode", "kilo-code", "kilo-gateway"),
|
|
"deepseek": ("deep-seek",),
|
|
"alibaba": ("dashscope", "aliyun", "qwen", "alibaba-cloud"),
|
|
"alibaba-coding-plan": ("alibaba_coding", "alibaba-coding", "alibaba_coding_plan"),
|
|
"huggingface": ("hf", "hugging-face", "huggingface-hub"),
|
|
"novita": ("novita-ai", "novitaai"),
|
|
"xiaomi": ("mimo", "xiaomi-mimo"),
|
|
"tencent-tokenhub": ("tencent", "tokenhub", "tencent-cloud", "tencentmaas"),
|
|
"tencent-tokenplan": ("tokenplan", "tencent-lkeap"),
|
|
"bedrock": ("aws", "aws-bedrock", "amazon-bedrock", "amazon"),
|
|
"arcee": ("arcee-ai", "arceeai"),
|
|
"gmi": ("gmi-cloud", "gmicloud"),
|
|
"fireworks": ("fireworks-ai", "fw"),
|
|
"upstage": ("solar",),
|
|
"actual": ("actual-computer", "actualcomputer", "aci"),
|
|
"nebius-token-factory": (
|
|
"nebius", "nebius-tokenfactory", "nebius-tf", "token-factory", "tokenfactory",
|
|
),
|
|
"lmstudio": ("lmstudio", "lm-studio", "lm_studio"),
|
|
"custom": ("ollama",),
|
|
"local": ("vllm", "llamacpp", "llama.cpp", "llama-cpp"),
|
|
}
|
|
ALIASES: Dict[str, str] = {alias: canon for canon, aliases in _ALIAS_GROUPS.items() for alias in aliases}
|
|
|
|
|
|
# -- Display labels -----------------------------------------------------------
|
|
# Built dynamically from models.dev + overlays. Fallback for providers
|
|
# not in the catalog.
|
|
|
|
_LABEL_OVERRIDES: Dict[str, str] = {
|
|
"moa": "Mixture of Agents",
|
|
"nous": "Nous Portal",
|
|
"openai-codex": "ChatGPT or Codex Subscription",
|
|
"copilot-acp": "GitHub Copilot ACP",
|
|
"stepfun": "StepFun Step Plan",
|
|
"xiaomi": "Xiaomi MiMo",
|
|
"gmi": "GMI Cloud",
|
|
"upstage": "Upstage Solar",
|
|
"actual": "Actual Computer",
|
|
"tencent-tokenhub": "Tencent TokenHub",
|
|
"nebius-token-factory": "Nebius Token Factory",
|
|
"tencent-tokenplan": "Tencent TokenPlan",
|
|
"lmstudio": "LM Studio",
|
|
"local": "Local endpoint",
|
|
"bedrock": "AWS Bedrock",
|
|
"vertex": "Google Vertex AI",
|
|
"ollama-cloud": "Ollama Cloud",
|
|
"xai-oauth": "xAI Grok OAuth (SuperGrok / Premium+)",
|
|
"opencode-free": "OpenCode Free",
|
|
}
|
|
|
|
|
|
# -- Transport → API mode mapping ---------------------------------------------
|
|
|
|
TRANSPORT_TO_API_MODE: Dict[str, str] = {
|
|
"openai_chat": "chat_completions",
|
|
"anthropic_messages": "anthropic_messages",
|
|
"codex_responses": "codex_responses",
|
|
"bedrock_converse": "bedrock_converse",
|
|
}
|
|
|
|
|
|
# -- Helper functions ---------------------------------------------------------
|
|
|
|
def normalize_provider(name: str) -> str:
|
|
"""Resolve aliases and normalise casing to a canonical provider id."""
|
|
key = name.strip().lower()
|
|
return ALIASES.get(key, key)
|
|
|
|
|
|
def get_provider(name: str, *, allow_network: bool = True) -> Optional[ProviderDef]:
|
|
"""Look up a built-in provider by id or alias.
|
|
|
|
Resolution order: 1. Hermes overlays (for providers not in models.dev: nous, openai-codex, etc.)
|
|
2. models.dev catalog + Hermes overlay
|
|
"""
|
|
canonical = normalize_provider(name)
|
|
|
|
# Try to get models.dev data
|
|
try:
|
|
from agent.models_dev import get_provider_info as _mdev_provider
|
|
# Keep the single-argument call on the default path: test sites
|
|
# monkeypatch get_provider_info with single-arg lambdas.
|
|
mdev_info = (
|
|
_mdev_provider(canonical)
|
|
if allow_network
|
|
else _mdev_provider(canonical, allow_network=False)
|
|
)
|
|
except Exception:
|
|
mdev_info = None
|
|
|
|
overlay = HERMES_OVERLAYS.get(canonical)
|
|
|
|
if mdev_info is not None:
|
|
# Merge models.dev + overlay (defaults when no overlay); env vars = models.dev + hermes extra
|
|
ov = overlay or HermesOverlay()
|
|
env_vars = list(mdev_info.env)
|
|
for ev in ov.extra_env_vars:
|
|
if ev not in env_vars:
|
|
env_vars.append(ev)
|
|
return ProviderDef(
|
|
id=canonical,
|
|
name=mdev_info.name,
|
|
transport=ov.transport,
|
|
api_key_env_vars=tuple(env_vars),
|
|
base_url=ov.base_url_override or mdev_info.api,
|
|
base_url_env_var=ov.base_url_env_var,
|
|
is_aggregator=ov.is_aggregator,
|
|
auth_type=ov.auth_type,
|
|
doc=mdev_info.doc,
|
|
source="models.dev",
|
|
)
|
|
|
|
if overlay is not None:
|
|
# Hermes-only provider (not in models.dev)
|
|
return ProviderDef(
|
|
id=canonical,
|
|
name=_LABEL_OVERRIDES.get(canonical, canonical),
|
|
transport=overlay.transport,
|
|
api_key_env_vars=overlay.extra_env_vars,
|
|
base_url=overlay.base_url_override,
|
|
base_url_env_var=overlay.base_url_env_var,
|
|
is_aggregator=overlay.is_aggregator,
|
|
auth_type=overlay.auth_type,
|
|
source="hermes",
|
|
)
|
|
|
|
# Plugin-registered provider profiles (plugins/model-providers/<name>/).
|
|
# Providers that ship only as plugin profiles (e.g. commandcode,
|
|
# tencent-tokenhub) are absent from models.dev and HERMES_OVERLAYS, so
|
|
# without this fallback they resolve as "Unknown provider" in /model,
|
|
# --provider, and the model-switch path even though the picker lists them
|
|
# (CANONICAL_PROVIDERS auto-extends from the same plugin registry).
|
|
try:
|
|
from providers import get_provider_profile as _profile
|
|
|
|
_prof = _profile(canonical)
|
|
# Only profiles with a concrete endpoint resolve here. Placeholder
|
|
# profiles like ``custom`` (aliases: ollama/local/vllm) ship with an
|
|
# empty base_url and are completed by config.yaml custom_providers —
|
|
# resolving them here would preempt resolve_provider_full's
|
|
# custom-provider step and collapse keyed IDs
|
|
# (``custom:local-...``) back to a bare, endpoint-less ``custom``.
|
|
if _prof is not None and (_prof.base_url or "").strip():
|
|
_api_mode_to_transport = {v: k for k, v in TRANSPORT_TO_API_MODE.items()}
|
|
_transport = _api_mode_to_transport.get(_prof.api_mode, "openai_chat")
|
|
return ProviderDef(
|
|
id=canonical,
|
|
name=_prof.display_name or _prof.name or canonical,
|
|
transport=_transport,
|
|
api_key_env_vars=tuple(_prof.env_vars or ()),
|
|
base_url=_prof.base_url or "",
|
|
auth_type=_prof.auth_type or "api_key",
|
|
source="plugin-profile",
|
|
)
|
|
except Exception:
|
|
pass
|
|
|
|
return None
|
|
|
|
|
|
def get_label(provider_id: str) -> str:
|
|
"""Get a human-readable display name for a provider."""
|
|
canonical = normalize_provider(provider_id)
|
|
|
|
# Check label overrides first
|
|
if canonical in _LABEL_OVERRIDES:
|
|
return _LABEL_OVERRIDES[canonical]
|
|
|
|
# Try models.dev
|
|
pdef = get_provider(canonical)
|
|
if pdef:
|
|
return pdef.name
|
|
|
|
return canonical
|
|
|
|
|
|
def is_aggregator(provider: str) -> bool:
|
|
"""Return True when the provider is a multi-model aggregator."""
|
|
provider_norm = normalize_provider(provider or "")
|
|
if provider_norm.startswith("custom:"):
|
|
return True
|
|
pdef = get_provider(provider_norm)
|
|
return pdef.is_aggregator if pdef else False
|
|
|
|
|
|
# Flat-namespace resellers (e.g. opencode-go, opencode-zen) are flagged
|
|
# ``is_aggregator=True`` because their live ``/v1/models`` returns bare model
|
|
# IDs ("deepseek-v4-flash") rather than ``vendor/model`` routing slugs — the
|
|
# model-switch resolver relies on that flag to search their flat catalog
|
|
# (see model_switch.py step d). But they are NOT routing aggregators: every
|
|
# model they list is a first-party model served under their own subscription,
|
|
# not a passthrough route to another provider's endpoint. The picker dedup
|
|
# (build_models_payload) must treat them differently from true routers like
|
|
# OpenRouter — a reseller's first-party "minimax-m3" must never be stripped
|
|
# just because a user's custom proxy also happens to serve a same-named model.
|
|
_FLAT_NAMESPACE_RESELLERS: frozenset[str] = frozenset({
|
|
# Use normalized provider IDs: normalize_provider("opencode-zen") -> "opencode".
|
|
"opencode-go",
|
|
"opencode",
|
|
})
|
|
|
|
|
|
def is_routing_aggregator(provider: str) -> bool:
|
|
"""True only for TRUE routing aggregators (OpenRouter, named ``custom:*`` proxies).
|
|
|
|
Unlike ``is_aggregator``, excludes flat-namespace resellers (opencode-go/zen) whose catalog is
|
|
first-party. Use for "would selecting this model silently re-route away from the intended
|
|
provider?" -- i.e. picker dedup; reseller rows must not be deduped against user proxies.
|
|
"""
|
|
provider_norm = normalize_provider(provider or "")
|
|
if provider_norm in _FLAT_NAMESPACE_RESELLERS:
|
|
return False
|
|
return is_aggregator(provider_norm)
|
|
|
|
|
|
def is_official_openai_host(base_url: str) -> bool:
|
|
"""True when *base_url* points at OpenAI's official API host family.
|
|
|
|
Hostname-parsed matching only — never substring — so lookalike hosts
|
|
(``api.openai.com.attacker.test``) and path-segment spoofs (``proxy.test/api.openai.com/v1``)
|
|
are rejected. A genuine ``*.api.openai.com`` subdomain requires control of openai.com DNS, so
|
|
the dot-suffix match does not reopen the #32243 spoofing hole.
|
|
"""
|
|
return base_url_host_matches(base_url, "api.openai.com")
|
|
|
|
|
|
def host_mandated_api_mode(base_url: str = "") -> Optional[str]:
|
|
"""Return the wire protocol a specific endpoint *requires*, or None.
|
|
|
|
Some hosts only accept one API mode and reject the others outright: - api.openai.com only
|
|
accepts the Responses API for its (reasoning) models when tools + reasoning are in play
|
|
(chat/completions 400s).
|
|
|
|
These are *mandatory* — a session carrying a stale api_mode (e.g. a /model switch that kept the
|
|
previous provider's ``chat_completions``) must be overridden to the host's required mode, not
|
|
merely filled in when empty.
|
|
"""
|
|
if not base_url:
|
|
return None
|
|
url_lower = base_url.rstrip("/").lower()
|
|
hostname = base_url_hostname(base_url)
|
|
# Exact-hostname matching only — never bare substring — so lookalike hosts
|
|
# (api.openai.com.attacker.test) and path-segment spoofs
|
|
# (proxy.test/api.openai.com/v1) are NOT treated as the real endpoint. (#32243)
|
|
if hostname == "api.kimi.com" and "/coding" in url_lower:
|
|
return "anthropic_messages"
|
|
if hostname == "api.anthropic.com" or url_lower.endswith("/anthropic"):
|
|
return "anthropic_messages"
|
|
# Official OpenAI host family: canonical + data-residency regional hosts
|
|
# (us./eu.api.openai.com) all mandate the Responses API for reasoning
|
|
# models with tools. Shared predicate keeps this lane in lockstep with
|
|
# catalog filtering and listing authority.
|
|
if is_official_openai_host(base_url):
|
|
return "codex_responses"
|
|
if hostname in _RESPONSES_NATIVE_HOSTS:
|
|
return "codex_responses"
|
|
if hostname.startswith("bedrock-runtime.") and base_url_host_matches(base_url, "amazonaws.com"):
|
|
return "bedrock_converse"
|
|
return None
|
|
|
|
|
|
# Exact hostnames (#32243) that are Responses-API-native:
|
|
# - api.meta.ai: Meta Model API only achieves prompt-cache hits on the Responses API with
|
|
# prompt_cache_retention; chat/completions stays cache-cold (0% vs 93-99% measured).
|
|
# - api.router.com: Ramp Router keeps reasoning-effort validation, reasoning summaries and prompt
|
|
# caching on /v1/responses; /v1/chat/completions is a minimal shim (docs.router.com/api/endpoint).
|
|
_RESPONSES_NATIVE_HOSTS: frozenset[str] = frozenset({"api.meta.ai", "api.router.com"})
|
|
|
|
|
|
def nous_api_mode(model: str = "") -> str:
|
|
"""Resolve the wire protocol for a Nous Portal model.
|
|
|
|
Portal serves its ``anthropic/*`` catalog on a native Anthropic Messages route
|
|
(``/v1/messages``) alongside the OpenAI-compatible ``/v1/chat/completions`` used by every other
|
|
model it proxies.
|
|
|
|
When *model* is empty/unknown, defaults to ``chat_completions`` — the historical Nous transport
|
|
— so callers that don't yet know the model stay on the safer OpenAI-compatible path.
|
|
"""
|
|
if str(model or "").strip().lower().startswith("anthropic/"):
|
|
return "anthropic_messages"
|
|
return "chat_completions"
|
|
|
|
|
|
def determine_api_mode(provider: str, base_url: str = "", model: str = "") -> str:
|
|
"""Determine the API mode (wire protocol) for a provider/endpoint.
|
|
|
|
Resolution order: 1. Host-mandated mode (special endpoints that only accept one protocol). 2.
|
|
Nous Portal dual-wire (model-derived; overlay alone is openai_chat). 3. Known provider →
|
|
transport → TRANSPORT_TO_API_MODE. 4. Direct provider checks (bedrock). 5. Default:
|
|
'chat_completions'.
|
|
"""
|
|
mandated = host_mandated_api_mode(base_url)
|
|
if mandated is not None:
|
|
return mandated
|
|
|
|
# Nous is dual-wire: anthropic/* → Messages, everything else →
|
|
# chat_completions. The Hermes overlay still advertises openai_chat
|
|
# (the majority of the Portal catalog), so the transport lookup below
|
|
# would pin Claude on the wrong wire without this carve-out.
|
|
provider_norm = (provider or "").strip().lower()
|
|
if provider_norm in {"nous", "nous-portal", "nousresearch"}:
|
|
return nous_api_mode(model)
|
|
|
|
pdef = get_provider(provider)
|
|
if pdef is not None:
|
|
return TRANSPORT_TO_API_MODE.get(pdef.transport, "chat_completions")
|
|
|
|
# Direct provider checks for providers not in HERMES_OVERLAYS
|
|
if provider == "bedrock":
|
|
return "bedrock_converse"
|
|
|
|
return "chat_completions"
|
|
|
|
|
|
# -- Provider from user config ------------------------------------------------
|
|
|
|
def resolve_user_provider(name: str, user_config: Dict[str, Any]) -> Optional[ProviderDef]:
|
|
"""Resolve a provider from the user's config.yaml ``providers:`` section."""
|
|
if not user_config or not isinstance(user_config, dict):
|
|
return None
|
|
|
|
entry = user_config.get(name)
|
|
if not isinstance(entry, dict):
|
|
return None
|
|
|
|
# Extract fields
|
|
display_name = entry.get("name", "") or name
|
|
api_url = entry.get("api", "") or entry.get("url", "") or entry.get("base_url", "") or ""
|
|
key_env = entry.get("key_env") or entry.get("api_key_env") or ""
|
|
transport = entry.get("transport", "openai_chat") or "openai_chat"
|
|
|
|
env_vars: List[str] = []
|
|
if key_env:
|
|
env_vars.append(key_env)
|
|
|
|
return ProviderDef(
|
|
id=name,
|
|
name=display_name,
|
|
transport=transport,
|
|
api_key_env_vars=tuple(env_vars),
|
|
base_url=api_url,
|
|
is_aggregator=False,
|
|
auth_type="api_key",
|
|
source="user-config",
|
|
)
|
|
|
|
|
|
def custom_provider_slug(display_name: str, provider_key: str = "") -> str:
|
|
"""Build the stable ``custom:`` identity for a configured provider.
|
|
|
|
Keyed ``providers:`` entries use their config key so the identity survives display-name
|
|
changes; legacy ``custom_providers:`` entries have no key, so their normalized display name
|
|
remains the identity.
|
|
"""
|
|
identity = str(provider_key or "").strip() or str(display_name or "").strip()
|
|
normalized = identity.lower().replace(" ", "-")
|
|
return normalized if normalized.startswith("custom:") else f"custom:{normalized}"
|
|
|
|
|
|
def custom_provider_aliases(
|
|
display_name: str,
|
|
provider_key: str = "",
|
|
) -> frozenset[str]:
|
|
"""Return every current and legacy identity accepted for one endpoint."""
|
|
aliases: set[str] = set()
|
|
for value in (display_name, provider_key):
|
|
raw = str(value or "").strip().lower()
|
|
if not raw:
|
|
continue
|
|
normalized = raw.replace(" ", "-")
|
|
aliases.update({raw, normalized, custom_provider_slug(normalized)})
|
|
if normalized.startswith("custom:"):
|
|
suffix = normalized.split(":", 1)[1]
|
|
if suffix:
|
|
aliases.update({suffix, f"custom:{normalized}"})
|
|
return frozenset(aliases)
|
|
|
|
|
|
def resolve_custom_provider(
|
|
name: str,
|
|
custom_providers: Optional[List[Dict[str, Any]]],
|
|
) -> Optional[ProviderDef]:
|
|
"""Resolve a provider from the user's config.yaml ``custom_providers`` list."""
|
|
if not custom_providers or not isinstance(custom_providers, list):
|
|
return None
|
|
|
|
requested = (name or "").strip().lower()
|
|
if not requested:
|
|
return None
|
|
|
|
# If the stored provider is the bare string "custom" (corrupt state
|
|
# from a prior model-switch bug), fall back to the first custom
|
|
# provider entry so existing configs self-heal. (GH #17478)
|
|
bare_custom_fallback = requested == "custom"
|
|
first_valid: Optional[ProviderDef] = None
|
|
|
|
for entry in custom_providers:
|
|
if not isinstance(entry, dict):
|
|
continue
|
|
|
|
display_name = (entry.get("name") or "").strip()
|
|
api_url = (
|
|
entry.get("base_url", "")
|
|
or entry.get("url", "")
|
|
or entry.get("api", "")
|
|
or ""
|
|
).strip()
|
|
if not display_name or not api_url:
|
|
continue
|
|
|
|
key_env = (entry.get("key_env") or "").strip()
|
|
provider_key = (entry.get("provider_key") or "").strip()
|
|
pdef = ProviderDef(
|
|
id=custom_provider_slug(display_name, provider_key),
|
|
name=display_name,
|
|
transport="openai_chat",
|
|
api_key_env_vars=(key_env,) if key_env else (),
|
|
base_url=api_url,
|
|
is_aggregator=False,
|
|
auth_type="api_key",
|
|
source="user-config",
|
|
)
|
|
|
|
# Stash the first valid entry for bare-"custom" fallback
|
|
if first_valid is None:
|
|
first_valid = pdef
|
|
|
|
if requested in custom_provider_aliases(display_name, provider_key):
|
|
return pdef
|
|
|
|
# Self-heal: bare "custom" matched nothing — return first valid entry
|
|
if bare_custom_fallback and first_valid:
|
|
return first_valid
|
|
|
|
return None
|
|
|
|
|
|
def resolve_provider_full(
|
|
name: str,
|
|
user_providers: Optional[Dict[str, Any]] = None,
|
|
custom_providers: Optional[List[Dict[str, Any]]] = None,
|
|
) -> Optional[ProviderDef]:
|
|
"""Full resolution chain: built-in → models.dev → user config."""
|
|
canonical = normalize_provider(name)
|
|
raw = name.strip().lower()
|
|
|
|
# 0. User-defined config providers win over the built-in alias table.
|
|
# A user who declares ``providers.<name>`` in config.yaml has stated
|
|
# explicit intent for that name — it must not be hijacked by a legacy
|
|
# vendor alias (e.g. bare "openai" → "openrouter"). Resolve the raw
|
|
# name against user config FIRST so a configured ``providers.openai``
|
|
# (pointing at api.openai.com) beats the alias that would otherwise
|
|
# silently route to OpenRouter. Only the raw (pre-alias) name is tried
|
|
# here; canonical/alias resolution still happens below.
|
|
if user_providers:
|
|
user_pdef = resolve_user_provider(raw, user_providers)
|
|
if user_pdef is not None:
|
|
return user_pdef
|
|
|
|
# 0.5 Exact Hermes provider IDs must win over LOSSY alias collapsing.
|
|
# Example: kimi-coding-cn should stay distinct from kimi-coding instead of
|
|
# normalizing through the shared models.dev alias "kimi-for-coding".
|
|
# A collapse is lossy only when MULTIPLE distinct registry providers
|
|
# normalize to the same canonical name — resolving through the alias
|
|
# would then lose which one the caller meant. Single-entry rewrites
|
|
# (e.g. "copilot" → "github-copilot") are correct routing and must keep
|
|
# resolving through the built-in chain below so overlay transports apply.
|
|
if canonical != raw:
|
|
try:
|
|
from hermes_cli.auth import PROVIDER_REGISTRY as _AUTH_PROVIDER_REGISTRY
|
|
_pcfg = _AUTH_PROVIDER_REGISTRY.get(raw)
|
|
if _pcfg is not None:
|
|
_collapsed_siblings = [
|
|
_rid
|
|
for _rid in _AUTH_PROVIDER_REGISTRY
|
|
if normalize_provider(_rid) == canonical
|
|
]
|
|
if len(_collapsed_siblings) > 1:
|
|
return ProviderDef(
|
|
id=_pcfg.id,
|
|
name=_pcfg.name,
|
|
transport="openai_chat",
|
|
api_key_env_vars=tuple(_pcfg.api_key_env_vars or ()),
|
|
base_url=_pcfg.inference_base_url or "",
|
|
source="hermes-auth-registry",
|
|
)
|
|
except Exception:
|
|
pass
|
|
|
|
# 1. Built-in (models.dev + overlays)
|
|
pdef = get_provider(canonical)
|
|
if pdef is not None:
|
|
return pdef
|
|
|
|
# 2. User-defined providers from config
|
|
if user_providers:
|
|
# Try canonical name
|
|
user_pdef = resolve_user_provider(canonical, user_providers)
|
|
if user_pdef is not None:
|
|
return user_pdef
|
|
# Try original name (in case alias didn't match)
|
|
user_pdef = resolve_user_provider(raw, user_providers)
|
|
if user_pdef is not None:
|
|
return user_pdef
|
|
|
|
# 2b. Saved custom providers from config
|
|
custom_pdef = resolve_custom_provider(name, custom_providers)
|
|
if custom_pdef is not None:
|
|
return custom_pdef
|
|
|
|
# 2c. Managed local runtime: the llamacpp aliases are a real provider
|
|
# whenever the managed server (or a detected external one) resolves —
|
|
# no credential and no providers: entry required, the credential is
|
|
# reachability. Without this rung the model-switch path rejected the
|
|
# very provider the Local Models 'Use' flow writes to config
|
|
# ("Unknown provider 'llamacpp'" from the desktop dropdown).
|
|
if raw in ("llamacpp", "llama.cpp", "llama-cpp"):
|
|
try:
|
|
from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint
|
|
|
|
endpoint = resolve_llamacpp_endpoint(wait_for_boot_s=0)
|
|
except Exception:
|
|
endpoint = None
|
|
if endpoint:
|
|
return ProviderDef(
|
|
id="llamacpp",
|
|
name="Local",
|
|
transport="openai_chat",
|
|
api_key_env_vars=(),
|
|
base_url=endpoint["base_url"],
|
|
source="local-runtime",
|
|
)
|
|
|
|
# 3. Try models.dev directly (for providers not in our ALIASES)
|
|
try:
|
|
from agent.models_dev import get_provider_info as _mdev_provider
|
|
mdev_info = _mdev_provider(canonical)
|
|
if mdev_info is not None:
|
|
return ProviderDef(
|
|
id=canonical,
|
|
name=mdev_info.name,
|
|
transport="openai_chat",
|
|
api_key_env_vars=mdev_info.env,
|
|
base_url=mdev_info.api,
|
|
source="models.dev",
|
|
)
|
|
except Exception:
|
|
pass
|
|
|
|
return None
|