diff --git a/hermes_cli/provider_catalog.py b/hermes_cli/provider_catalog.py index d80e3dbe49..8f381e6a05 100644 --- a/hermes_cli/provider_catalog.py +++ b/hermes_cli/provider_catalog.py @@ -1,35 +1,26 @@ """Unified provider catalog — one source of truth for the provider universe. The provider list shown by ``hermes model`` (CLI/TUI) and the desktop Settings → Providers tabs -(Accounts + API keys) **must be the same set**. Every provider added after those lists were written -silently went missing from the GUI — e.g. - -* ``auth_type`` / ``api_key_env_vars`` / ``base_url_env_var`` from -:data:`hermes_cli.auth.PROVIDER_REGISTRY` (credential truth), and * ``display_name`` / -``description`` / ``signup_url`` from the provider's :class:`providers.base.ProviderProfile` when -one exists, falling back to the ``CANONICAL_PROVIDERS`` entry's ``label`` / ``tui_desc`` and the -``OPTIONAL_ENV_VARS`` signup URL otherwise (many profiles leave these blank, and four canonical -providers have no profile at all — lmstudio, openai-api, tencent-tokenhub, xai-oauth — so the -fallbacks are load-bearing). +(Accounts + API keys) **must be the same set**; providers added after those lists were written +silently went missing from the GUI. ``auth_type`` / ``api_key_env_vars`` / ``base_url_env_var`` +come from :data:`hermes_cli.auth.PROVIDER_REGISTRY` (credential truth); ``display_name`` / +``description`` / ``signup_url`` from the provider's :class:`providers.base.ProviderProfile`, falling +back to the ``CANONICAL_PROVIDERS`` entry's ``label`` / ``tui_desc`` and the ``OPTIONAL_ENV_VARS`` +signup URL (many profiles leave these blank, and lmstudio, openai-api, tencent-tokenhub, xai-oauth +have no profile at all — the fallbacks are load-bearing). """ from __future__ import annotations from dataclasses import dataclass -# Auth types that authenticate via an account / sign-in flow rather than a -# pasted API key. These route to the desktop "Accounts" tab; everything else -# (api_key, and aws_sdk which is configured via AWS_REGION/AWS_PROFILE) routes -# to the "API keys" tab. Mirrors the auth_type strings used in -# hermes_cli.auth.PROVIDER_REGISTRY and providers.base.ProviderProfile. +# Auth types that authenticate via an account / sign-in flow rather than a pasted API key; these +# route to the desktop "Accounts" tab, everything else (api_key, and aws_sdk configured via +# AWS_REGION/AWS_PROFILE) to "API keys". Mirrors the auth_type strings in PROVIDER_REGISTRY and +# ProviderProfile: external_process = copilot-acp (spawns `copilot --acp --stdio`), copilot = GitHub +# Copilot token / gh auth. _ACCOUNTS_AUTH_TYPES: frozenset[str] = frozenset( - { - "oauth_device_code", - "oauth_external", - "oauth_minimax", - "external_process", # copilot-acp: spawns `copilot --acp --stdio` - "copilot", # GitHub Copilot token / gh auth - } + {"oauth_device_code", "oauth_external", "oauth_minimax", "external_process", "copilot"} ) @@ -54,19 +45,19 @@ def tab_for_auth_type(auth_type: str) -> str: return "accounts" if auth_type in _ACCOUNTS_AUTH_TYPES else "keys" +def _is_url_var(name: str) -> bool: + return name.endswith("_BASE_URL") or name.endswith("_URL") + + def _split_env_vars(env_vars: tuple[str, ...]) -> tuple[tuple[str, ...], str]: """Split a profile's ``env_vars`` into (api_key_vars, base_url_var).""" - keys = tuple(v for v in env_vars if not (v.endswith("_BASE_URL") or v.endswith("_URL"))) - base = next((v for v in env_vars if v.endswith("_BASE_URL") or v.endswith("_URL")), "") - return keys, base + return tuple(v for v in env_vars if not _is_url_var(v)), next((v for v in env_vars if _is_url_var(v)), "") def _safe_import(module: str, attr: str, default): - """Import ``attr`` from ``module``; return ``default`` on ANY failure. - - This module is on the import path of the web server and the CLI, and a - provider-plugin import error must never blank the whole catalog. - """ + """Import ``attr`` from ``module``; ``default`` on ANY failure — this module is on the import + path of the web server and the CLI, and a provider-plugin import error must never blank the + whole catalog.""" try: return getattr(__import__(module, fromlist=[attr]), attr) except Exception: @@ -74,74 +65,47 @@ def _safe_import(module: str, attr: str, default): def provider_catalog() -> list[ProviderDescriptor]: - """Return one descriptor per provider in the ``hermes model`` universe. - - Membership is :data:`CANONICAL_PROVIDERS` (auto-extended by provider plugins). Auth/env come - from ``PROVIDER_REGISTRY``; display metadata from ``ProviderProfile`` with canonical/env - fallbacks so providers without a profile still resolve sensibly. - """ + """One descriptor per provider in the ``hermes model`` universe (:data:`CANONICAL_PROVIDERS`, + auto-extended by provider plugins). Auth/env from ``PROVIDER_REGISTRY``; display metadata from + ``ProviderProfile`` with canonical/env fallbacks so profile-less providers still resolve.""" from hermes_cli.models import CANONICAL_PROVIDERS - PROVIDER_REGISTRY = _safe_import("hermes_cli.auth", "PROVIDER_REGISTRY", {}) OPTIONAL_ENV_VARS = _safe_import("hermes_cli.config", "OPTIONAL_ENV_VARS", {}) - # Hermes overlays carry auth_type for providers that have no registry/profile - # entry of their own — notably the ``moa`` virtual provider (auth_type - # "virtual"), which has no real credential and no network endpoint. + # Overlays carry auth_type for providers with no registry/profile entry — notably the ``moa`` + # virtual provider (auth_type "virtual"), which has no credential and no network endpoint. HERMES_OVERLAYS = _safe_import("hermes_cli.providers", "HERMES_OVERLAYS", {}) try: from providers import list_providers - profiles = {p.name: p for p in list_providers()} except Exception: profiles = {} - out: list[ProviderDescriptor] = [] for order, entry in enumerate(CANONICAL_PROVIDERS): slug = entry.slug cfg = PROVIDER_REGISTRY.get(slug) prof = profiles.get(slug) overlay = HERMES_OVERLAYS.get(slug) - - # auth_type: registry is authoritative; fall back to profile, then the - # Hermes overlay (e.g. moa → "virtual"), then api_key. - auth_type = ( - (cfg.auth_type if cfg else "") - or (prof.auth_type if prof else "") - or (overlay.auth_type if overlay else "") - or "api_key" - ) - - # Credential env vars: registry first (it already normalizes these), - # else derive from the profile's env_vars tuple. + # auth_type: registry is authoritative; then profile, then overlay (moa → "virtual"), then api_key. + auth_type = ((cfg.auth_type if cfg else "") or (prof.auth_type if prof else "") + or (overlay.auth_type if overlay else "") or "api_key") + # Credential env vars: registry first (already normalized), else derived from the profile. if cfg and cfg.api_key_env_vars: - api_key_vars = tuple(cfg.api_key_env_vars) - base_url_var = cfg.base_url_env_var or "" + api_key_vars, base_url_var = tuple(cfg.api_key_env_vars), cfg.base_url_env_var or "" elif prof and prof.env_vars: api_key_vars, base_url_var = _split_env_vars(tuple(prof.env_vars)) else: api_key_vars, base_url_var = (), "" - label = (prof.display_name if prof else "") or entry.label or slug - description = (prof.description if prof else "") or entry.tui_desc or label signup_url = (prof.signup_url if prof else "") or "" if not signup_url and api_key_vars: - info = OPTIONAL_ENV_VARS.get(api_key_vars[0]) or {} - signup_url = info.get("url") or "" - + signup_url = (OPTIONAL_ENV_VARS.get(api_key_vars[0]) or {}).get("url") or "" out.append( ProviderDescriptor( - slug=slug, - label=label, - description=description, - auth_type=auth_type, - tab=tab_for_auth_type(auth_type), - api_key_env_vars=api_key_vars, - base_url_env_var=base_url_var, - signup_url=signup_url, - order=order, - # Keyless providers (e.g. opencode-free) are served - # anonymously: there is no credential to configure, so the - # GUI renders no key card and contract tests exempt them. + slug=slug, label=label, description=(prof.description if prof else "") or entry.tui_desc or label, + auth_type=auth_type, tab=tab_for_auth_type(auth_type), api_key_env_vars=api_key_vars, + base_url_env_var=base_url_var, signup_url=signup_url, order=order, + # Keyless providers (opencode-free) are served anonymously: no key card in the GUI, + # and contract tests exempt them. keyless=bool(overlay.keyless) if overlay else False, ) ) diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py index 306e433c89..3e91e69f14 100644 --- a/hermes_cli/providers.py +++ b/hermes_cli/providers.py @@ -11,8 +11,7 @@ from utils import base_url_host_matches, base_url_hostname logger = logging.getLogger(__name__) -# -- Hermes overlay ---------------------------------------------------------- -# Hermes-specific metadata that models.dev doesn't provide. +# -- Hermes overlay: metadata models.dev doesn't provide ---------------------- @dataclass(frozen=True) class HermesOverlay: @@ -30,153 +29,76 @@ class HermesOverlay: HERMES_OVERLAYS: Dict[str, HermesOverlay] = { "moa": HermesOverlay(auth_type="virtual", base_url_override="moa://local"), "openrouter": HermesOverlay(is_aggregator=True, base_url_env_var="OPENROUTER_BASE_URL"), - "nous": HermesOverlay( - auth_type="oauth_device_code", - base_url_override="https://inference-api.nousresearch.com/v1", - ), - "openai-codex": HermesOverlay( - transport="codex_responses", - auth_type="oauth_external", - base_url_override="https://chatgpt.com/backend-api/codex", - ), - "openai-api": HermesOverlay( - transport="codex_responses", - base_url_override="https://api.openai.com/v1", - base_url_env_var="OPENAI_BASE_URL", - ), - "xai-oauth": HermesOverlay( - transport="codex_responses", - auth_type="oauth_external", - base_url_override="https://api.x.ai/v1", - base_url_env_var="XAI_BASE_URL", - ), - "qwen-oauth": HermesOverlay( - auth_type="oauth_external", - base_url_override="https://portal.qwen.ai/v1", - base_url_env_var="HERMES_QWEN_BASE_URL", - ), - "lmstudio": HermesOverlay( - extra_env_vars=("LM_API_KEY",), - base_url_override="http://127.0.0.1:1234/v1", - base_url_env_var="LM_BASE_URL", - ), - "copilot-acp": HermesOverlay( - transport="codex_responses", - auth_type="external_process", - base_url_override="acp://copilot", - base_url_env_var="COPILOT_ACP_BASE_URL", - ), + "nous": HermesOverlay(auth_type="oauth_device_code", base_url_override="https://inference-api.nousresearch.com/v1"), + "openai-codex": HermesOverlay(transport="codex_responses", auth_type="oauth_external", + base_url_override="https://chatgpt.com/backend-api/codex"), + "openai-api": HermesOverlay(transport="codex_responses", base_url_override="https://api.openai.com/v1", + base_url_env_var="OPENAI_BASE_URL"), + "xai-oauth": HermesOverlay(transport="codex_responses", auth_type="oauth_external", + base_url_override="https://api.x.ai/v1", base_url_env_var="XAI_BASE_URL"), + "qwen-oauth": HermesOverlay(auth_type="oauth_external", base_url_override="https://portal.qwen.ai/v1", + base_url_env_var="HERMES_QWEN_BASE_URL"), + "lmstudio": HermesOverlay(extra_env_vars=("LM_API_KEY",), base_url_override="http://127.0.0.1:1234/v1", + base_url_env_var="LM_BASE_URL"), + "copilot-acp": HermesOverlay(transport="codex_responses", auth_type="external_process", + base_url_override="acp://copilot", base_url_env_var="COPILOT_ACP_BASE_URL"), "github-copilot": HermesOverlay(extra_env_vars=("COPILOT_GITHUB_TOKEN", "GH_TOKEN")), - "anthropic": HermesOverlay( - transport="anthropic_messages", - extra_env_vars=("ANTHROPIC_TOKEN", "CLAUDE_CODE_OAUTH_TOKEN"), - ), - "zai": HermesOverlay( - extra_env_vars=("GLM_API_KEY", "ZAI_API_KEY", "Z_AI_API_KEY"), - base_url_env_var="GLM_BASE_URL", - ), + "anthropic": HermesOverlay(transport="anthropic_messages", extra_env_vars=("ANTHROPIC_TOKEN", "CLAUDE_CODE_OAUTH_TOKEN")), + "zai": HermesOverlay(extra_env_vars=("GLM_API_KEY", "ZAI_API_KEY", "Z_AI_API_KEY"), base_url_env_var="GLM_BASE_URL"), "kimi-for-coding": HermesOverlay(base_url_env_var="KIMI_BASE_URL"), - "stepfun": HermesOverlay( - extra_env_vars=("STEPFUN_API_KEY",), - base_url_override="https://api.stepfun.ai/step_plan/v1", - base_url_env_var="STEPFUN_BASE_URL", - ), + "stepfun": HermesOverlay(extra_env_vars=("STEPFUN_API_KEY",), + base_url_override="https://api.stepfun.ai/step_plan/v1", + base_url_env_var="STEPFUN_BASE_URL"), "minimax": HermesOverlay(transport="anthropic_messages", base_url_env_var="MINIMAX_BASE_URL"), - "minimax-oauth": HermesOverlay( - transport="anthropic_messages", - auth_type="oauth_external", - base_url_override="https://api.minimax.io/anthropic", - ), - "minimax-cn": HermesOverlay( - transport="anthropic_messages", - base_url_env_var="MINIMAX_CN_BASE_URL", - ), + "minimax-oauth": HermesOverlay(transport="anthropic_messages", auth_type="oauth_external", + base_url_override="https://api.minimax.io/anthropic"), + "minimax-cn": HermesOverlay(transport="anthropic_messages", base_url_env_var="MINIMAX_CN_BASE_URL"), "deepseek": HermesOverlay(base_url_env_var="DEEPSEEK_BASE_URL"), "alibaba": HermesOverlay(base_url_env_var="DASHSCOPE_BASE_URL"), "alibaba-coding-plan": HermesOverlay(base_url_env_var="ALIBABA_CODING_PLAN_BASE_URL"), "vercel": HermesOverlay(is_aggregator=True), "opencode": HermesOverlay(is_aggregator=True, base_url_env_var="OPENCODE_ZEN_BASE_URL"), "opencode-go": HermesOverlay(is_aggregator=True, base_url_env_var="OPENCODE_GO_BASE_URL"), - "opencode-free": HermesOverlay( - is_aggregator=True, - base_url_override="https://opencode.ai/zen/v1", - keyless=True, - ), + "opencode-free": HermesOverlay(is_aggregator=True, base_url_override="https://opencode.ai/zen/v1", keyless=True), "kilo": HermesOverlay(is_aggregator=True, base_url_env_var="KILOCODE_BASE_URL"), "huggingface": HermesOverlay(is_aggregator=True, base_url_env_var="HF_BASE_URL"), "novita": HermesOverlay(is_aggregator=True, base_url_env_var="NOVITA_BASE_URL"), - "xai": HermesOverlay( - transport="codex_responses", - base_url_override="https://api.x.ai/v1", - base_url_env_var="XAI_BASE_URL", - ), - "nvidia": HermesOverlay( - base_url_override="https://integrate.api.nvidia.com/v1", - base_url_env_var="NVIDIA_BASE_URL", - ), + "xai": HermesOverlay(transport="codex_responses", base_url_override="https://api.x.ai/v1", base_url_env_var="XAI_BASE_URL"), + "nvidia": HermesOverlay(base_url_override="https://integrate.api.nvidia.com/v1", base_url_env_var="NVIDIA_BASE_URL"), "xiaomi": HermesOverlay(base_url_env_var="XIAOMI_BASE_URL"), "tencent-tokenhub": HermesOverlay(base_url_env_var="TOKENHUB_BASE_URL"), - "tencent-tokenplan": HermesOverlay( - transport="anthropic_messages", - base_url_override="https://api.lkeap.cloud.tencent.com/plan/anthropic", - base_url_env_var="TOKENPLAN_BASE_URL", - ), - "arcee": HermesOverlay( - base_url_override="https://api.arcee.ai/api/v1", - base_url_env_var="ARCEE_BASE_URL", - ), - "gmi": HermesOverlay( - extra_env_vars=("GMI_API_KEY",), - base_url_override="https://api.gmi-serving.com/v1", - base_url_env_var="GMI_BASE_URL", - ), - "fireworks": HermesOverlay( - extra_env_vars=("FIREWORKS_API_KEY",), - base_url_override="https://api.fireworks.ai/inference/v1", - ), - "actual": HermesOverlay( - transport="codex_responses", - extra_env_vars=("ACTUAL_API_KEY", "ACTUAL_BASE_URL"), - base_url_override="https://api.actual.inc/v1", - base_url_env_var="ACTUAL_BASE_URL", - ), - "upstage": HermesOverlay( - extra_env_vars=("UPSTAGE_API_KEY",), - base_url_override="https://api.upstage.ai/v1", - base_url_env_var="UPSTAGE_BASE_URL", - ), - "nebius-token-factory": HermesOverlay( - extra_env_vars=("NEBIUS_API_KEY", "NEBIUS_TOKEN_FACTORY_API_KEY"), - base_url_override="https://api.tokenfactory.nebius.com/v1", - base_url_env_var="NEBIUS_BASE_URL", - ), - "ollama-cloud": HermesOverlay( - base_url_override="https://ollama.com/v1", - base_url_env_var="OLLAMA_BASE_URL", - ), - # Azure Foundry: supports both OpenAI-style and Anthropic-style endpoints. - # The transport is determined at runtime from config.yaml model.api_mode. - "azure-foundry": HermesOverlay(base_url_env_var="AZURE_FOUNDRY_BASE_URL"), # openai_chat default; api_mode overrides + "tencent-tokenplan": HermesOverlay(transport="anthropic_messages", + base_url_override="https://api.lkeap.cloud.tencent.com/plan/anthropic", + base_url_env_var="TOKENPLAN_BASE_URL"), + "arcee": HermesOverlay(base_url_override="https://api.arcee.ai/api/v1", base_url_env_var="ARCEE_BASE_URL"), + "gmi": HermesOverlay(extra_env_vars=("GMI_API_KEY",), base_url_override="https://api.gmi-serving.com/v1", + base_url_env_var="GMI_BASE_URL"), + "fireworks": HermesOverlay(extra_env_vars=("FIREWORKS_API_KEY",), + base_url_override="https://api.fireworks.ai/inference/v1"), + "actual": HermesOverlay(transport="codex_responses", extra_env_vars=("ACTUAL_API_KEY", "ACTUAL_BASE_URL"), + base_url_override="https://api.actual.inc/v1", base_url_env_var="ACTUAL_BASE_URL"), + "upstage": HermesOverlay(extra_env_vars=("UPSTAGE_API_KEY",), base_url_override="https://api.upstage.ai/v1", + base_url_env_var="UPSTAGE_BASE_URL"), + "nebius-token-factory": HermesOverlay(extra_env_vars=("NEBIUS_API_KEY", "NEBIUS_TOKEN_FACTORY_API_KEY"), + base_url_override="https://api.tokenfactory.nebius.com/v1", + base_url_env_var="NEBIUS_BASE_URL"), + "ollama-cloud": HermesOverlay(base_url_override="https://ollama.com/v1", base_url_env_var="OLLAMA_BASE_URL"), + # Azure Foundry serves OpenAI- and Anthropic-style endpoints; transport comes from model.api_mode. + "azure-foundry": HermesOverlay(base_url_env_var="AZURE_FOUNDRY_BASE_URL"), "bedrock": HermesOverlay(transport="bedrock_converse", auth_type="aws_sdk"), - # Vertex authenticates via OAuth2 (service-account JSON / ADC), not a - # static API key or models.dev entry — resolved specially by - # agent/vertex_adapter.py, like bedrock's aws_sdk. Without an overlay - # entry get_provider("vertex") returns None, which makes - # _preserve_provider_with_base_url() in agent/auxiliary_client.py treat - # a Vertex MoA slot's resolved (base_url, api_key) pair as an unknown - # custom endpoint instead of "vertex" — losing the provider identity - # that _refresh_provider_credentials() needs to re-mint an expired - # OAuth2 token on a 401. + # Vertex is OAuth2 (service-account JSON / ADC), resolved by agent/vertex_adapter.py. Without an + # overlay get_provider("vertex") is None and auxiliary_client._preserve_provider_with_base_url + # would treat a Vertex MoA slot as an unknown custom endpoint, losing the identity + # _refresh_provider_credentials() needs to re-mint an expired token on 401. "vertex": HermesOverlay(auth_type="vertex"), } # -- Resolved provider ------------------------------------------------------- -# The merged result of models.dev + overlay + user config. @dataclass class ProviderDef: - """Complete provider definition — merged from all sources.""" + """Complete provider definition — merged from models.dev + overlay + user config.""" id: str name: str @@ -190,86 +112,50 @@ class ProviderDef: source: str = "" # "models.dev", "hermes", "user-config" -# -- Aliases ------------------------------------------------------------------ -# Maps human-friendly / legacy names to canonical provider IDs. -# Uses models.dev IDs where possible. - -# Aliases grouped by canonical provider id; ``ALIASES`` is the inverted lookup table. +# -- Aliases: human-friendly / legacy names grouped by canonical (models.dev where possible) id; +# ``ALIASES`` is the inverted lookup table. --------------------------------------------------- _ALIAS_GROUPS: Dict[str, Tuple[str, ...]] = { - "openrouter": ("openai",), - "zai": ("glm", "z-ai", "z.ai", "zhipu"), - "xai": ("x-ai", "x.ai", "grok"), + "openrouter": ("openai",), "zai": ("glm", "z-ai", "z.ai", "zhipu"), "xai": ("x-ai", "x.ai", "grok"), "xai-oauth": ("grok-oauth", "xai-oauth", "x-ai-oauth", "xai-grok-oauth"), "nvidia": ("nim", "nvidia-nim", "build-nvidia", "nemotron"), "kimi-for-coding": ("kimi", "kimi-coding", "kimi-coding-cn", "moonshot"), - "stepfun": ("step", "stepfun-coding-plan"), - "minimax-cn": ("minimax-china", "minimax_cn"), - "anthropic": ("claude", "claude-code"), - "github-copilot": ("copilot", "github"), - "copilot-acp": ("github-copilot-acp",), - "vercel": ("ai-gateway", "aigateway", "vercel-ai-gateway"), - "opencode": ("opencode-zen", "zen"), - "opencode-go": ("go", "opencode-go-sub"), - "opencode-free": ("free", "opencode_free"), - "kilo": ("kilocode", "kilo-code", "kilo-gateway"), - "deepseek": ("deep-seek",), - "alibaba": ("dashscope", "aliyun", "qwen", "alibaba-cloud"), + "stepfun": ("step", "stepfun-coding-plan"), "minimax-cn": ("minimax-china", "minimax_cn"), + "anthropic": ("claude", "claude-code"), "github-copilot": ("copilot", "github"), + "copilot-acp": ("github-copilot-acp",), "vercel": ("ai-gateway", "aigateway", "vercel-ai-gateway"), + "opencode": ("opencode-zen", "zen"), "opencode-go": ("go", "opencode-go-sub"), + "opencode-free": ("free", "opencode_free"), "kilo": ("kilocode", "kilo-code", "kilo-gateway"), + "deepseek": ("deep-seek",), "alibaba": ("dashscope", "aliyun", "qwen", "alibaba-cloud"), "alibaba-coding-plan": ("alibaba_coding", "alibaba-coding", "alibaba_coding_plan"), - "huggingface": ("hf", "hugging-face", "huggingface-hub"), - "novita": ("novita-ai", "novitaai"), - "xiaomi": ("mimo", "xiaomi-mimo"), - "tencent-tokenhub": ("tencent", "tokenhub", "tencent-cloud", "tencentmaas"), + "huggingface": ("hf", "hugging-face", "huggingface-hub"), "novita": ("novita-ai", "novitaai"), + "xiaomi": ("mimo", "xiaomi-mimo"), "tencent-tokenhub": ("tencent", "tokenhub", "tencent-cloud", "tencentmaas"), "tencent-tokenplan": ("tokenplan", "tencent-lkeap"), - "bedrock": ("aws", "aws-bedrock", "amazon-bedrock", "amazon"), - "arcee": ("arcee-ai", "arceeai"), - "gmi": ("gmi-cloud", "gmicloud"), - "fireworks": ("fireworks-ai", "fw"), - "upstage": ("solar",), + "bedrock": ("aws", "aws-bedrock", "amazon-bedrock", "amazon"), "arcee": ("arcee-ai", "arceeai"), + "gmi": ("gmi-cloud", "gmicloud"), "fireworks": ("fireworks-ai", "fw"), "upstage": ("solar",), "actual": ("actual-computer", "actualcomputer", "aci"), - "nebius-token-factory": ( - "nebius", "nebius-tokenfactory", "nebius-tf", "token-factory", "tokenfactory", - ), - "lmstudio": ("lmstudio", "lm-studio", "lm_studio"), - "custom": ("ollama",), + "nebius-token-factory": ("nebius", "nebius-tokenfactory", "nebius-tf", "token-factory", "tokenfactory"), + "lmstudio": ("lmstudio", "lm-studio", "lm_studio"), "custom": ("ollama",), "local": ("vllm", "llamacpp", "llama.cpp", "llama-cpp"), } ALIASES: Dict[str, str] = {alias: canon for canon, aliases in _ALIAS_GROUPS.items() for alias in aliases} -# -- Display labels ----------------------------------------------------------- -# Built dynamically from models.dev + overlays. Fallback for providers -# not in the catalog. +# -- Display labels for providers not in the models.dev catalog --------------- _LABEL_OVERRIDES: Dict[str, str] = { - "moa": "Mixture of Agents", - "nous": "Nous Portal", - "openai-codex": "ChatGPT or Codex Subscription", - "copilot-acp": "GitHub Copilot ACP", - "stepfun": "StepFun Step Plan", - "xiaomi": "Xiaomi MiMo", - "gmi": "GMI Cloud", - "upstage": "Upstage Solar", - "actual": "Actual Computer", - "tencent-tokenhub": "Tencent TokenHub", - "nebius-token-factory": "Nebius Token Factory", - "tencent-tokenplan": "Tencent TokenPlan", - "lmstudio": "LM Studio", - "local": "Local endpoint", - "bedrock": "AWS Bedrock", - "vertex": "Google Vertex AI", - "ollama-cloud": "Ollama Cloud", - "xai-oauth": "xAI Grok OAuth (SuperGrok / Premium+)", - "opencode-free": "OpenCode Free", + "moa": "Mixture of Agents", "nous": "Nous Portal", "openai-codex": "ChatGPT or Codex Subscription", + "copilot-acp": "GitHub Copilot ACP", "stepfun": "StepFun Step Plan", "xiaomi": "Xiaomi MiMo", "gmi": "GMI Cloud", + "upstage": "Upstage Solar", "actual": "Actual Computer", "tencent-tokenhub": "Tencent TokenHub", + "nebius-token-factory": "Nebius Token Factory", "tencent-tokenplan": "Tencent TokenPlan", "lmstudio": "LM Studio", + "local": "Local endpoint", "bedrock": "AWS Bedrock", "vertex": "Google Vertex AI", "ollama-cloud": "Ollama Cloud", + "xai-oauth": "xAI Grok OAuth (SuperGrok / Premium+)", "opencode-free": "OpenCode Free", } # -- Transport → API mode mapping --------------------------------------------- TRANSPORT_TO_API_MODE: Dict[str, str] = { - "openai_chat": "chat_completions", - "anthropic_messages": "anthropic_messages", - "codex_responses": "codex_responses", - "bedrock_converse": "bedrock_converse", + "openai_chat": "chat_completions", "anthropic_messages": "anthropic_messages", + "codex_responses": "codex_responses", "bedrock_converse": "bedrock_converse", } @@ -281,111 +167,66 @@ def normalize_provider(name: str) -> str: return ALIASES.get(key, key) -def get_provider(name: str, *, allow_network: bool = True) -> Optional[ProviderDef]: - """Look up a built-in provider by id or alias. - - Resolution order: 1. Hermes overlays (for providers not in models.dev: nous, openai-codex, etc.) - 2. models.dev catalog + Hermes overlay - """ - canonical = normalize_provider(name) - - # Try to get models.dev data +def _models_dev_info(canonical: str, allow_network: bool = True): + """models.dev entry or None. Single-arg call on the default path: test sites monkeypatch + ``get_provider_info`` with single-arg lambdas.""" try: from agent.models_dev import get_provider_info as _mdev_provider - # Keep the single-argument call on the default path: test sites - # monkeypatch get_provider_info with single-arg lambdas. - mdev_info = ( - _mdev_provider(canonical) - if allow_network - else _mdev_provider(canonical, allow_network=False) - ) + return _mdev_provider(canonical) if allow_network else _mdev_provider(canonical, allow_network=False) except Exception: - mdev_info = None + return None + +def _overlay_pdef(canonical, ov: HermesOverlay, name, env_vars, base_url, doc, source) -> ProviderDef: + return ProviderDef(id=canonical, name=name, transport=ov.transport, api_key_env_vars=env_vars, base_url=base_url, + base_url_env_var=ov.base_url_env_var, is_aggregator=ov.is_aggregator, auth_type=ov.auth_type, doc=doc, + source=source) + + +def get_provider(name: str, *, allow_network: bool = True) -> Optional[ProviderDef]: + """Look up a built-in provider by id or alias: models.dev catalog merged with the Hermes overlay; + Hermes-only overlay (nous, openai-codex, …); plugin provider profiles with a concrete endpoint.""" + canonical = normalize_provider(name) + mdev_info = _models_dev_info(canonical, allow_network) overlay = HERMES_OVERLAYS.get(canonical) - if mdev_info is not None: - # Merge models.dev + overlay (defaults when no overlay); env vars = models.dev + hermes extra ov = overlay or HermesOverlay() env_vars = list(mdev_info.env) for ev in ov.extra_env_vars: if ev not in env_vars: env_vars.append(ev) - return ProviderDef( - id=canonical, - name=mdev_info.name, - transport=ov.transport, - api_key_env_vars=tuple(env_vars), - base_url=ov.base_url_override or mdev_info.api, - base_url_env_var=ov.base_url_env_var, - is_aggregator=ov.is_aggregator, - auth_type=ov.auth_type, - doc=mdev_info.doc, - source="models.dev", - ) - + return _overlay_pdef(canonical, ov, mdev_info.name, tuple(env_vars), ov.base_url_override or mdev_info.api, + mdev_info.doc, "models.dev") if overlay is not None: - # Hermes-only provider (not in models.dev) - return ProviderDef( - id=canonical, - name=_LABEL_OVERRIDES.get(canonical, canonical), - transport=overlay.transport, - api_key_env_vars=overlay.extra_env_vars, - base_url=overlay.base_url_override, - base_url_env_var=overlay.base_url_env_var, - is_aggregator=overlay.is_aggregator, - auth_type=overlay.auth_type, - source="hermes", - ) - - # Plugin-registered provider profiles (plugins/model-providers//). - # Providers that ship only as plugin profiles (e.g. commandcode, - # tencent-tokenhub) are absent from models.dev and HERMES_OVERLAYS, so - # without this fallback they resolve as "Unknown provider" in /model, - # --provider, and the model-switch path even though the picker lists them - # (CANONICAL_PROVIDERS auto-extends from the same plugin registry). + return _overlay_pdef(canonical, overlay, _LABEL_OVERRIDES.get(canonical, canonical), overlay.extra_env_vars, + overlay.base_url_override, "", "hermes") + # Plugin-registered profiles (plugins/model-providers//) absent from models.dev and + # HERMES_OVERLAYS would otherwise be "Unknown provider" in /model, --provider and model-switch + # even though the picker lists them. Only profiles with a concrete endpoint resolve here: + # placeholder profiles like ``custom`` (aliases ollama/local/vllm) ship an empty base_url and + # are completed by config.yaml custom_providers — resolving them would preempt + # resolve_provider_full's custom step and collapse keyed ``custom:`` ids to bare custom. try: from providers import get_provider_profile as _profile - _prof = _profile(canonical) - # Only profiles with a concrete endpoint resolve here. Placeholder - # profiles like ``custom`` (aliases: ollama/local/vllm) ship with an - # empty base_url and are completed by config.yaml custom_providers — - # resolving them here would preempt resolve_provider_full's - # custom-provider step and collapse keyed IDs - # (``custom:local-...``) back to a bare, endpoint-less ``custom``. if _prof is not None and (_prof.base_url or "").strip(): _api_mode_to_transport = {v: k for k, v in TRANSPORT_TO_API_MODE.items()} - _transport = _api_mode_to_transport.get(_prof.api_mode, "openai_chat") - return ProviderDef( - id=canonical, - name=_prof.display_name or _prof.name or canonical, - transport=_transport, - api_key_env_vars=tuple(_prof.env_vars or ()), - base_url=_prof.base_url or "", - auth_type=_prof.auth_type or "api_key", - source="plugin-profile", - ) + return ProviderDef(id=canonical, name=_prof.display_name or _prof.name or canonical, + transport=_api_mode_to_transport.get(_prof.api_mode, "openai_chat"), + api_key_env_vars=tuple(_prof.env_vars or ()), base_url=_prof.base_url or "", + auth_type=_prof.auth_type or "api_key", source="plugin-profile") except Exception: pass - return None def get_label(provider_id: str) -> str: - """Get a human-readable display name for a provider.""" + """Human-readable display name: label override, else models.dev name, else the id.""" canonical = normalize_provider(provider_id) - - # Check label overrides first if canonical in _LABEL_OVERRIDES: return _LABEL_OVERRIDES[canonical] - - # Try models.dev pdef = get_provider(canonical) - if pdef: - return pdef.name - - return canonical + return pdef.name if pdef else canonical def is_aggregator(provider: str) -> bool: @@ -397,30 +238,19 @@ def is_aggregator(provider: str) -> bool: return pdef.is_aggregator if pdef else False -# Flat-namespace resellers (e.g. opencode-go, opencode-zen) are flagged -# ``is_aggregator=True`` because their live ``/v1/models`` returns bare model -# IDs ("deepseek-v4-flash") rather than ``vendor/model`` routing slugs — the -# model-switch resolver relies on that flag to search their flat catalog -# (see model_switch.py step d). But they are NOT routing aggregators: every -# model they list is a first-party model served under their own subscription, -# not a passthrough route to another provider's endpoint. The picker dedup -# (build_models_payload) must treat them differently from true routers like -# OpenRouter — a reseller's first-party "minimax-m3" must never be stripped -# just because a user's custom proxy also happens to serve a same-named model. -_FLAT_NAMESPACE_RESELLERS: frozenset[str] = frozenset({ - # Use normalized provider IDs: normalize_provider("opencode-zen") -> "opencode". - "opencode-go", - "opencode", -}) +# Flat-namespace resellers (opencode-go, opencode-zen) are flagged ``is_aggregator=True`` because +# their live ``/v1/models`` returns bare model IDs ("deepseek-v4-flash") rather than +# ``vendor/model`` routing slugs — model_switch searches their flat catalog on that flag. But they +# are NOT routing aggregators: every listed model is first-party under their own subscription, so +# picker dedup (build_models_payload) must not strip a reseller's "minimax-m3" just because a +# user's custom proxy serves a same-named model. Normalized ids: "opencode-zen" -> "opencode". +_FLAT_NAMESPACE_RESELLERS: frozenset[str] = frozenset({"opencode-go", "opencode"}) def is_routing_aggregator(provider: str) -> bool: - """True only for TRUE routing aggregators (OpenRouter, named ``custom:*`` proxies). - - Unlike ``is_aggregator``, excludes flat-namespace resellers (opencode-go/zen) whose catalog is - first-party. Use for "would selecting this model silently re-route away from the intended - provider?" -- i.e. picker dedup; reseller rows must not be deduped against user proxies. - """ + """True only for TRUE routing aggregators (OpenRouter, named ``custom:*`` proxies) — excludes + flat-namespace resellers whose catalog is first-party. Use for "would selecting this model + silently re-route away from the intended provider?" (picker dedup).""" provider_norm = normalize_provider(provider or "") if provider_norm in _FLAT_NAMESPACE_RESELLERS: return False @@ -428,154 +258,99 @@ def is_routing_aggregator(provider: str) -> bool: def is_official_openai_host(base_url: str) -> bool: - """True when *base_url* points at OpenAI's official API host family. - - Hostname-parsed matching only — never substring — so lookalike hosts - (``api.openai.com.attacker.test``) and path-segment spoofs (``proxy.test/api.openai.com/v1``) - are rejected. A genuine ``*.api.openai.com`` subdomain requires control of openai.com DNS, so - the dot-suffix match does not reopen the #32243 spoofing hole. - """ + """True when *base_url* points at OpenAI's official API host family. Hostname-parsed matching + only — never substring — so lookalike hosts (``api.openai.com.attacker.test``) and path-segment + spoofs (``proxy.test/api.openai.com/v1``) are rejected; a genuine ``*.api.openai.com`` + subdomain requires control of openai.com DNS.""" return base_url_host_matches(base_url, "api.openai.com") +# Exact hostnames that are Responses-API-native: api.meta.ai only achieves prompt-cache hits on +# Responses with prompt_cache_retention (chat/completions stays cache-cold); api.router.com (Ramp +# Router) keeps reasoning validation/summaries and prompt caching on /v1/responses and serves +# /v1/chat/completions as a minimal shim. +_RESPONSES_NATIVE_HOSTS: frozenset[str] = frozenset({"api.meta.ai", "api.router.com"}) + + def host_mandated_api_mode(base_url: str = "") -> Optional[str]: - """Return the wire protocol a specific endpoint *requires*, or None. - - Some hosts only accept one API mode and reject the others outright: - api.openai.com only - accepts the Responses API for its (reasoning) models when tools + reasoning are in play - (chat/completions 400s). - - These are *mandatory* — a session carrying a stale api_mode (e.g. a /model switch that kept the - previous provider's ``chat_completions``) must be overridden to the host's required mode, not - merely filled in when empty. - """ + """Return the wire protocol a specific endpoint *requires*, or None. Some hosts accept exactly + one API mode (api.openai.com 400s chat/completions for reasoning models with tools); these are + *mandatory*: a session carrying a stale api_mode (a /model switch that kept the previous + provider's ``chat_completions``) must be overridden, not merely filled in when empty. + Exact-hostname matching only — never substring — so lookalike hosts and path-segment spoofs are + not treated as the real endpoint.""" if not base_url: return None url_lower = base_url.rstrip("/").lower() hostname = base_url_hostname(base_url) - # Exact-hostname matching only — never bare substring — so lookalike hosts - # (api.openai.com.attacker.test) and path-segment spoofs - # (proxy.test/api.openai.com/v1) are NOT treated as the real endpoint. (#32243) if hostname == "api.kimi.com" and "/coding" in url_lower: return "anthropic_messages" if hostname == "api.anthropic.com" or url_lower.endswith("/anthropic"): return "anthropic_messages" - # Official OpenAI host family: canonical + data-residency regional hosts - # (us./eu.api.openai.com) all mandate the Responses API for reasoning - # models with tools. Shared predicate keeps this lane in lockstep with - # catalog filtering and listing authority. - if is_official_openai_host(base_url): - return "codex_responses" - if hostname in _RESPONSES_NATIVE_HOSTS: + # Official OpenAI host family (canonical + us./eu. data-residency hosts) mandates Responses; + # the shared predicate keeps this in lockstep with catalog filtering and listing authority. + if is_official_openai_host(base_url) or hostname in _RESPONSES_NATIVE_HOSTS: return "codex_responses" if hostname.startswith("bedrock-runtime.") and base_url_host_matches(base_url, "amazonaws.com"): return "bedrock_converse" return None -# Exact hostnames (#32243) that are Responses-API-native: -# - api.meta.ai: Meta Model API only achieves prompt-cache hits on the Responses API with -# prompt_cache_retention; chat/completions stays cache-cold (0% vs 93-99% measured). -# - api.router.com: Ramp Router keeps reasoning-effort validation, reasoning summaries and prompt -# caching on /v1/responses; /v1/chat/completions is a minimal shim (docs.router.com/api/endpoint). -_RESPONSES_NATIVE_HOSTS: frozenset[str] = frozenset({"api.meta.ai", "api.router.com"}) - - def nous_api_mode(model: str = "") -> str: - """Resolve the wire protocol for a Nous Portal model. - - Portal serves its ``anthropic/*`` catalog on a native Anthropic Messages route - (``/v1/messages``) alongside the OpenAI-compatible ``/v1/chat/completions`` used by every other - model it proxies. - - When *model* is empty/unknown, defaults to ``chat_completions`` — the historical Nous transport - — so callers that don't yet know the model stay on the safer OpenAI-compatible path. - """ + """Wire protocol for a Nous Portal model: Portal serves its ``anthropic/*`` catalog on a native + Messages route alongside OpenAI-compatible chat/completions for everything else. Empty/unknown + model defaults to ``chat_completions`` (the historical Nous transport) as the safer path.""" if str(model or "").strip().lower().startswith("anthropic/"): return "anthropic_messages" return "chat_completions" def determine_api_mode(provider: str, base_url: str = "", model: str = "") -> str: - """Determine the API mode (wire protocol) for a provider/endpoint. - - Resolution order: 1. Host-mandated mode (special endpoints that only accept one protocol). 2. - Nous Portal dual-wire (model-derived; overlay alone is openai_chat). 3. Known provider → - transport → TRANSPORT_TO_API_MODE. 4. Direct provider checks (bedrock). 5. Default: - 'chat_completions'. - """ + """API mode (wire protocol) for a provider/endpoint: host-mandated mode, then Nous dual-wire + (model-derived — the overlay alone says openai_chat and would pin Claude on the wrong wire), + then the known provider's transport, then bedrock, else ``chat_completions``.""" mandated = host_mandated_api_mode(base_url) if mandated is not None: return mandated - - # Nous is dual-wire: anthropic/* → Messages, everything else → - # chat_completions. The Hermes overlay still advertises openai_chat - # (the majority of the Portal catalog), so the transport lookup below - # would pin Claude on the wrong wire without this carve-out. - provider_norm = (provider or "").strip().lower() - if provider_norm in {"nous", "nous-portal", "nousresearch"}: + if (provider or "").strip().lower() in {"nous", "nous-portal", "nousresearch"}: return nous_api_mode(model) - pdef = get_provider(provider) if pdef is not None: return TRANSPORT_TO_API_MODE.get(pdef.transport, "chat_completions") - - # Direct provider checks for providers not in HERMES_OVERLAYS if provider == "bedrock": return "bedrock_converse" - return "chat_completions" # -- Provider from user config ------------------------------------------------ +def _user_pdef(pid: str, name: str, base_url: str, key_env: str, transport: str = "openai_chat") -> ProviderDef: + """``source="user-config"`` ProviderDef shared by ``providers:`` and ``custom_providers:`` entries.""" + return ProviderDef(id=pid, name=name, transport=transport, api_key_env_vars=(key_env,) if key_env else (), + base_url=base_url, is_aggregator=False, auth_type="api_key", source="user-config") + + def resolve_user_provider(name: str, user_config: Dict[str, Any]) -> Optional[ProviderDef]: """Resolve a provider from the user's config.yaml ``providers:`` section.""" - if not user_config or not isinstance(user_config, dict): - return None - - entry = user_config.get(name) + entry = user_config.get(name) if isinstance(user_config, dict) and user_config else None if not isinstance(entry, dict): return None - - # Extract fields - display_name = entry.get("name", "") or name - api_url = entry.get("api", "") or entry.get("url", "") or entry.get("base_url", "") or "" - key_env = entry.get("key_env") or entry.get("api_key_env") or "" - transport = entry.get("transport", "openai_chat") or "openai_chat" - - env_vars: List[str] = [] - if key_env: - env_vars.append(key_env) - - return ProviderDef( - id=name, - name=display_name, - transport=transport, - api_key_env_vars=tuple(env_vars), - base_url=api_url, - is_aggregator=False, - auth_type="api_key", - source="user-config", - ) + return _user_pdef(name, entry.get("name", "") or name, + entry.get("api", "") or entry.get("url", "") or entry.get("base_url", "") or "", + entry.get("key_env") or entry.get("api_key_env") or "", + entry.get("transport", "openai_chat") or "openai_chat") def custom_provider_slug(display_name: str, provider_key: str = "") -> str: - """Build the stable ``custom:`` identity for a configured provider. - - Keyed ``providers:`` entries use their config key so the identity survives display-name - changes; legacy ``custom_providers:`` entries have no key, so their normalized display name - remains the identity. - """ + """Stable ``custom:`` identity for a configured provider: keyed ``providers:`` entries use their + config key (survives display-name changes); legacy ``custom_providers:`` entries have no key, + so their normalized display name is the identity.""" identity = str(provider_key or "").strip() or str(display_name or "").strip() normalized = identity.lower().replace(" ", "-") return normalized if normalized.startswith("custom:") else f"custom:{normalized}" -def custom_provider_aliases( - display_name: str, - provider_key: str = "", -) -> frozenset[str]: +def custom_provider_aliases(display_name: str, provider_key: str = "") -> frozenset[str]: """Return every current and legacy identity accepted for one endpoint.""" aliases: set[str] = set() for value in (display_name, provider_key): @@ -591,175 +366,105 @@ def custom_provider_aliases( return frozenset(aliases) -def resolve_custom_provider( - name: str, - custom_providers: Optional[List[Dict[str, Any]]], -) -> Optional[ProviderDef]: - """Resolve a provider from the user's config.yaml ``custom_providers`` list.""" - if not custom_providers or not isinstance(custom_providers, list): - return None - +def resolve_custom_provider(name: str, custom_providers: Optional[List[Dict[str, Any]]]) -> Optional[ProviderDef]: + """Resolve a provider from the user's config.yaml ``custom_providers`` list. A stored bare + ``"custom"`` (corrupt state from a prior model-switch bug) falls back to the first valid entry + so existing configs self-heal.""" requested = (name or "").strip().lower() - if not requested: + if not requested or not custom_providers or not isinstance(custom_providers, list): return None - - # If the stored provider is the bare string "custom" (corrupt state - # from a prior model-switch bug), fall back to the first custom - # provider entry so existing configs self-heal. (GH #17478) - bare_custom_fallback = requested == "custom" first_valid: Optional[ProviderDef] = None - for entry in custom_providers: if not isinstance(entry, dict): continue - display_name = (entry.get("name") or "").strip() - api_url = ( - entry.get("base_url", "") - or entry.get("url", "") - or entry.get("api", "") - or "" - ).strip() + api_url = (entry.get("base_url", "") or entry.get("url", "") or entry.get("api", "") or "").strip() if not display_name or not api_url: continue - - key_env = (entry.get("key_env") or "").strip() provider_key = (entry.get("provider_key") or "").strip() - pdef = ProviderDef( - id=custom_provider_slug(display_name, provider_key), - name=display_name, - transport="openai_chat", - api_key_env_vars=(key_env,) if key_env else (), - base_url=api_url, - is_aggregator=False, - auth_type="api_key", - source="user-config", - ) - - # Stash the first valid entry for bare-"custom" fallback + pdef = _user_pdef(custom_provider_slug(display_name, provider_key), display_name, api_url, + (entry.get("key_env") or "").strip()) if first_valid is None: first_valid = pdef - if requested in custom_provider_aliases(display_name, provider_key): return pdef - - # Self-heal: bare "custom" matched nothing — return first valid entry - if bare_custom_fallback and first_valid: + if requested == "custom" and first_valid: return first_valid - return None -def resolve_provider_full( - name: str, - user_providers: Optional[Dict[str, Any]] = None, - custom_providers: Optional[List[Dict[str, Any]]] = None, -) -> Optional[ProviderDef]: - """Full resolution chain: built-in → models.dev → user config.""" +def _lossy_alias_registry_pdef(raw: str, canonical: str) -> Optional[ProviderDef]: + """Exact Hermes registry ids win over LOSSY alias collapsing (kimi-coding-cn must stay distinct + from kimi-coding instead of collapsing through the shared models.dev alias "kimi-for-coding"). + A collapse is lossy only when MULTIPLE registry providers normalize to the same canonical name; + single-entry rewrites ("copilot" -> "github-copilot") are correct routing and keep resolving + through the built-in chain so overlay transports apply.""" + try: + from hermes_cli.auth import PROVIDER_REGISTRY as _AUTH_PROVIDER_REGISTRY + _pcfg = _AUTH_PROVIDER_REGISTRY.get(raw) + if _pcfg is None: + return None + if sum(1 for _rid in _AUTH_PROVIDER_REGISTRY if normalize_provider(_rid) == canonical) > 1: + return ProviderDef(id=_pcfg.id, name=_pcfg.name, transport="openai_chat", + api_key_env_vars=tuple(_pcfg.api_key_env_vars or ()), base_url=_pcfg.inference_base_url or "", + source="hermes-auth-registry") + except Exception: + pass + return None + + +def _llamacpp_pdef() -> Optional[ProviderDef]: + """The llamacpp aliases are a real provider whenever the managed server (or a detected external + one) resolves — reachability is the credential. Without this rung model-switch rejected the very + provider the Local Models 'Use' flow writes to config.""" + try: + from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint + endpoint = resolve_llamacpp_endpoint(wait_for_boot_s=0) + except Exception: + endpoint = None + if not endpoint: + return None + return ProviderDef(id="llamacpp", name="Local", transport="openai_chat", api_key_env_vars=(), base_url=endpoint["base_url"], + source="local-runtime") + + +def resolve_provider_full(name: str, user_providers: Optional[Dict[str, Any]] = None, + custom_providers: Optional[List[Dict[str, Any]]] = None) -> Optional[ProviderDef]: + """Full resolution chain: user ``providers.`` -> lossy-alias registry id -> built-in + (models.dev + overlays) -> user providers (canonical, then raw) -> ``custom_providers`` -> + managed llamacpp -> models.dev directly. User-defined ``providers.`` is tried FIRST on + the raw (pre-alias) name: a configured ``providers.openai`` pointing at api.openai.com must not + be hijacked by the legacy "openai" -> "openrouter" alias.""" canonical = normalize_provider(name) raw = name.strip().lower() - - # 0. User-defined config providers win over the built-in alias table. - # A user who declares ``providers.`` in config.yaml has stated - # explicit intent for that name — it must not be hijacked by a legacy - # vendor alias (e.g. bare "openai" → "openrouter"). Resolve the raw - # name against user config FIRST so a configured ``providers.openai`` - # (pointing at api.openai.com) beats the alias that would otherwise - # silently route to OpenRouter. Only the raw (pre-alias) name is tried - # here; canonical/alias resolution still happens below. if user_providers: user_pdef = resolve_user_provider(raw, user_providers) if user_pdef is not None: return user_pdef - - # 0.5 Exact Hermes provider IDs must win over LOSSY alias collapsing. - # Example: kimi-coding-cn should stay distinct from kimi-coding instead of - # normalizing through the shared models.dev alias "kimi-for-coding". - # A collapse is lossy only when MULTIPLE distinct registry providers - # normalize to the same canonical name — resolving through the alias - # would then lose which one the caller meant. Single-entry rewrites - # (e.g. "copilot" → "github-copilot") are correct routing and must keep - # resolving through the built-in chain below so overlay transports apply. if canonical != raw: - try: - from hermes_cli.auth import PROVIDER_REGISTRY as _AUTH_PROVIDER_REGISTRY - _pcfg = _AUTH_PROVIDER_REGISTRY.get(raw) - if _pcfg is not None: - _collapsed_siblings = [ - _rid - for _rid in _AUTH_PROVIDER_REGISTRY - if normalize_provider(_rid) == canonical - ] - if len(_collapsed_siblings) > 1: - return ProviderDef( - id=_pcfg.id, - name=_pcfg.name, - transport="openai_chat", - api_key_env_vars=tuple(_pcfg.api_key_env_vars or ()), - base_url=_pcfg.inference_base_url or "", - source="hermes-auth-registry", - ) - except Exception: - pass - - # 1. Built-in (models.dev + overlays) + pdef = _lossy_alias_registry_pdef(raw, canonical) + if pdef is not None: + return pdef pdef = get_provider(canonical) if pdef is not None: return pdef - - # 2. User-defined providers from config if user_providers: - # Try canonical name - user_pdef = resolve_user_provider(canonical, user_providers) - if user_pdef is not None: - return user_pdef - # Try original name (in case alias didn't match) - user_pdef = resolve_user_provider(raw, user_providers) - if user_pdef is not None: - return user_pdef - - # 2b. Saved custom providers from config + for candidate in (canonical, raw): + user_pdef = resolve_user_provider(candidate, user_providers) + if user_pdef is not None: + return user_pdef custom_pdef = resolve_custom_provider(name, custom_providers) if custom_pdef is not None: return custom_pdef - - # 2c. Managed local runtime: the llamacpp aliases are a real provider - # whenever the managed server (or a detected external one) resolves — - # no credential and no providers: entry required, the credential is - # reachability. Without this rung the model-switch path rejected the - # very provider the Local Models 'Use' flow writes to config - # ("Unknown provider 'llamacpp'" from the desktop dropdown). if raw in ("llamacpp", "llama.cpp", "llama-cpp"): - try: - from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint - - endpoint = resolve_llamacpp_endpoint(wait_for_boot_s=0) - except Exception: - endpoint = None - if endpoint: - return ProviderDef( - id="llamacpp", - name="Local", - transport="openai_chat", - api_key_env_vars=(), - base_url=endpoint["base_url"], - source="local-runtime", - ) - - # 3. Try models.dev directly (for providers not in our ALIASES) + pdef = _llamacpp_pdef() + if pdef is not None: + return pdef try: - from agent.models_dev import get_provider_info as _mdev_provider - mdev_info = _mdev_provider(canonical) + mdev_info = _models_dev_info(canonical) if mdev_info is not None: - return ProviderDef( - id=canonical, - name=mdev_info.name, - transport="openai_chat", - api_key_env_vars=mdev_info.env, - base_url=mdev_info.api, - source="models.dev", - ) + return ProviderDef(id=canonical, name=mdev_info.name, transport="openai_chat", api_key_env_vars=mdev_info.env, + base_url=mdev_info.api, source="models.dev") except Exception: pass - return None diff --git a/hermes_cli/route_identity.py b/hermes_cli/route_identity.py index 9e7d45194b..6ada71c1d0 100644 --- a/hermes_cli/route_identity.py +++ b/hermes_cli/route_identity.py @@ -28,7 +28,6 @@ def normalize_route_base_url(base_url: Any) -> str: port = parsed.port except (TypeError, ValueError): return raw - route_host = parsed.netloc.rsplit("@", 1)[-1] if route_host.startswith("[") or ":" in host: host = f"[{host}]" @@ -36,68 +35,33 @@ def normalize_route_base_url(base_url: Any) -> str: host = f"{host}:{port}" if "@" in parsed.netloc: host = f"{parsed.netloc.rsplit('@', 1)[0]}@{host}" - path = parsed.path if path.endswith("/") and not had_query_delimiter: path = path[:-1] - normalized = urlunsplit((scheme, host, path, parsed.query, "")) if had_query_delimiter and not parsed.query: normalized += "?" return normalized -def should_clear_context_pin( - configured_model: Any, - active_model: Any, - configured_base_url: Any, - active_base_url: Any, - configured_provider: Any, - active_provider: Any, -) -> bool: +def should_clear_context_pin(configured_model: Any, active_model: Any, configured_base_url: Any, active_base_url: Any, + configured_provider: Any, active_provider: Any) -> bool: """True when a configured ``model.context_length`` pin no longer matches its runtime route. - Fail-closed: any error during route comparison returns ``True`` (drop the pin) so a stale window - never silently inflates the compression threshold. - """ + never silently inflates the compression threshold.""" configured_model = str(configured_model or "").strip() if configured_model and configured_model != str(active_model or "").strip(): return True try: from agent.agent_init import _context_route_mismatch - - return _context_route_mismatch( - configured_base_url, - active_base_url, - configured_provider, - active_provider, - ) + return _context_route_mismatch(configured_base_url, active_base_url, configured_provider, active_provider) except Exception: return True -async def should_clear_context_pin_async( - configured_model: Any, - active_model: Any, - configured_base_url: Any, - active_base_url: Any, - configured_provider: Any, - active_provider: Any, -) -> bool: - """Async wrapper for ``should_clear_context_pin``. - - Offloads the route comparison to a worker thread so async gateway handlers never run it on the - event loop — the resolution chain is cache-only (``allow_network=False``) but can still do cold- - start disk I/O. Shares all logic with the sync version — no code duplication. - """ +async def should_clear_context_pin_async(*args: Any) -> bool: + """``should_clear_context_pin`` on a worker thread so async gateway handlers never run it on the + event loop — the resolution chain is cache-only (``allow_network=False``) but can still do + cold-start disk I/O.""" import asyncio - - return await asyncio.to_thread( - should_clear_context_pin, - configured_model, - active_model, - configured_base_url, - active_base_url, - configured_provider, - active_provider, - ) + return await asyncio.to_thread(should_clear_context_pin, *args) diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index 38e6fe461c..7d70ceb8cc 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -1,11 +1,9 @@ -"""Shared runtime provider resolution for CLI, gateway, cron, and helpers. - -Layout: this module owns the resolution ORDER (:func:`resolve_runtime_provider`), the api_mode / -base_url helpers and the pool / OAuth / explicit paths. Custom-provider lookup lives in -:mod:`hermes_cli.runtime_provider_custom`; Azure Foundry, OpenRouter/bare-custom, Bedrock and -external-process builders in :mod:`hermes_cli.runtime_provider_backends`. Both are re-exported -here so ``hermes_cli.runtime_provider.`` imports and test patches keep working. -""" +"""Shared runtime provider resolution for CLI, gateway, cron, and helpers: the resolution ORDER +(:func:`resolve_runtime_provider`), api_mode / base_url helpers and the pool / OAuth / explicit paths. +Custom-provider lookup lives in :mod:`hermes_cli.runtime_provider_custom`; Azure Foundry, +OpenRouter/bare-custom, Bedrock and external-process builders in +:mod:`hermes_cli.runtime_provider_backends` — both re-exported here so +``hermes_cli.runtime_provider.`` imports and test patches keep working.""" from __future__ import annotations @@ -18,67 +16,44 @@ from urllib.parse import urlparse logger = logging.getLogger(__name__) from hermes_cli import auth as auth_mod -from agent.credential_pool import ( - CredentialPool, - PooledCredential, - credential_pool_matches_provider, - custom_provider_pool_key_candidates, # noqa: F401 — read via origin by runtime_provider_custom (patchable) +from agent.credential_pool import ( # custom_provider_pool_key_candidates is read via origin by runtime_provider_custom + CredentialPool, PooledCredential, credential_pool_matches_provider, custom_provider_pool_key_candidates, # noqa: F401 load_pool, ) from agent.secret_scope import get_secret as _get_secret -from hermes_cli.auth import ( - ACTUAL_LOCAL_NOAUTH_PLACEHOLDER, - AuthError, - DEFAULT_CODEX_BASE_URL, - DEFAULT_QWEN_BASE_URL, - DEFAULT_XAI_OAUTH_BASE_URL, - PROVIDER_REGISTRY, - _agent_key_is_usable, - _nous_inference_env_override, - format_auth_error, - resolve_provider, - resolve_nous_runtime_credentials, - resolve_codex_runtime_credentials, - resolve_xai_oauth_runtime_credentials, - resolve_qwen_runtime_credentials, - resolve_api_key_provider_credentials, - resolve_external_process_provider_credentials, # noqa: F401 — read via origin by runtime_provider_backends - has_usable_secret, - is_actual_local_base_url, - normalize_actual_base_url, +from hermes_cli.auth import ( # resolve_external_process_provider_credentials is read via origin by runtime_provider_backends + ACTUAL_LOCAL_NOAUTH_PLACEHOLDER, AuthError, DEFAULT_CODEX_BASE_URL, DEFAULT_QWEN_BASE_URL, DEFAULT_XAI_OAUTH_BASE_URL, + PROVIDER_REGISTRY, _agent_key_is_usable, _nous_inference_env_override, format_auth_error, resolve_provider, + resolve_nous_runtime_credentials, resolve_codex_runtime_credentials, resolve_xai_oauth_runtime_credentials, + resolve_qwen_runtime_credentials, resolve_api_key_provider_credentials, + resolve_external_process_provider_credentials, # noqa: F401 + has_usable_secret, is_actual_local_base_url, normalize_actual_base_url, ) from hermes_cli import config as _config_mod +from hermes_cli import models as _models # attribute access keeps ``hermes_cli.models.`` patches effective from hermes_constants import OPENROUTER_BASE_URL -from hermes_cli.providers import is_official_openai_host +from hermes_cli.providers import determine_api_mode, is_official_openai_host, nous_api_mode from utils import base_url_host_matches, base_url_hostname, env_int +# Late-bound delegates, deliberately NOT module-level from-imports: this module is often imported +# lazily, so its first import can happen while a test has ``hermes_cli.config.load_config`` patched +# — a from-import would bind the MagicMock permanently and poison every later caller. def load_config(): - """Late-bound delegate to :func:`hermes_cli.config.load_config`. - - Deliberately NOT a module-level from-import: this module is often imported lazily, so its - first import can happen while a test has ``hermes_cli.config.load_config`` patched — a - from-import would bind the MagicMock permanently and poison every later caller. - """ return _config_mod.load_config() def get_compatible_custom_providers(config=None): - """Late-bound delegate — see :func:`load_config` for why.""" return _config_mod.get_compatible_custom_providers(config) def normalize_extra_headers(value): - """Late-bound delegate — see :func:`load_config` for why.""" return _config_mod.normalize_extra_headers(value) def _getenv(name: str, default: str = "") -> str: - """Profile-scoped ``os.getenv`` for credential/provider reads. - - Identical to ``os.getenv`` when multiplexing is off; scope-aware (fail-closed on an unscoped - read) when on. Genuinely-global vars are handled inside ``get_secret``. - """ + """Profile-scoped ``os.getenv`` for credential/provider reads: identical to ``os.getenv`` when + multiplexing is off; scope-aware (fail-closed on an unscoped read) when on.""" val = _get_secret(name, default) return val if val is not None else default @@ -96,13 +71,11 @@ def _resolves_to_custom(name: str) -> bool: def _config_base_url_trustworthy_for_bare_custom(cfg_base_url: str, cfg_provider: str) -> bool: - """Whether ``model.base_url`` may back bare ``custom`` runtime resolution. - - The model picker can select Custom while ``model.provider`` still reflects a previous - provider. Non-loopback URLs are rejected unless the YAML provider is already ``custom`` or a - local-server alias (ollama/vllm/llamacpp — else a legit LAN ollama endpoint silently falls - through to OpenRouter), so a stale OpenRouter/Z.ai base_url cannot hijack local sessions. - """ + """Whether ``model.base_url`` may back bare ``custom`` runtime resolution. The picker can select + Custom while ``model.provider`` still names a previous provider, so non-loopback URLs are rejected + unless the YAML provider is already ``custom`` or a local-server alias (ollama/vllm/llamacpp — + else a legit LAN ollama endpoint falls through to OpenRouter): a stale OpenRouter/Z.ai base_url + cannot hijack local sessions.""" cfg_provider_norm = (cfg_provider or "").strip().lower() bu = (cfg_base_url or "").strip() if not bu: @@ -120,31 +93,20 @@ def _config_base_url_trustworthy_for_bare_custom(cfg_base_url: str, cfg_provider # so the runtime resolver stays in lockstep: api.meta.ai — prompt caching only on Responses; # api.router.com — /v1/chat/completions is a minimal shim; api.anthropic.com — native Messages. _HOST_MANDATED_API_MODES = { - "api.x.ai": "codex_responses", - "api.meta.ai": "codex_responses", - "api.actual.inc": "codex_responses", - "api.router.com": "codex_responses", - "api.anthropic.com": "anthropic_messages", + "api.x.ai": "codex_responses", "api.meta.ai": "codex_responses", "api.actual.inc": "codex_responses", + "api.router.com": "codex_responses", "api.anthropic.com": "anthropic_messages", } -_VALID_API_MODES = { - "chat_completions", - "codex_responses", - "anthropic_messages", - "bedrock_converse", - # Opt-in: hand the whole turn to a `codex app-server` subprocess (Codex's own tool runtime). - # Gated on `model.openai_runtime == "codex_app_server"` AND provider in {openai, openai-codex}. - "codex_app_server", -} +# codex_app_server is opt-in: hand the whole turn to a `codex app-server` subprocess (Codex's own +# tool runtime), gated on `model.openai_runtime == "codex_app_server"` AND provider in {openai, openai-codex}. +_VALID_API_MODES = {"chat_completions", "codex_responses", "anthropic_messages", "bedrock_converse", "codex_app_server"} def _detect_api_mode_for_url(base_url: str) -> Optional[str]: - """Auto-detect api_mode from the resolved base URL, or None. - - Exact-hostname matches reject lookalike subdomains (api.anthropic.com.attacker.test) and - path-segment spoofing (proxy.test/api.anthropic.com/v1). Official OpenAI hosts (incl. the - data-residency us./eu. regional hosts) need Responses for GPT-5.x tool calls with reasoning. - """ + """Auto-detect api_mode from the resolved base URL, or None. Exact-hostname matches reject + lookalike subdomains (api.anthropic.com.attacker.test) and path-segment spoofing + (proxy.test/api.anthropic.com/v1). Official OpenAI hosts (incl. us./eu. data-residency hosts) + need Responses for GPT-5.x tool calls with reasoning.""" normalized = (base_url or "").strip().lower().rstrip("/") hostname = base_url_hostname(base_url) mandated = _HOST_MANDATED_API_MODES.get(hostname) @@ -165,9 +127,7 @@ def _parse_api_mode(raw: Any) -> Optional[str]: ``anthropic``, ``responses``, …) are canonicalized first so old configs keep their transport instead of silently falling through to hostname-based detection.""" if isinstance(raw, str): - from hermes_cli.config import _canonical_api_mode - - normalized = _canonical_api_mode(raw).lower() + normalized = _config_mod._canonical_api_mode(raw).lower() if normalized in _VALID_API_MODES: return normalized return None @@ -176,14 +136,9 @@ def _parse_api_mode(raw: Any) -> Optional[str]: def _fallback_api_mode(provider: str, base_url: str, model: str = "") -> str: """api_mode when no explicit/persisted mode applies: URL detection (host-mandated wire shapes) first, then the transport the provider overlay declares via ``providers.determine_api_mode`` - (which was never consulted before — ``openai-api`` pointed at us.api.openai.com 400'd on every - tool call), then ``chat_completions``.""" - detected = _detect_api_mode_for_url(base_url) - if detected: - return detected - from hermes_cli.providers import determine_api_mode - - return determine_api_mode(provider, base_url, model) or "chat_completions" + (``openai-api`` pointed at us.api.openai.com 400'd on every tool call without it), then + ``chat_completions``.""" + return _detect_api_mode_for_url(base_url) or determine_api_mode(provider, base_url, model) or "chat_completions" def _resolve_plain_custom_api_mode(model_cfg: Dict[str, Any], base_url: str) -> str: @@ -192,10 +147,7 @@ def _resolve_plain_custom_api_mode(model_cfg: Dict[str, Any], base_url: str) -> configured_mode = _parse_api_mode(model_cfg.get("api_mode")) detected_mode = _detect_api_mode_for_url(base_url) if configured_mode == "codex_responses" and detected_mode != "codex_responses": - logger.info( - "Ignoring persisted custom api_mode=codex_responses for non-OpenAI endpoint %s", - base_url or "(unknown)", - ) + logger.info("Ignoring persisted custom api_mode=codex_responses for non-OpenAI endpoint %s", base_url or "(unknown)") configured_mode = None return configured_mode or detected_mode or "chat_completions" @@ -212,19 +164,29 @@ def _provider_supports_explicit_api_mode(provider: Optional[str], configured_pro return normalized_configured == normalized_provider -def _copilot_runtime_api_mode(model_cfg: Dict[str, Any], api_key: str, *, target_model: Optional[str] = None) -> str: +def _configured_api_mode(provider: str, model_cfg: Dict[str, Any]) -> Optional[str]: + """Persisted ``model.api_mode`` when valid and recorded for this provider, else None.""" configured_mode = _parse_api_mode(model_cfg.get("api_mode")) - if configured_mode and _provider_supports_explicit_api_mode("copilot", _cfg_provider(model_cfg)): + if configured_mode and _provider_supports_explicit_api_mode(provider, _cfg_provider(model_cfg)): + return configured_mode + return None + + +def _effective_model(model_cfg: Dict[str, Any], target_model: Optional[str]) -> str: + """The caller's target model (e.g. /model switch) beats the persisted default, else api_mode + is computed from a stale default.""" + return target_model or model_cfg.get("default") or "" + + +def _copilot_runtime_api_mode(model_cfg: Dict[str, Any], api_key: str, *, target_model: Optional[str] = None) -> str: + configured_mode = _configured_api_mode("copilot", model_cfg) + if configured_mode: return configured_mode # Use the model being resolved, not the persisted default: a Claude MoA slot inheriting # codex_responses from a GPT-5 default fails with "model ... does not support Responses API". - model_name = str(target_model or model_cfg.get("default") or "").strip() - if not model_name: - return "chat_completions" + model_name = str(_effective_model(model_cfg, target_model)).strip() try: - from hermes_cli.models import copilot_model_api_mode - - return copilot_model_api_mode(model_name, api_key=api_key) + return _models.copilot_model_api_mode(model_name, api_key=api_key) if model_name else "chat_completions" except Exception: return "chat_completions" @@ -235,38 +197,23 @@ def _azure_inferred_api_mode(effective_model: str, api_mode: str) -> str: if not effective_model or api_mode == "anthropic_messages": return api_mode try: - from hermes_cli.models import azure_foundry_model_api_mode - - inferred = azure_foundry_model_api_mode(effective_model) + return _models.azure_foundry_model_api_mode(effective_model) or api_mode except Exception: - inferred = None - return inferred or api_mode + return api_mode -def _configured_or_fallback_api_mode( - provider: str, model_cfg: Dict[str, Any], base_url: str, effective_model: Any, *, opencode_by_model: bool -) -> str: +def _configured_or_fallback_api_mode(provider: str, model_cfg: Dict[str, Any], base_url: str, effective_model: Any, *, + opencode_by_model: bool) -> str: """Persisted ``model.api_mode`` when it belongs to this provider, else URL/transport fallback. - OpenCode Zen/Go serve both anthropic_messages and chat_completions models, so (when - ``opencode_by_model``) their mode is always re-derived from the effective model. - """ - if opencode_by_model: - from hermes_cli.models import opencode_provider_family - - if opencode_provider_family(provider) is not None: - from hermes_cli.models import opencode_model_api_mode - - return opencode_model_api_mode(provider, effective_model) - configured_mode = _parse_api_mode(model_cfg.get("api_mode")) - if configured_mode and _provider_supports_explicit_api_mode(provider, _cfg_provider(model_cfg)): - return configured_mode - return _fallback_api_mode(provider, base_url, effective_model) + ``opencode_by_model``) their mode is always re-derived from the effective model.""" + if opencode_by_model and _models.opencode_provider_family(provider) is not None: + return _models.opencode_model_api_mode(provider, effective_model) + return _configured_api_mode(provider, model_cfg) or _fallback_api_mode(provider, base_url, effective_model) -def _api_key_provider_api_mode( - provider: str, model_cfg: Dict[str, Any], api_key: str, base_url: str, effective_model: Any, *, opencode_by_model: bool -) -> str: +def _api_key_provider_api_mode(provider: str, model_cfg: Dict[str, Any], api_key: str, base_url: str, effective_model: Any, *, + opencode_by_model: bool) -> str: """api_mode for a registry ``api_key`` provider (explicit and env/config paths).""" if provider == "copilot": return _copilot_runtime_api_mode(model_cfg, api_key, target_model=effective_model) @@ -278,11 +225,7 @@ def _api_key_provider_api_mode( def _maybe_apply_codex_app_server_runtime(*, provider: str, api_mode: str, model_cfg: Optional[Dict[str, Any]]) -> str: """Opt-in rewrite to "codex_app_server" via ``model.openai_runtime``; only ``openai`` / ``openai-codex`` are eligible. No-op when unset, "auto", or empty.""" - if ( - model_cfg - and provider in {"openai", "openai-codex"} - and str(model_cfg.get("openai_runtime") or "").strip().lower() == "codex_app_server" - ): + if model_cfg and provider in {"openai", "openai-codex"} and str(model_cfg.get("openai_runtime") or "").strip().lower() == "codex_app_server": return "codex_app_server" return api_mode @@ -290,10 +233,8 @@ def _maybe_apply_codex_app_server_runtime(*, provider: str, api_mode: str, model # ── base_url / credential helpers ────────────────────────────────────────────────────────── _ANTHROPIC_DEFAULT_BASE_URL = "https://api.anthropic.com" -_NO_ANTHROPIC_CREDENTIALS_MSG = ( - "No Anthropic credentials found. Set ANTHROPIC_TOKEN or ANTHROPIC_API_KEY, " - "run 'claude setup-token', or authenticate with 'claude /login'." -) +_NO_ANTHROPIC_CREDENTIALS_MSG = ("No Anthropic credentials found. Set ANTHROPIC_TOKEN or ANTHROPIC_API_KEY, " + "run 'claude setup-token', or authenticate with 'claude /login'.") def _runtime(provider: str, api_mode: str, base_url: Any, api_key: Any, **extra: Any) -> Dict[str, Any]: @@ -333,6 +274,14 @@ def _anthropic_cfg_base_url(model_cfg: Dict[str, Any]) -> str: return cfg_base_url if _anthropic_base_url_override_ok(cfg_base_url) else "" +def _anthropic_token_or_raise() -> str: + from agent.anthropic_adapter import resolve_anthropic_token + token = resolve_anthropic_token() + if not token: + raise AuthError(_NO_ANTHROPIC_CREDENTIALS_MSG) + return token + + def _host_derived_api_key(base_url: str) -> str: """``_API_KEY`` from the env, vendor = registrable hostname label (``api.deepseek.com`` → ``deepseek``). Lookalike hosts pick the ATTACKER's label (api.deepseek.com.attacker.test → @@ -344,9 +293,7 @@ def _host_derived_api_key(base_url: str) -> str: labels = [lbl for lbl in hostname.split(".") if lbl] while labels and labels[0] in ("api", "www"): labels.pop(0) - if len(labels) < 2: - return "" - sanitized = "".join(ch if ch.isalnum() else "_" for ch in labels[-2]).upper() + sanitized = "".join(ch if ch.isalnum() else "_" for ch in labels[-2]).upper() if len(labels) >= 2 else "" if not sanitized or not sanitized[0].isalpha() or sanitized in ("OPENAI", "OPENROUTER", "OLLAMA"): return "" return (_getenv(f"{sanitized}_API_KEY", "") or "").strip() @@ -354,11 +301,9 @@ def _host_derived_api_key(base_url: str) -> str: def _host_gated_env_key_candidates(base_url: str, *, ollama: bool) -> list: """Env API keys gated on their authoritative hosts, then the host-derived ``_API_KEY``. - Sending OPENAI/OPENROUTER/OLLAMA keys to an unrelated endpoint leaks credentials - (GHSA-76xc-57q6-vm5m); match on HOST, not substring. ``_host_derived_api_key`` skips OLLAMA, - so callers that want it opt in via ``ollama``. - """ + (GHSA-76xc-57q6-vm5m); match on HOST, not substring. ``_host_derived_api_key`` skips OLLAMA, so + callers that want it opt in via ``ollama``.""" is_openai = base_url_host_matches(base_url, "openai.com") or base_url_host_matches(base_url, "openai.azure.com") candidates = [] if ollama: @@ -379,19 +324,8 @@ def _pool_entry_base_url(entry: Any) -> str: return getattr(entry, "runtime_base_url", None) or getattr(entry, "base_url", None) or "" -def _nous_pool_state(entry: Any) -> Dict[str, Any]: - return {k: getattr(entry, k, None) for k in ("agent_key", "agent_key_expires_at", "scope")} - - -def _registry_base_url(provider: str) -> str: - pconfig = PROVIDER_REGISTRY.get(provider) - return pconfig.inference_base_url if pconfig else "" - - -def _nous_inference_base_url_override() -> str: - """Trusted ``NOUS_INFERENCE_BASE_URL`` override (bypasses the network host allowlist); one - normalization path via ``auth._nous_inference_env_override``.""" - return _nous_inference_env_override() or "" +def _nous_entry_key_usable(entry: Any, min_ttl: int) -> bool: + return _agent_key_is_usable({k: getattr(entry, k, None) for k in ("agent_key", "agent_key_expires_at", "scope")}, min_ttl) def _nous_min_key_ttl() -> int: @@ -402,27 +336,12 @@ def _resolve_nous_creds() -> Dict[str, Any]: return resolve_nous_runtime_credentials(timeout_seconds=float(_getenv("HERMES_NOUS_TIMEOUT_SECONDS", "15"))) -def _nous_api_mode(model: str) -> str: - from hermes_cli.providers import nous_api_mode - - return nous_api_mode(model) - - -def _normalize_opencode_runtime_base_url(provider: str, api_mode: str, base_url: str) -> str: - """OpenCode base URLs end with /v1 for OpenAI-compatible models, but the Anthropic SDK prepends - its own /v1/messages: strip /v1 for anthropic_messages, re-append otherwise.""" - from hermes_cli.models import opencode_provider_family - - if opencode_provider_family(provider) is None: - return base_url - from hermes_cli.models import normalize_opencode_base_url - - return normalize_opencode_base_url(provider, api_mode, base_url) - - def _finalize_base_url(provider: str, api_mode: str, base_url: str) -> str: - """Shared tail for pool-entry and api-key paths: OpenCode /v1 rule, then LM Studio normalization.""" - base_url = _normalize_opencode_runtime_base_url(provider, api_mode, base_url) + """Shared tail for pool-entry and api-key paths: OpenCode /v1 rule (OpenCode URLs end with /v1 + for OpenAI-compatible models but the Anthropic SDK prepends its own /v1/messages — strip for + anthropic_messages, re-append otherwise), then LM Studio normalization.""" + if _models.opencode_provider_family(provider) is not None: + base_url = _models.normalize_opencode_base_url(provider, api_mode, base_url) if provider == "lmstudio": base_url = auth_mod._normalize_lmstudio_runtime_base_url(base_url) return base_url @@ -437,11 +356,8 @@ def _auto_detect_local_model(base_url: str) -> str: return "" try: import requests - url = base_url.rstrip("/") - if not url.endswith("/v1"): - url += "/v1" - resp = requests.get(url + "/models", timeout=(2, 3)) + resp = requests.get((url if url.endswith("/v1") else url + "/v1") + "/models", timeout=(2, 3)) if resp.ok: models = resp.json().get("data", []) if len(models) == 1 and models[0].get("id", ""): @@ -465,9 +381,7 @@ def _get_model_config() -> Dict[str, Any]: cfg["default"] = cfg["model"] _default = cfg.get("default") if isinstance(_default, dict): - from hermes_cli.config import split_model_config_default - - cfg_model, cfg_provider = split_model_config_default(_default) + cfg_model, cfg_provider = _config_mod.split_model_config_default(_default) cfg_provider = cfg_provider or str(model_cfg.get("provider") or "") cfg["default"] = cfg_model if cfg_provider and not cfg.get("provider"): @@ -496,31 +410,15 @@ def resolve_requested_provider(requested: Optional[str] = None) -> str: # ── extracted collaborators (re-exported; see module docstring) ──────────────────────────── from hermes_cli.runtime_provider_custom import ( # noqa: E402,F401 - _apply_custom_provider_extras, - _custom_provider_request_overrides, - _filter_capabilities, - _find_custom_identity, - _get_named_custom_provider, - _lift_common_custom_fields, - _lift_extra_headers, - _lift_max_output_tokens, - _lift_model_capabilities, - _normalize_base_url_for_match, - _normalize_custom_provider_name, - _resolve_named_custom_runtime, - _try_resolve_from_custom_pool, - canonical_custom_identity, - find_custom_provider_identity, - find_custom_provider_identity_by_model, - has_named_custom_provider, - is_routable_provider, + _apply_custom_provider_extras, _custom_provider_request_overrides, _filter_capabilities, _find_custom_identity, + _get_named_custom_provider, _lift_common_custom_fields, _lift_extra_headers, _lift_max_output_tokens, + _lift_model_capabilities, _normalize_base_url_for_match, _normalize_custom_provider_name, _resolve_named_custom_runtime, + _try_resolve_from_custom_pool, canonical_custom_identity, find_custom_provider_identity, + find_custom_provider_identity_by_model, has_named_custom_provider, is_routable_provider, ) from hermes_cli.runtime_provider_backends import ( # noqa: E402,F401 - _is_external_process_provider, - _resolve_azure_foundry_runtime, - _resolve_bedrock_runtime, - _resolve_external_process_runtime, - _resolve_openrouter_runtime, + _is_external_process_provider, _resolve_azure_foundry_runtime, _resolve_bedrock_runtime, + _resolve_external_process_runtime, _resolve_openrouter_runtime, ) @@ -531,64 +429,50 @@ from hermes_cli.runtime_provider_backends import ( # noqa: E402,F401 # are valid only against the Anthropic Messages endpoint, so a stale model.api_mode from a prior # OpenAI-compatible provider is never honoured for it (it would 404 on /chat/completions). _POOL_ENTRY_SIMPLE_MODES: Dict[str, tuple] = { - "openai-codex": ("codex_responses", DEFAULT_CODEX_BASE_URL), - "xai-oauth": ("codex_responses", DEFAULT_XAI_OAUTH_BASE_URL), - "qwen-oauth": ("chat_completions", DEFAULT_QWEN_BASE_URL), - "minimax-oauth": ("anthropic_messages", lambda: _registry_base_url("minimax-oauth")), - "openrouter": ("chat_completions", OPENROUTER_BASE_URL), + "openai-codex": ("codex_responses", DEFAULT_CODEX_BASE_URL), "xai-oauth": ("codex_responses", DEFAULT_XAI_OAUTH_BASE_URL), + "qwen-oauth": ("chat_completions", DEFAULT_QWEN_BASE_URL), "openrouter": ("chat_completions", OPENROUTER_BASE_URL), + "minimax-oauth": ("anthropic_messages", lambda: getattr(PROVIDER_REGISTRY.get("minimax-oauth"), "inference_base_url", "")), "xai": ("codex_responses", ""), } -def _resolve_runtime_from_pool_entry( - *, - provider: str, - entry: PooledCredential, - requested_provider: str, - model_cfg: Optional[Dict[str, Any]] = None, - pool: Optional[CredentialPool] = None, - target_model: Optional[str] = None, -) -> Dict[str, Any]: - model_cfg = model_cfg or _get_model_config() - # The caller's target model (e.g. /model switch) beats the persisted default, else api_mode is - # computed from a stale default. - effective_model = target_model or model_cfg.get("default") or "" - base_url = _pool_entry_base_url(entry).rstrip("/") - api_key = _pool_entry_api_key(entry) +def _pool_entry_mode_and_url(provider, entry, model_cfg, effective_model, base_url) -> tuple: + """(api_mode, base_url) for a pool entry of ``provider``.""" if provider in _POOL_ENTRY_SIMPLE_MODES: api_mode, default_url = _POOL_ENTRY_SIMPLE_MODES[provider] - base_url = base_url or (default_url() if callable(default_url) else default_url) - elif provider == "anthropic": - api_mode = "anthropic_messages" - base_url = _anthropic_cfg_base_url(model_cfg) or base_url or _ANTHROPIC_DEFAULT_BASE_URL - elif provider == "nous": - api_mode = _nous_api_mode(effective_model) - base_url = _nous_inference_base_url_override() or base_url - elif provider == "copilot": + return api_mode, base_url or (default_url() if callable(default_url) else default_url) + if provider == "anthropic": + return "anthropic_messages", _anthropic_cfg_base_url(model_cfg) or base_url or _ANTHROPIC_DEFAULT_BASE_URL + if provider == "nous": + return nous_api_mode(effective_model), (_nous_inference_env_override() or "") or base_url + if provider == "copilot": api_mode = _copilot_runtime_api_mode(model_cfg, getattr(entry, "runtime_api_key", ""), target_model=effective_model) - base_url = base_url or PROVIDER_REGISTRY["copilot"].inference_base_url - elif provider == "azure-foundry": + return api_mode, base_url or PROVIDER_REGISTRY["copilot"].inference_base_url + if provider == "azure-foundry": api_mode = "chat_completions" if _cfg_provider(model_cfg) == "azure-foundry": base_url = _config_base_url_for_provider(model_cfg, "azure-foundry") or base_url api_mode = _parse_api_mode(model_cfg.get("api_mode")) or api_mode api_mode = _azure_inferred_api_mode(effective_model, api_mode) - if api_mode == "anthropic_messages": - base_url = re.sub(r"/v1/?$", "", base_url) - else: - # Honour model.base_url only when the pool entry carries no explicit base_url (i.e. it - # fell back to the registry default). Env var overrides win. - pconfig = PROVIDER_REGISTRY.get(provider) - if pconfig and base_url.rstrip("/") == pconfig.inference_base_url.rstrip("/"): - base_url = _config_base_url_for_provider(model_cfg, provider) or base_url - api_mode = _configured_or_fallback_api_mode(provider, model_cfg, base_url, effective_model, opencode_by_model=True) + return api_mode, (re.sub(r"/v1/?$", "", base_url) if api_mode == "anthropic_messages" else base_url) + # Honour model.base_url only when the pool entry carries no explicit base_url (i.e. it fell + # back to the registry default). Env var overrides win. + pconfig = PROVIDER_REGISTRY.get(provider) + if pconfig and base_url.rstrip("/") == pconfig.inference_base_url.rstrip("/"): + base_url = _config_base_url_for_provider(model_cfg, provider) or base_url + return _configured_or_fallback_api_mode(provider, model_cfg, base_url, effective_model, opencode_by_model=True), base_url + +def _resolve_runtime_from_pool_entry(*, provider: str, entry: PooledCredential, requested_provider: str, + model_cfg: Optional[Dict[str, Any]] = None, pool: Optional[CredentialPool] = None, + target_model: Optional[str] = None) -> Dict[str, Any]: + model_cfg = model_cfg or _get_model_config() + api_mode, base_url = _pool_entry_mode_and_url(provider, entry, model_cfg, _effective_model(model_cfg, target_model), + _pool_entry_base_url(entry).rstrip("/")) base_url = _finalize_base_url(provider, api_mode, base_url) api_mode = _maybe_apply_codex_app_server_runtime(provider=provider, api_mode=api_mode, model_cfg=model_cfg) - return _runtime( - provider, api_mode, base_url, api_key, - source=getattr(entry, "source", "pool"), credential_pool=pool, requested_provider=requested_provider, - ) + return _runtime(provider, api_mode, base_url, _pool_entry_api_key(entry), source=getattr(entry, "source", "pool"), + credential_pool=pool, requested_provider=requested_provider) def _openrouter_should_use_pool(requested_provider, model_cfg, explicit_api_key, explicit_base_url) -> bool: @@ -597,8 +481,7 @@ def _openrouter_should_use_pool(requested_provider, model_cfg, explicit_api_key, env_openai_base_url = _getenv("OPENAI_BASE_URL", "").strip() env_openrouter_base_url = _getenv("OPENROUTER_BASE_URL", "").strip() has_custom_endpoint = bool(explicit_base_url or env_openai_base_url or env_openrouter_base_url) or bool( - cfg_base_url and _cfg_provider(model_cfg) in {"auto", "custom"} - ) + cfg_base_url and _cfg_provider(model_cfg) in {"auto", "custom"}) return requested_provider in {"openrouter", "auto"} and not has_custom_endpoint and not bool(explicit_api_key or explicit_base_url) @@ -607,7 +490,7 @@ def _refresh_nous_pool_entry(pool: CredentialPool, entry: Any, pool_api_key: str selection (avoids network calls in `hermes auth list`); refresh here before falling back to singleton auth resolution. Returns (entry, pool_api_key) — key "" when still unusable.""" min_ttl = _nous_min_key_ttl() - if _agent_key_is_usable(_nous_pool_state(entry), min_ttl): + if _nous_entry_key_usable(entry, min_ttl): return entry, pool_api_key logger.debug("Nous pool entry agent_key expired/missing, refreshing selected pool entry") try: @@ -616,21 +499,18 @@ def _refresh_nous_pool_entry(pool: CredentialPool, entry: Any, pool_api_key: str logger.debug("Nous pool entry refresh failed: %s", exc) refreshed = None if refreshed is not None: - entry = refreshed - pool_api_key = _pool_entry_api_key(entry) - if not pool_api_key or not _agent_key_is_usable(_nous_pool_state(entry), min_ttl): + entry, pool_api_key = refreshed, _pool_entry_api_key(refreshed) + if not pool_api_key or not _nous_entry_key_usable(entry, min_ttl): logger.debug("Nous pool entry agent_key still unavailable, falling through to runtime resolution") pool_api_key = "" return entry, pool_api_key -def _resolve_from_pool( - provider: str, requested_provider: str, model_cfg: Dict[str, Any], explicit_api_key, explicit_base_url, target_model -) -> Optional[Dict[str, Any]]: +def _resolve_from_pool(provider: str, requested_provider: str, model_cfg: Dict[str, Any], explicit_api_key, explicit_base_url, + target_model) -> Optional[Dict[str, Any]]: """Runtime from the provider's credential pool, or None to continue down the ladder.""" - should_use_pool = provider != "openrouter" or _openrouter_should_use_pool( - requested_provider, model_cfg, explicit_api_key, explicit_base_url - ) + should_use_pool = provider != "openrouter" or _openrouter_should_use_pool(requested_provider, model_cfg, explicit_api_key, + explicit_base_url) try: pool = load_pool(provider) if should_use_pool else None except Exception: @@ -638,18 +518,14 @@ def _resolve_from_pool( if not (pool and pool.has_credentials()): return None entry = pool.select() - pool_api_key = _pool_entry_api_key(entry) if entry is not None else "" - if provider == "nous" and entry is not None: + if entry is None: + return None + pool_api_key = _pool_entry_api_key(entry) + if provider == "nous": entry, pool_api_key = _refresh_nous_pool_entry(pool, entry, pool_api_key) - if ( - entry is not None - and pool_api_key - and credential_pool_matches_provider(pool, provider, base_url=_pool_entry_base_url(entry)) - ): - return _resolve_runtime_from_pool_entry( - provider=provider, entry=entry, requested_provider=requested_provider, - model_cfg=model_cfg, pool=pool, target_model=target_model, - ) + if pool_api_key and credential_pool_matches_provider(pool, provider, base_url=_pool_entry_base_url(entry)): + return _resolve_runtime_from_pool_entry(provider=provider, entry=entry, requested_provider=requested_provider, + model_cfg=model_cfg, pool=pool, target_model=target_model) return None @@ -658,57 +534,54 @@ def _resolve_from_pool( def _explicit_anthropic(requested_provider, model_cfg, api_key, base_url, target_model): base_url = base_url or _anthropic_cfg_base_url(model_cfg) or _ANTHROPIC_DEFAULT_BASE_URL - if not api_key: - from agent.anthropic_adapter import resolve_anthropic_token - - api_key = resolve_anthropic_token() - if not api_key: - raise AuthError(_NO_ANTHROPIC_CREDENTIALS_MSG) + api_key = api_key or _anthropic_token_or_raise() return _runtime("anthropic", "anthropic_messages", base_url, api_key, source="explicit", requested_provider=requested_provider) +def _creds_fallback(api_key, explicit_base_url, base_url, expiry, expiry_key, resolve): + """When no explicit key was given, take api_key / expiry / base_url from stored credentials + (an explicit --base-url still wins over the stored one).""" + if api_key: + return api_key, base_url, expiry + creds = resolve() + return creds.get("api_key", ""), explicit_base_url or creds.get("base_url", "").rstrip("/") or base_url, creds.get(expiry_key) + + def _explicit_codex(requested_provider, model_cfg, api_key, explicit_base_url, target_model): - base_url = explicit_base_url or DEFAULT_CODEX_BASE_URL - last_refresh = None - if not api_key: - creds = resolve_codex_runtime_credentials() - api_key = creds.get("api_key", "") - last_refresh = creds.get("last_refresh") - base_url = explicit_base_url or creds.get("base_url", "").rstrip("/") or base_url - return _runtime( - "openai-codex", "codex_responses", base_url, api_key, - source="explicit", last_refresh=last_refresh, requested_provider=requested_provider, - ) + api_key, base_url, last_refresh = _creds_fallback(api_key, explicit_base_url, explicit_base_url or DEFAULT_CODEX_BASE_URL, + None, "last_refresh", resolve_codex_runtime_credentials) + return _runtime("openai-codex", "codex_responses", base_url, api_key, source="explicit", last_refresh=last_refresh, + requested_provider=requested_provider) def _explicit_nous(requested_provider, model_cfg, api_key, explicit_base_url, target_model): state = auth_mod.get_provider_auth_state("nous") or {} - base_url = ( - explicit_base_url - or _nous_inference_base_url_override() - or str(state.get("inference_base_url") or auth_mod.DEFAULT_NOUS_INFERENCE_URL).strip().rstrip("/") - ) + base_url = (explicit_base_url or _nous_inference_env_override() + or str(state.get("inference_base_url") or auth_mod.DEFAULT_NOUS_INFERENCE_URL).strip().rstrip("/")) # The agent_key compatibility field is used for inference only when it holds a NAS invoke JWT; # raw OAuth access_token fallback is handled by resolve_nous_runtime_credentials(). - api_key = api_key or ( - str(state.get("agent_key") or "").strip() if _agent_key_is_usable(state, _nous_min_key_ttl()) else "" - ) - expires_at = state.get("agent_key_expires_at") or state.get("expires_at") - if not api_key: - creds = _resolve_nous_creds() - api_key = creds.get("api_key", "") - expires_at = creds.get("expires_at") - base_url = explicit_base_url or creds.get("base_url", "").rstrip("/") or base_url - return _runtime( - "nous", _nous_api_mode(target_model or model_cfg.get("default") or ""), base_url, api_key, - source="explicit", expires_at=expires_at, requested_provider=requested_provider, - ) + api_key = api_key or (str(state.get("agent_key") or "").strip() if _agent_key_is_usable(state, _nous_min_key_ttl()) else "") + api_key, base_url, expires_at = _creds_fallback(api_key, explicit_base_url, base_url, + state.get("agent_key_expires_at") or state.get("expires_at"), "expires_at", + _resolve_nous_creds) + return _runtime("nous", nous_api_mode(_effective_model(model_cfg, target_model)), base_url, api_key, source="explicit", + expires_at=expires_at, requested_provider=requested_provider) def _explicit_azure_foundry(requested_provider, model_cfg, api_key, base_url, target_model): - return _resolve_azure_foundry_runtime( - requested_provider=requested_provider, model_cfg=model_cfg, explicit_api_key=api_key, explicit_base_url=base_url - ) + return _resolve_azure_foundry_runtime(requested_provider=requested_provider, model_cfg=model_cfg, explicit_api_key=api_key, + explicit_base_url=base_url) + + +def _actual_local_key(provider: str, api_key: str, base_url: str) -> str: + """Actual Computer's loopback daemon speaks a no-auth local API — substitute the placeholder key.""" + if provider == "actual" and not api_key and is_actual_local_base_url(base_url): + return ACTUAL_LOCAL_NOAUTH_PLACEHOLDER + return api_key + + +def _actual_url(provider: str, base_url: str) -> str: + return normalize_actual_base_url(base_url) if provider == "actual" else base_url def _explicit_api_key_provider(provider, pconfig, requested_provider, model_cfg, api_key, base_url, target_model): @@ -718,42 +591,29 @@ def _explicit_api_key_provider(provider, pconfig, requested_provider, model_cfg, else: env_url = _getenv(pconfig.base_url_env_var, "").strip().rstrip("/") if pconfig.base_url_env_var else "" base_url = env_url or pconfig.inference_base_url - if provider == "actual": - base_url = normalize_actual_base_url(base_url) + base_url = _actual_url(provider, base_url) if not api_key: creds = resolve_api_key_provider_credentials(provider) api_key = creds.get("api_key", "") if not base_url: - base_url = creds.get("base_url", "").rstrip("/") - if provider == "actual": - base_url = normalize_actual_base_url(base_url) - api_mode = _api_key_provider_api_mode( - provider, model_cfg, api_key, base_url, target_model or model_cfg.get("default", ""), opencode_by_model=False - ) - if provider == "actual" and not api_key and is_actual_local_base_url(base_url): - api_key = ACTUAL_LOCAL_NOAUTH_PLACEHOLDER + base_url = _actual_url(provider, creds.get("base_url", "").rstrip("/")) + api_mode = _api_key_provider_api_mode(provider, model_cfg, api_key, base_url, target_model or model_cfg.get("default", ""), + opencode_by_model=False) + api_key = _actual_local_key(provider, api_key, base_url) return _runtime(provider, api_mode, base_url.rstrip("/"), api_key, source="explicit", requested_provider=requested_provider) # Providers with a dedicated explicit-credential builder; everything else goes through the # registry ``api_key`` path (or None when the provider takes no explicit creds). _EXPLICIT_RESOLVERS: Dict[str, Callable[..., Dict[str, Any]]] = { - "anthropic": _explicit_anthropic, - "openai-codex": _explicit_codex, - "nous": _explicit_nous, + "anthropic": _explicit_anthropic, "openai-codex": _explicit_codex, "nous": _explicit_nous, "azure-foundry": _explicit_azure_foundry, } -def _resolve_explicit_runtime( - *, - provider: str, - requested_provider: str, - model_cfg: Dict[str, Any], - explicit_api_key: Optional[str] = None, - explicit_base_url: Optional[str] = None, - target_model: Optional[str] = None, -) -> Optional[Dict[str, Any]]: +def _resolve_explicit_runtime(*, provider: str, requested_provider: str, model_cfg: Dict[str, Any], + explicit_api_key: Optional[str] = None, explicit_base_url: Optional[str] = None, + target_model: Optional[str] = None) -> Optional[Dict[str, Any]]: explicit_api_key = str(explicit_api_key or "").strip() explicit_base_url = str(explicit_base_url or "").strip().rstrip("/") if not explicit_api_key and not explicit_base_url: @@ -762,11 +622,10 @@ def _resolve_explicit_runtime( if resolver is not None: return resolver(requested_provider, model_cfg, explicit_api_key, explicit_base_url, target_model) pconfig = PROVIDER_REGISTRY.get(provider) - if pconfig and pconfig.auth_type == "api_key": - return _explicit_api_key_provider( - provider, pconfig, requested_provider, model_cfg, explicit_api_key, explicit_base_url, target_model - ) - return None + if not (pconfig and pconfig.auth_type == "api_key"): + return None + return _explicit_api_key_provider(provider, pconfig, requested_provider, model_cfg, explicit_api_key, explicit_base_url, + target_model) # ── OAuth / auth-store providers ─────────────────────────────────────────────────────────── @@ -787,23 +646,14 @@ class _OAuthRuntimeSpec: # ``resolve`` entries are late-bound lambdas so tests can monkeypatch the module-level # ``resolve_*_runtime_credentials`` names. _OAUTH_RUNTIME_PROVIDERS: Dict[str, _OAuthRuntimeSpec] = { - "nous": _OAuthRuntimeSpec( - _resolve_nous_creds, _nous_api_mode, "portal", "expires_at", - "Auto-detected Nous provider but credentials failed", - ), - "openai-codex": _OAuthRuntimeSpec( - lambda: resolve_codex_runtime_credentials(), "codex_responses", "hermes-auth-store", "last_refresh", - "Auto-detected Codex provider but credentials failed", - ), - "xai-oauth": _OAuthRuntimeSpec( - lambda: resolve_xai_oauth_runtime_credentials(), "codex_responses", "hermes-auth-store", "last_refresh", - "Auto-detected xAI OAuth provider but credentials failed", - default_base_url=DEFAULT_XAI_OAUTH_BASE_URL, - ), - "qwen-oauth": _OAuthRuntimeSpec( - lambda: resolve_qwen_runtime_credentials(), "chat_completions", "qwen-cli", "expires_at_ms", - "Qwen OAuth credentials failed", - ), + "nous": _OAuthRuntimeSpec(_resolve_nous_creds, nous_api_mode, "portal", "expires_at", + "Auto-detected Nous provider but credentials failed"), + "openai-codex": _OAuthRuntimeSpec(lambda: resolve_codex_runtime_credentials(), "codex_responses", "hermes-auth-store", + "last_refresh", "Auto-detected Codex provider but credentials failed"), + "xai-oauth": _OAuthRuntimeSpec(lambda: resolve_xai_oauth_runtime_credentials(), "codex_responses", "hermes-auth-store", + "last_refresh", "Auto-detected xAI OAuth provider but credentials failed", DEFAULT_XAI_OAUTH_BASE_URL), + "qwen-oauth": _OAuthRuntimeSpec(lambda: resolve_qwen_runtime_credentials(), "chat_completions", "qwen-cli", + "expires_at_ms", "Qwen OAuth credentials failed"), } @@ -819,56 +669,51 @@ def _resolve_oauth_runtime(provider, requested_provider, model_cfg, target_model raise logger.info("%s; falling through to next provider.", spec.failure_msg) return None - api_mode = spec.api_mode - if callable(api_mode): - api_mode = api_mode(target_model or model_cfg.get("default") or "") - return _runtime( - provider, api_mode, - (creds.get("base_url") or "").rstrip("/") or spec.default_base_url, - creds.get("api_key", ""), - source=creds.get("source", spec.default_source), - **{spec.expiry_key: creds.get(spec.expiry_key)}, - requested_provider=requested_provider, - ) + api_mode = spec.api_mode(_effective_model(model_cfg, target_model)) if callable(spec.api_mode) else spec.api_mode + return _runtime(provider, api_mode, (creds.get("base_url") or "").rstrip("/") or spec.default_base_url, + creds.get("api_key", ""), source=creds.get("source", spec.default_source), + **{spec.expiry_key: creds.get(spec.expiry_key)}, requested_provider=requested_provider) + + +def _minimax_oauth_runtime(provider, requested_provider) -> Optional[Dict[str, Any]]: + pconfig = PROVIDER_REGISTRY.get(provider) + if not (pconfig and pconfig.auth_type == "oauth_minimax"): + return None + creds = auth_mod.resolve_minimax_oauth_runtime_credentials() + return _runtime(provider, "anthropic_messages", creds["base_url"], creds["api_key"], source=creds.get("source", "oauth"), + requested_provider=requested_provider) # ── env/config paths for anthropic and registry api_key providers ────────────────────────── +def _azure_anthropic_env_key(model_cfg: Dict[str, Any]) -> str: + """Azure Anthropic key: `key_env` / `api_key_env` hints on the model config, then an inline + api_key (multi-profile setups), then the historical fixed names.""" + for hint_key in ("key_env", "api_key_env"): + env_var = str(model_cfg.get(hint_key) or "").strip() + if env_var: + token = _getenv(env_var, "").strip() + if token: + return token + return (str(model_cfg.get("api_key") or "").strip() or _getenv("AZURE_ANTHROPIC_KEY", "").strip() + or _getenv("ANTHROPIC_API_KEY", "").strip()) + + def _anthropic_env_runtime(requested_provider: str, model_cfg: Dict[str, Any]) -> Dict[str, Any]: """Native Anthropic (Messages API) from env/auth store; ``model.base_url`` honoured only when the configured provider is anthropic (else a Codex endpoint would leak into Anthropic requests).""" - cfg_base_url = _anthropic_cfg_base_url(model_cfg) - base_url = cfg_base_url or _ANTHROPIC_DEFAULT_BASE_URL + base_url = _anthropic_cfg_base_url(model_cfg) or _ANTHROPIC_DEFAULT_BASE_URL # Microsoft Foundry endpoints reject Claude Code OAuth tokens, which resolve_anthropic_token() - # would return first — use the env key directly: `key_env` / `api_key_env` hints on the model - # config, then an inline api_key (multi-profile setups), then the historical fixed names. - if base_url_host_matches(base_url, "azure.com") or (cfg_base_url and base_url_host_matches(cfg_base_url, "azure.com")): - token = "" - for hint_key in ("key_env", "api_key_env"): - env_var = str(model_cfg.get(hint_key) or "").strip() - if env_var: - token = _getenv(env_var, "").strip() - if token: - break - token = ( - token - or str(model_cfg.get("api_key") or "").strip() - or _getenv("AZURE_ANTHROPIC_KEY", "").strip() - or _getenv("ANTHROPIC_API_KEY", "").strip() - ) + # would return first — use the env key directly. + if base_url_host_matches(base_url, "azure.com"): + token = _azure_anthropic_env_key(model_cfg) if not token: - raise AuthError( - "No Azure Anthropic API key found. Set AZURE_ANTHROPIC_KEY or " - "ANTHROPIC_API_KEY, or point key_env/api_key_env in your " - "config.yaml model section at a custom env var." - ) + raise AuthError("No Azure Anthropic API key found. Set AZURE_ANTHROPIC_KEY or " + "ANTHROPIC_API_KEY, or point key_env/api_key_env in your " + "config.yaml model section at a custom env var.") else: - from agent.anthropic_adapter import resolve_anthropic_token - - token = resolve_anthropic_token() - if not token: - raise AuthError(_NO_ANTHROPIC_CREDENTIALS_MSG) + token = _anthropic_token_or_raise() return _runtime("anthropic", "anthropic_messages", base_url, token, source="env", requested_provider=requested_provider) @@ -880,9 +725,7 @@ def _api_key_provider_runtime(provider, pconfig, requested_provider, model_cfg, if provider == "actual" and not has_usable_secret(creds.get("api_key")): cfg_url = _config_base_url_for_provider(model_cfg, provider) if is_actual_local_base_url(normalize_actual_base_url(cfg_url or creds.get("base_url", "").rstrip("/"))): - creds = dict(creds) - creds["api_key"] = ACTUAL_LOCAL_NOAUTH_PLACEHOLDER - creds["source"] = creds.get("source") or "local-offline" + creds = {**creds, "api_key": ACTUAL_LOCAL_NOAUTH_PLACEHOLDER, "source": creds.get("source") or "local-offline"} # An explicitly selected API-key provider is authoritative: an empty key would defer failure # to the first request and make a later fallback look like a silent provider switch. if not has_usable_secret(creds.get("api_key")): @@ -890,16 +733,11 @@ def _api_key_provider_runtime(provider, pconfig, requested_provider, model_cfg, hint = f" Set {env_names}." if env_names else "" raise AuthError(f"No usable credentials found for provider '{provider}'.{hint}", provider=provider, code="missing_api_key") # Honour model.base_url when the configured provider matches (e.g. api.minimaxi.com China endpoint). - base_url = _config_base_url_for_provider(model_cfg, provider) or creds.get("base_url", "").rstrip("/") - if provider == "actual": - base_url = normalize_actual_base_url(base_url) - api_mode = _api_key_provider_api_mode( - provider, model_cfg, creds.get("api_key", ""), base_url, target_model or model_cfg.get("default", ""), opencode_by_model=True - ) + base_url = _actual_url(provider, _config_base_url_for_provider(model_cfg, provider) or creds.get("base_url", "").rstrip("/")) + api_mode = _api_key_provider_api_mode(provider, model_cfg, creds.get("api_key", ""), base_url, + target_model or model_cfg.get("default", ""), opencode_by_model=True) base_url = _finalize_base_url(provider, api_mode, base_url) - api_key = creds.get("api_key", "") - if provider == "actual" and not api_key and is_actual_local_base_url(base_url): - api_key = ACTUAL_LOCAL_NOAUTH_PLACEHOLDER + api_key = _actual_local_key(provider, creds.get("api_key", ""), base_url) return _runtime(provider, api_mode, base_url, api_key, source=creds.get("source", "env"), requested_provider=requested_provider) @@ -912,16 +750,12 @@ _LOCAL_BYPASS_CLOUD_HOSTS = ("openrouter.ai", "anthropic.com", "openai.com") def _raise_if_provider_disabled(requested_provider: str) -> None: """Honour ``providers..enabled: false`` for built-ins too (the custom lookup gate only covers custom blocks); a typed error lets the fallback chain advance.""" - from hermes_cli.config import is_provider_enabled, load_config - - full_cfg = load_config() + full_cfg = _config_mod.load_config() provs_cfg = full_cfg.get("providers") if isinstance(full_cfg, dict) else None block = provs_cfg.get(requested_provider) if isinstance(provs_cfg, dict) else None - if isinstance(block, dict) and not is_provider_enabled(block): - raise ValueError( - f"provider {requested_provider!r} is disabled in config " - f"(providers.{requested_provider}.enabled: false)" - ) + if isinstance(block, dict) and not _config_mod.is_provider_enabled(block): + raise ValueError(f"provider {requested_provider!r} is disabled in config " + f"(providers.{requested_provider}.enabled: false)") def _resolve_vertex_runtime(requested_provider: str) -> Dict[str, Any]: @@ -929,48 +763,36 @@ def _resolve_vertex_runtime(requested_provider: str) -> Dict[str, Any]: treated as a static API key; a short-lived token is minted per call, and mid-session expiry is recovered on 401 by run_agent._try_refresh_vertex_client_credentials().""" from agent.vertex_adapter import get_vertex_config - token, base_url = get_vertex_config() if not token or not base_url: - raise AuthError( - "Vertex AI credentials could not be resolved. Vertex uses " - "OAuth2 (not a static API key): provide a service-account JSON " - "via GOOGLE_APPLICATION_CREDENTIALS (or VERTEX_CREDENTIALS_PATH) " - "in ~/.hermes/.env, or run 'gcloud auth application-default " - "login' for ADC. Set the GCP project/region under vertex: in " - "config.yaml if they aren't embedded in the credentials. " - "Run `hermes setup` to install Vertex support." - ) + raise AuthError("Vertex AI credentials could not be resolved. Vertex uses " + "OAuth2 (not a static API key): provide a service-account JSON " + "via GOOGLE_APPLICATION_CREDENTIALS (or VERTEX_CREDENTIALS_PATH) " + "in ~/.hermes/.env, or run 'gcloud auth application-default " + "login' for ADC. Set the GCP project/region under vertex: in " + "config.yaml if they aren't embedded in the credentials. " + "Run `hermes setup` to install Vertex support.") return _runtime("vertex", "chat_completions", base_url.rstrip("/"), token, source="vertex-oauth", requested_provider=requested_provider) def _resolve_requested_shortcuts(requested_provider, explicit_api_key, explicit_base_url, target_model) -> Optional[Dict[str, Any]]: """Providers decided on the REQUESTED name alone, before custom / pool / generic paths.""" if requested_provider == "moa": - return _runtime( - "moa", "chat_completions", "moa://local", "moa-virtual-provider", - source="moa-virtual-provider", requested_provider=requested_provider, - ) + return _runtime("moa", "chat_completions", "moa://local", "moa-virtual-provider", source="moa-virtual-provider", + requested_provider=requested_provider) # Azure Anthropic short-circuit: an explicit Azure endpoint with provider="anthropic" must # bypass _resolve_named_custom_runtime (which would yield custom/chat_completions/no key). eff_base = (explicit_base_url or "").strip() if requested_provider == "anthropic" and base_url_host_matches(eff_base, "azure.com"): - azure_key = ( - (explicit_api_key or "").strip() - or _getenv("AZURE_ANTHROPIC_KEY", "").strip() - or _getenv("ANTHROPIC_API_KEY", "").strip() - ) - return _runtime( - "anthropic", "anthropic_messages", eff_base.rstrip("/"), azure_key, - source="azure-explicit", requested_provider=requested_provider, - ) + azure_key = (explicit_api_key or "").strip() or _getenv("AZURE_ANTHROPIC_KEY", "").strip() or _getenv("ANTHROPIC_API_KEY", "").strip() + return _runtime("anthropic", "anthropic_messages", eff_base.rstrip("/"), azure_key, source="azure-explicit", + requested_provider=requested_provider) # Azure Foundry resolves before the custom-runtime / pool / generic paths so its config is # always picked up from model.base_url + model.api_mode, with or without explicit_* args. if requested_provider == "azure-foundry": - return _resolve_azure_foundry_runtime( - requested_provider=requested_provider, model_cfg=_get_model_config(), - explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, target_model=target_model, - ) + return _resolve_azure_foundry_runtime(requested_provider=requested_provider, model_cfg=_get_model_config(), + explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, + target_model=target_model) if requested_provider in _VERTEX_NAMES: return _resolve_vertex_runtime(requested_provider) return None @@ -987,9 +809,12 @@ def _local_endpoint_bypass(requested_provider: str, explicit_api_key, explicit_b return None if any(base_url_host_matches(cfg_base_url, host) for host in _LOCAL_BYPASS_CLOUD_HOSTS): return None - runtime = _resolve_openrouter_runtime( - requested_provider=requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url - ) + return _openrouter_fallback(requested_provider, explicit_api_key, explicit_base_url) + + +def _openrouter_fallback(requested_provider, explicit_api_key, explicit_base_url) -> Dict[str, Any]: + runtime = _resolve_openrouter_runtime(requested_provider=requested_provider, explicit_api_key=explicit_api_key, + explicit_base_url=explicit_base_url) runtime["requested_provider"] = requested_provider return runtime @@ -998,27 +823,19 @@ def _opencode_free_runtime(provider, requested_provider, model_cfg, target_model """OpenCode Zen free tier (*-free slugs) is served ANONYMOUSLY on the Zen relay only: unknown bearers 401 and the Go relay rejects free models, so free slugs route through the keyless Zen runtime BEFORE the pool / explicit / api_key paths.""" - from hermes_cli.models import opencode_provider_family, opencode_zen_free_runtime - - if opencode_provider_family(provider) is None: + if _models.opencode_provider_family(provider) is None: return None model = str(target_model or model_cfg.get("default") or model_cfg.get("model") or "").strip() - free_runtime = opencode_zen_free_runtime(provider, model) + free_runtime = _models.opencode_zen_free_runtime(provider, model) if free_runtime is not None: free_runtime["requested_provider"] = requested_provider return free_runtime -def resolve_runtime_provider( - *, - requested: Optional[str] = None, - explicit_api_key: Optional[str] = None, - explicit_base_url: Optional[str] = None, - target_model: Optional[str] = None, -) -> Dict[str, Any]: - """Resolve runtime provider credentials for agent execution. - - Ladder (order is behavior — each rung returns or raises, else falls to the next): +def resolve_runtime_provider(*, requested: Optional[str] = None, explicit_api_key: Optional[str] = None, + explicit_base_url: Optional[str] = None, target_model: Optional[str] = None) -> Dict[str, Any]: + """Resolve runtime provider credentials for agent execution. Ladder (order is behavior — each + rung returns or raises, else falls to the next): 1. disabled-provider guard (``providers..enabled: false``) 2. requested-name shortcuts: moa, anthropic@azure, azure-foundry, vertex 3. named custom provider / llamacpp alias / bare-custom direct alias @@ -1028,82 +845,49 @@ def resolve_runtime_provider( 7. OAuth specs (nous/codex/xai/qwen; "auto" swallows AuthError and logs) → minimax-oauth → external-process → anthropic env → bedrock → registry api_key providers 8. OpenRouter / bare-custom fallback - - target_model: overrides model_cfg["default"] when computing provider-specific api_mode - (e.g. OpenCode Zen/Go where different models route through different API surfaces). - """ + target_model overrides model_cfg["default"] when computing provider-specific api_mode (e.g. + OpenCode Zen/Go where different models route through different API surfaces).""" requested_provider = resolve_requested_provider(requested) _raise_if_provider_disabled(requested_provider) + return next(r for r in _ladder_rungs(requested_provider, explicit_api_key, explicit_base_url, target_model) if r) - runtime = _resolve_requested_shortcuts(requested_provider, explicit_api_key, explicit_base_url, target_model) - if runtime: - return runtime - runtime = _resolve_named_custom_runtime( - requested_provider=requested_provider, explicit_api_key=explicit_api_key, - explicit_base_url=explicit_base_url, target_model=target_model, - ) +def _named_custom_rung(requested_provider, explicit_api_key, explicit_base_url, target_model): + runtime = _resolve_named_custom_runtime(requested_provider=requested_provider, explicit_api_key=explicit_api_key, + explicit_base_url=explicit_base_url, target_model=target_model) if runtime: runtime["requested_provider"] = requested_provider - return runtime + return runtime + +def _ladder_rungs(requested_provider, explicit_api_key, explicit_base_url, target_model): + """Ladder rungs 2-8, yielded lazily so each is evaluated only when the previous one returned + nothing; the last rung (OpenRouter / bare-custom fallback) always yields a runtime.""" + yield _resolve_requested_shortcuts(requested_provider, explicit_api_key, explicit_base_url, target_model) + yield _named_custom_rung(requested_provider, explicit_api_key, explicit_base_url, target_model) if not explicit_base_url and not explicit_api_key: - runtime = _local_endpoint_bypass(requested_provider, explicit_api_key, explicit_base_url) - if runtime: - return runtime - + yield _local_endpoint_bypass(requested_provider, explicit_api_key, explicit_base_url) provider = resolve_provider(requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url) model_cfg = _get_model_config() - - runtime = _opencode_free_runtime(provider, requested_provider, model_cfg, target_model) - if runtime is not None: - return runtime - - runtime = _resolve_explicit_runtime( - provider=provider, requested_provider=requested_provider, model_cfg=model_cfg, - explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, target_model=target_model, - ) - if runtime: - return runtime - - runtime = _resolve_from_pool(provider, requested_provider, model_cfg, explicit_api_key, explicit_base_url, target_model) - if runtime: - return runtime - + yield _opencode_free_runtime(provider, requested_provider, model_cfg, target_model) + yield _resolve_explicit_runtime(provider=provider, requested_provider=requested_provider, model_cfg=model_cfg, + explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, + target_model=target_model) + yield _resolve_from_pool(provider, requested_provider, model_cfg, explicit_api_key, explicit_base_url, target_model) if provider in _OAUTH_RUNTIME_PROVIDERS: - runtime = _resolve_oauth_runtime(provider, requested_provider, model_cfg, target_model) - if runtime: - return runtime - + yield _resolve_oauth_runtime(provider, requested_provider, model_cfg, target_model) if provider == "minimax-oauth": - pconfig = PROVIDER_REGISTRY.get(provider) - if pconfig and pconfig.auth_type == "oauth_minimax": - from hermes_cli.auth import resolve_minimax_oauth_runtime_credentials - - creds = resolve_minimax_oauth_runtime_credentials() - return _runtime( - provider, "anthropic_messages", creds["base_url"], creds["api_key"], - source=creds.get("source", "oauth"), requested_provider=requested_provider, - ) - + yield _minimax_oauth_runtime(provider, requested_provider) if _is_external_process_provider(provider): - return _resolve_external_process_runtime(provider, requested_provider) - + yield _resolve_external_process_runtime(provider, requested_provider) if provider == "anthropic": - return _anthropic_env_runtime(requested_provider, model_cfg) - + yield _anthropic_env_runtime(requested_provider, model_cfg) if provider == "bedrock": - return _resolve_bedrock_runtime(requested_provider, model_cfg, target_model) - + yield _resolve_bedrock_runtime(requested_provider, model_cfg, target_model) pconfig = PROVIDER_REGISTRY.get(provider) if pconfig and pconfig.auth_type == "api_key": - return _api_key_provider_runtime(provider, pconfig, requested_provider, model_cfg, target_model) - - runtime = _resolve_openrouter_runtime( - requested_provider=requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url - ) - runtime["requested_provider"] = requested_provider - return runtime + yield _api_key_provider_runtime(provider, pconfig, requested_provider, model_cfg, target_model) + yield _openrouter_fallback(requested_provider, explicit_api_key, explicit_base_url) def format_runtime_provider_error(error: Exception) -> str: diff --git a/hermes_cli/runtime_provider_backends.py b/hermes_cli/runtime_provider_backends.py index d9c662f200..400d4bc204 100644 --- a/hermes_cli/runtime_provider_backends.py +++ b/hermes_cli/runtime_provider_backends.py @@ -1,10 +1,8 @@ -"""Provider-specific runtime builders for :mod:`hermes_cli.runtime_provider`. - -Azure Foundry, the OpenRouter / bare-custom fallback resolver, Bedrock, and external-process -providers. Origin-internal collaborators are resolved on the origin module at call time via -:func:`_rp` so test patches on ``hermes_cli.runtime_provider.*`` (``_get_model_config``, -``load_config``, ``has_usable_secret``, ``_try_resolve_from_custom_pool``, …) still apply. -""" +"""Provider-specific runtime builders for :mod:`hermes_cli.runtime_provider`: Azure Foundry, the +OpenRouter / bare-custom fallback resolver, Bedrock, and external-process providers. Origin-internal +collaborators are resolved on the origin module at call time via :func:`_rp` so test patches on +``hermes_cli.runtime_provider.*`` (``_get_model_config``, ``load_config``, ``has_usable_secret``, +``_try_resolve_from_custom_pool``, …) still apply.""" from __future__ import annotations @@ -18,15 +16,9 @@ from utils import base_url_host_matches def _rp(): import hermes_cli.runtime_provider as origin - return origin -def _strip_v1(base_url: str) -> str: - """Anthropic SDK appends /v1/messages itself — drop an inherited trailing /v1.""" - return re.sub(r"/v1/?$", "", base_url) - - # ── Azure Foundry ────────────────────────────────────────────────────────────────────────── @@ -35,11 +27,7 @@ def _azure_entra_credentials(cfg_entra: Dict[str, Any]) -> Any: ``build_anthropic_client`` injects the bearer via an httpx hook).""" AuthError = _rp().AuthError try: - from agent.azure_identity_adapter import ( - SCOPE_AI_AZURE_DEFAULT, - EntraIdentityConfig, - build_token_provider, - ) + from agent.azure_identity_adapter import SCOPE_AI_AZURE_DEFAULT, EntraIdentityConfig, build_token_provider except Exception as exc: raise AuthError( "Azure Foundry Entra ID auth requires the 'azure-identity' " @@ -53,68 +41,15 @@ def _azure_entra_credentials(cfg_entra: Dict[str, Any]) -> Any: raise AuthError(str(exc)) from exc -def _resolve_azure_foundry_runtime( - *, - requested_provider: str, - model_cfg: Dict[str, Any], - explicit_api_key: Optional[str] = None, - explicit_base_url: Optional[str] = None, - target_model: Optional[str] = None, -) -> Dict[str, Any]: - """Azure Foundry: ``model.base_url`` + ``model.api_mode`` (or explicit overrides), API key from - ``.env``/env or a per-request Entra ID token, trailing ``/v1`` stripped for Anthropic-style - endpoints.""" - rp = _rp() - explicit_api_key = str(explicit_api_key or "").strip() - explicit_base_url_clean = str(explicit_base_url or "").strip().rstrip("/") - - cfg_base_url, cfg_api_mode, cfg_auth_mode, cfg_entra = "", "chat_completions", "api_key", {} - if rp._cfg_provider(model_cfg) == "azure-foundry": - cfg_base_url = rp._config_base_url_for_provider(model_cfg, "azure-foundry") - cfg_api_mode = rp._parse_api_mode(model_cfg.get("api_mode")) or "chat_completions" - cfg_auth_mode = str(model_cfg.get("auth_mode") or "api_key").strip().lower() or "api_key" - if isinstance(model_cfg.get("entra"), dict): - cfg_entra = model_cfg["entra"] - - # GPT-5.x / codex / o1-o4 deployments are Responses-API-only on Foundry. - effective_model = str(target_model or model_cfg.get("default") or "").strip() - cfg_api_mode = rp._azure_inferred_api_mode(effective_model, cfg_api_mode) - - env_base_url = rp._getenv("AZURE_FOUNDRY_BASE_URL", "").strip().rstrip("/") - base_url = explicit_base_url_clean or cfg_base_url or env_base_url - if not base_url: - raise rp.AuthError( - "Azure Foundry requires a base URL. Set it via 'hermes model' or " - "the AZURE_FOUNDRY_BASE_URL environment variable." - ) - if cfg_api_mode == "anthropic_messages": - base_url = _strip_v1(base_url) - - if cfg_auth_mode == "entra_id": - if explicit_api_key: - # --api-key on the CLI while config says entra_id: honour the explicit string - # (escape hatch for one-off testing). - api_key, source, auth_mode = explicit_api_key, "explicit", "api_key" - else: - api_key, source, auth_mode = _azure_entra_credentials(cfg_entra), "entra_id", "entra_id" - clean_entra = {} - configured_scope = str(cfg_entra.get("scope") or "").strip() - if auth_mode == "entra_id" and configured_scope: - clean_entra["scope"] = configured_scope - return rp._runtime( - "azure-foundry", cfg_api_mode, base_url, api_key, - auth_mode=auth_mode, entra=clean_entra, source=source, requested_provider=requested_provider, - ) - - api_key = explicit_api_key - if not api_key: - try: - from hermes_cli.config import get_env_value - - api_key = get_env_value("AZURE_FOUNDRY_API_KEY") or "" - except Exception: - api_key = "" - api_key = api_key or rp._getenv("AZURE_FOUNDRY_API_KEY", "").strip() +def _azure_foundry_api_key(rp, explicit_api_key: str) -> str: + if explicit_api_key: + return explicit_api_key + try: + from hermes_cli.config import get_env_value + api_key = get_env_value("AZURE_FOUNDRY_API_KEY") or "" + except Exception: + api_key = "" + api_key = api_key or rp._getenv("AZURE_FOUNDRY_API_KEY", "").strip() if not api_key: raise rp.AuthError( "Azure Foundry requires an API key. Set AZURE_FOUNDRY_API_KEY in " @@ -123,92 +58,103 @@ def _resolve_azure_foundry_runtime( "model.auth_mode: entra_id in config.yaml (or pick " "'Microsoft Entra ID' in 'hermes model')." ) - return rp._runtime( - "azure-foundry", cfg_api_mode, base_url, api_key, - auth_mode="api_key", - source="explicit" if (explicit_api_key or explicit_base_url) else "config", - requested_provider=requested_provider, - ) + return api_key + + +def _resolve_azure_foundry_runtime(*, requested_provider: str, model_cfg: Dict[str, Any], + explicit_api_key: Optional[str] = None, explicit_base_url: Optional[str] = None, + target_model: Optional[str] = None) -> Dict[str, Any]: + """Azure Foundry: ``model.base_url`` + ``model.api_mode`` (or explicit overrides), API key from + ``.env``/env or a per-request Entra ID token, trailing ``/v1`` stripped for Anthropic-style + endpoints (the Anthropic SDK appends /v1/messages itself).""" + rp = _rp() + explicit_api_key = str(explicit_api_key or "").strip() + explicit_base_url_clean = str(explicit_base_url or "").strip().rstrip("/") + cfg_base_url, cfg_api_mode, cfg_auth_mode, cfg_entra = "", "chat_completions", "api_key", {} + if rp._cfg_provider(model_cfg) == "azure-foundry": + cfg_base_url = rp._config_base_url_for_provider(model_cfg, "azure-foundry") + cfg_api_mode = rp._parse_api_mode(model_cfg.get("api_mode")) or "chat_completions" + cfg_auth_mode = str(model_cfg.get("auth_mode") or "api_key").strip().lower() or "api_key" + if isinstance(model_cfg.get("entra"), dict): + cfg_entra = model_cfg["entra"] + # GPT-5.x / codex / o1-o4 deployments are Responses-API-only on Foundry. + effective_model = str(target_model or model_cfg.get("default") or "").strip() + cfg_api_mode = rp._azure_inferred_api_mode(effective_model, cfg_api_mode) + env_base_url = rp._getenv("AZURE_FOUNDRY_BASE_URL", "").strip().rstrip("/") + base_url = explicit_base_url_clean or cfg_base_url or env_base_url + if not base_url: + raise rp.AuthError( + "Azure Foundry requires a base URL. Set it via 'hermes model' or " + "the AZURE_FOUNDRY_BASE_URL environment variable." + ) + if cfg_api_mode == "anthropic_messages": + base_url = re.sub(r"/v1/?$", "", base_url) + if cfg_auth_mode == "entra_id": + # --api-key on the CLI while config says entra_id: honour the explicit string (escape hatch + # for one-off testing). + if explicit_api_key: + api_key, source, auth_mode, entra = explicit_api_key, "explicit", "api_key", {} + else: + scope = str(cfg_entra.get("scope") or "").strip() + api_key, source, auth_mode, entra = _azure_entra_credentials(cfg_entra), "entra_id", "entra_id", ( + {"scope": scope} if scope else {} + ) + return rp._runtime("azure-foundry", cfg_api_mode, base_url, api_key, auth_mode=auth_mode, entra=entra, source=source, + requested_provider=requested_provider) + return rp._runtime("azure-foundry", cfg_api_mode, base_url, _azure_foundry_api_key(rp, explicit_api_key), + auth_mode="api_key", source="explicit" if (explicit_api_key or explicit_base_url) else "config", + requested_provider=requested_provider) # ── OpenRouter / bare custom fallback ────────────────────────────────────────────────────── def _resolve_openrouter_runtime( - *, - requested_provider: str, - explicit_api_key: Optional[str] = None, - explicit_base_url: Optional[str] = None, + *, requested_provider: str, explicit_api_key: Optional[str] = None, explicit_base_url: Optional[str] = None ) -> Dict[str, Any]: - """Terminal resolver: OpenRouter, or a bare/aliased ``custom`` endpoint. - - base_url precedence: explicit > CUSTOM_BASE_URL > trusted ``model.base_url`` > OPENROUTER_BASE_URL - > default. OPENAI_BASE_URL is deliberately NOT consulted — config.yaml is the single source of - truth for endpoint URLs. OpenRouter contexts prefer OPENROUTER_API_KEY; custom endpoints never - receive the OpenRouter key and only get env keys gated on their authoritative hosts. - """ + """Terminal resolver: OpenRouter, or a bare/aliased ``custom`` endpoint. base_url precedence: + explicit > CUSTOM_BASE_URL > trusted ``model.base_url`` > OPENROUTER_BASE_URL > default. + OPENAI_BASE_URL is deliberately NOT consulted — config.yaml is the single source of truth for + endpoint URLs. OpenRouter contexts prefer OPENROUTER_API_KEY; custom endpoints never receive the + OpenRouter key and only get env keys gated on their authoritative hosts.""" rp = _rp() model_cfg = rp._get_model_config() cfg_base_url = model_cfg.get("base_url") if isinstance(model_cfg.get("base_url"), str) else "" - cfg_provider = model_cfg.get("provider") if isinstance(model_cfg.get("provider"), str) else "" - cfg_api_key = next( - (v.strip() for v in (model_cfg.get("api_key"), model_cfg.get("api")) if isinstance(v, str) and v.strip()), - "", - ) + cfg_provider = (model_cfg.get("provider") if isinstance(model_cfg.get("provider"), str) else "").strip().lower() + cfg_api_key = next((v.strip() for v in (model_cfg.get("api_key"), model_cfg.get("api")) if isinstance(v, str) and v.strip()), "") requested_norm = (requested_provider or "").strip().lower() - cfg_provider = cfg_provider.strip().lower() # Aliases resolving to "custom" (ollama, vllm, …) follow bare-custom trust + routing rules. if requested_norm and requested_norm != "custom" and rp._resolves_to_custom(requested_norm): requested_norm = "custom" - env_openrouter_base_url = rp._getenv("OPENROUTER_BASE_URL", "").strip() env_custom_base_url = rp._getenv("CUSTOM_BASE_URL", "").strip() - use_config_base_url = bool(cfg_base_url.strip()) and not explicit_base_url and ( (requested_norm == "auto" and cfg_provider in ("", "auto")) or (requested_norm == "custom" and rp._config_base_url_trustworthy_for_bare_custom(cfg_base_url, cfg_provider)) ) - base_url = ( - (explicit_base_url or "").strip() - or env_custom_base_url - or (cfg_base_url.strip() if use_config_base_url else "") - or env_openrouter_base_url - or OPENROUTER_BASE_URL - ).rstrip("/") - + base_url = ((explicit_base_url or "").strip() or env_custom_base_url or (cfg_base_url.strip() if use_config_base_url else "") + or env_openrouter_base_url or OPENROUTER_BASE_URL).rstrip("/") is_openrouter_url = base_url_host_matches(base_url, "openrouter.ai") # Explicitly-configured OpenRouter mirrors (OPENROUTER_BASE_URL + provider=openrouter) still # count as OpenRouter for key selection. is_openrouter_context = is_openrouter_url or ( - requested_norm == "openrouter" - and (env_openrouter_base_url or base_url == env_openrouter_base_url) + requested_norm == "openrouter" and (env_openrouter_base_url or base_url == env_openrouter_base_url) and base_url == (env_openrouter_base_url or "").rstrip("/") ) if is_openrouter_context: candidates = [explicit_api_key, rp._getenv("OPENROUTER_API_KEY"), rp._getenv("OPENAI_API_KEY")] else: - candidates = [ - explicit_api_key, - (cfg_api_key if use_config_base_url else ""), - *rp._host_gated_env_key_candidates(base_url, ollama=True), - ] + candidates = [explicit_api_key, (cfg_api_key if use_config_base_url else ""), + *rp._host_gated_env_key_candidates(base_url, ollama=True)] api_key = next((str(c or "").strip() for c in candidates if rp.has_usable_secret(c)), "") source = "explicit" if (explicit_api_key or explicit_base_url) else "env/config" - + cfg_api_mode = rp._parse_api_mode(model_cfg.get("api_mode")) # Explicit "custom" stays "custom" rather than relabeling to "openrouter". if requested_norm != "custom": - return rp._runtime( - "openrouter", - rp._parse_api_mode(model_cfg.get("api_mode")) or rp._detect_api_mode_for_url(base_url) or "chat_completions", - base_url, api_key, source=source, - ) + return rp._runtime("openrouter", cfg_api_mode or rp._detect_api_mode_for_url(base_url) or "chat_completions", base_url, + api_key, source=source) if base_url: - # provider_name makes pool lookup prefer name match over base_url (fixes credential - # mix-ups when multiple custom providers share a base_url). - pool_result = rp._try_resolve_from_custom_pool( - base_url, "custom", rp._parse_api_mode(model_cfg.get("api_mode")), - provider_name=requested_provider if requested_norm != "custom" else None, - ) + pool_result = rp._try_resolve_from_custom_pool(base_url, "custom", cfg_api_mode, provider_name=None) if pool_result: return pool_result # Local no-auth servers get a placeholder key — the OpenAI SDK requires a non-empty string. @@ -236,17 +182,10 @@ def _resolve_bedrock_runtime(requested_provider: str, model_cfg: Dict[str, Any], Claude → AnthropicBedrock SDK (prompt caching, thinking budgets); others → Converse API. AWS_BEARER_TOKEN_BEDROCK auth is unsupported by AnthropicBedrock (SigV4 only), so bearer users go through Converse regardless of model.""" - from agent.bedrock_adapter import ( - bedrock_openai_base_url, - has_aws_credentials, - is_anthropic_bedrock_model, - is_openai_bedrock_model, - resolve_aws_auth_env_var, - resolve_bedrock_bearer_token, - resolve_bedrock_runtime_region, - ) + from agent.bedrock_adapter import (bedrock_openai_base_url, has_aws_credentials, is_anthropic_bedrock_model, + is_openai_bedrock_model, resolve_aws_auth_env_var, resolve_bedrock_bearer_token, + resolve_bedrock_runtime_region) from hermes_cli.config import load_config # direct (not the origin delegate), as before - rp = _rp() # Explicitly selected bedrock trusts boto3's credential chain (IMDS, ECS/Lambda roles, SSO) # which the env-var check can't detect. @@ -267,20 +206,12 @@ def _resolve_bedrock_runtime(requested_provider: str, model_cfg: Dict[str, Any], guardrail_config = _bedrock_guardrail_config(bedrock_cfg) current_model = str(target_model or model_cfg.get("default") or "").strip() has_bearer_token = bool(os.environ.get("AWS_BEARER_TOKEN_BEDROCK", "").strip()) - runtime = rp._runtime( - "bedrock", "bedrock_converse", f"https://bedrock-runtime.{region}.amazonaws.com", "aws-sdk", - source=auth_source, region=region, requested_provider=requested_provider, - ) + runtime = rp._runtime("bedrock", "bedrock_converse", f"https://bedrock-runtime.{region}.amazonaws.com", "aws-sdk", + source=auth_source, region=region, requested_provider=requested_provider) if is_openai_bedrock_model(current_model): bearer = resolve_bedrock_bearer_token() - runtime.update( - api_mode="codex_responses", - base_url=bedrock_openai_base_url(region), - api_key=bearer or "aws-sdk", - source="AWS_BEARER_TOKEN_BEDROCK" if bearer else auth_source, - model=current_model, - bedrock_openai=True, - ) + runtime.update(api_mode="codex_responses", base_url=bedrock_openai_base_url(region), api_key=bearer or "aws-sdk", + source="AWS_BEARER_TOKEN_BEDROCK" if bearer else auth_source, model=current_model, bedrock_openai=True) elif is_anthropic_bedrock_model(current_model) and not has_bearer_token: runtime.update(api_mode="anthropic_messages", bedrock_anthropic=True) if guardrail_config: @@ -298,9 +229,7 @@ def _is_external_process_provider(provider: str) -> bool: if not name: return False try: - from hermes_cli.auth import PROVIDER_REGISTRY - - pconfig = PROVIDER_REGISTRY.get(name) + pconfig = _rp().PROVIDER_REGISTRY.get(name) if pconfig is not None: return pconfig.auth_type == "external_process" except Exception: @@ -317,8 +246,6 @@ def _is_external_process_provider(provider: str) -> bool: def _resolve_external_process_runtime(provider: str, requested_provider: str) -> Dict[str, Any]: rp = _rp() creds = rp.resolve_external_process_provider_credentials(provider) - return rp._runtime( - provider, "chat_completions", creds.get("base_url", "").rstrip("/"), creds.get("api_key", ""), - command=creds.get("command", ""), args=list(creds.get("args") or []), - source=creds.get("source", "process"), requested_provider=requested_provider, - ) + return rp._runtime(provider, "chat_completions", creds.get("base_url", "").rstrip("/"), creds.get("api_key", ""), + command=creds.get("command", ""), args=list(creds.get("args") or []), + source=creds.get("source", "process"), requested_provider=requested_provider) diff --git a/hermes_cli/runtime_provider_custom.py b/hermes_cli/runtime_provider_custom.py index 1de8c6771d..48f9f64428 100644 --- a/hermes_cli/runtime_provider_custom.py +++ b/hermes_cli/runtime_provider_custom.py @@ -1,12 +1,9 @@ -"""Custom-provider resolution: ``providers:`` / ``custom_providers:`` lookup, identity -recovery, custom credential pools, and the named-custom runtime builder. - -Extracted from :mod:`hermes_cli.runtime_provider`; every public/private name here is -re-exported there. Origin-internal collaborators (``load_config``, ``_get_model_config``, -``load_pool``, ``has_usable_secret``, ``custom_provider_pool_key_candidates``, …) are looked up -on the origin module AT CALL TIME via :func:`_rp` so ``monkeypatch.setattr(runtime_provider, -name, …)`` in tests keeps working for moved bodies. -""" +"""Custom-provider resolution: ``providers:`` / ``custom_providers:`` lookup, identity recovery, +custom credential pools, and the named-custom runtime builder. Extracted from +:mod:`hermes_cli.runtime_provider` (every name re-exported there); origin-internal collaborators +(``load_config``, ``_get_model_config``, ``load_pool``, ``has_usable_secret``, …) are looked up on +the origin module AT CALL TIME via :func:`_rp` so ``monkeypatch.setattr(runtime_provider, name, …)`` +keeps working for moved bodies.""" from __future__ import annotations @@ -19,11 +16,12 @@ from utils import base_url_hostname logger = logging.getLogger("hermes_cli.runtime_provider") +_LLAMACPP_ALIASES = ("llamacpp", "llama.cpp", "llama-cpp") + def _rp(): """Origin module, late-bound so test patches on ``hermes_cli.runtime_provider.*`` apply.""" import hermes_cli.runtime_provider as origin - return origin @@ -65,11 +63,9 @@ def _lift_model_capabilities(entry: Dict[str, Any], model: Optional[str], result def _lift_max_output_tokens(entry: Dict[str, Any], result: Dict[str, Any]) -> None: - """``max_output_tokens`` or ``max_tokens`` on a provider entry pins its own output limit. - - Gateway/CLI map it onto ``AIAgent.max_tokens`` only when top-level ``model.max_tokens`` is - unset, so the documented global key still wins. - """ + """``max_output_tokens`` or ``max_tokens`` on a provider entry pins its own output limit; + gateway/CLI map it onto ``AIAgent.max_tokens`` only when top-level ``model.max_tokens`` is + unset, so the documented global key still wins.""" for key in ("max_output_tokens", "max_tokens"): value = entry.get(key) if isinstance(value, int) and value > 0: @@ -84,14 +80,8 @@ def _lift_extra_headers(entry: Dict[str, Any], result: Dict[str, Any]) -> None: result["extra_headers"] = extra_headers -def _lift_common_custom_fields( - entry: Dict[str, Any], - result: Dict[str, Any], - *, - provider_key: str, - key_env: str, - api_mode: Optional[str], -) -> None: +def _lift_common_custom_fields(entry: Dict[str, Any], result: Dict[str, Any], *, provider_key: str, key_env: str, + api_mode: Optional[str]) -> None: """Copy the optional fields shared by ``providers:`` and legacy ``custom_providers:`` entries.""" if key_env: result["key_env"] = key_env @@ -104,23 +94,19 @@ def _lift_common_custom_fields( if api_mode: result["api_mode"] = api_mode _lift_max_output_tokens(entry, result) - capabilities = _filter_capabilities(entry.get("capabilities")) - if capabilities: - result["capabilities"] = capabilities + _lift_model_capabilities(entry, None, result) # ── config lookup ────────────────────────────────────────────────────────────────────────── def _shadowed_by_builtin(requested_norm: str) -> bool: - """Raw names map to custom providers only when they are not canonical built-ins. - - Explicit ``custom:`` keys always target the saved entry, and bare ``custom`` is - exempt: a user may literally name a ``providers:`` entry "custom" (returning None before - the config scan made such cron jobs fail with ``auth_unavailable``). Defer to the built-in - only when the raw name IS the canonical provider (``nous``); an entry matching merely an - alias (``kimi`` → ``kimi-coding``) is the user's target. - """ + """Raw names map to custom providers only when they are not canonical built-ins. Explicit + ``custom:`` keys always target the saved entry, and bare ``custom`` is exempt: a user may + literally name a ``providers:`` entry "custom" (returning None before the config scan made such + cron jobs fail with ``auth_unavailable``). Defer to the built-in only when the raw name IS the + canonical provider (``nous``); an entry matching merely an alias (``kimi`` → ``kimi-coding``) + is the user's target.""" if requested_norm == "custom" or requested_norm.startswith("custom:"): return False rp = _rp() @@ -134,7 +120,6 @@ def _shadowed_by_builtin(requested_norm: str) -> bool: def _match_new_style_provider(requested_norm: str, providers: Dict[str, Any]) -> Optional[Dict[str, Any]]: """Scan ``providers:`` (new-style, keyed) for ``requested_norm``.""" from hermes_cli.config import is_provider_enabled - rp = _rp() for ep_name, entry in providers.items(): # ``providers..enabled: false`` entries stay in config but are invisible here. @@ -149,12 +134,8 @@ def _match_new_style_provider(requested_norm: str, providers: Dict[str, Any]) -> base_url = _entry_url(entry) if not base_url: continue - result: Dict[str, Any] = { - "name": entry.get("name", ep_name), - "base_url": base_url.strip(), - "api_key": api_key or _clean(entry.get("api_key", "")), - "model": entry.get("default_model", ""), - } + result: Dict[str, Any] = {"name": entry.get("name", ep_name), "base_url": base_url.strip(), + "api_key": api_key or _clean(entry.get("api_key", "")), "model": entry.get("default_model", "")} # Command that PRINTS a short-lived credential; wrapped in a per-request token provider. key_cmd = _clean(entry.get("key_cmd", "")) if key_cmd: @@ -162,9 +143,7 @@ def _match_new_style_provider(requested_norm: str, providers: Dict[str, Any]) -> # v12 migration writes ``transport``; hand-edited configs may still use ``api_mode``. # Accept both or migrated configs silently downgrade to chat_completions. _lift_common_custom_fields( - entry, result, - provider_key=_clean(ep_name), - key_env=key_env, + entry, result, provider_key=_clean(ep_name), key_env=key_env, api_mode=rp._parse_api_mode(entry.get("api_mode") or entry.get("transport")), ) return result @@ -174,9 +153,7 @@ def _match_new_style_provider(requested_norm: str, providers: Dict[str, Any]) -> def _match_legacy_custom_provider(requested_norm: str, custom_providers) -> Optional[Dict[str, Any]]: """Scan the legacy ``custom_providers:`` list for ``requested_norm``.""" for entry in custom_providers: - if not isinstance(entry, dict): - continue - name, base_url = entry.get("name"), entry.get("base_url") + name, base_url = (entry.get("name"), entry.get("base_url")) if isinstance(entry, dict) else (None, None) if not isinstance(name, str) or not isinstance(base_url, str): continue provider_key = _clean(entry.get("provider_key", "")) @@ -186,12 +163,8 @@ def _match_legacy_custom_provider(requested_norm: str, custom_providers) -> Opti model_name = _clean(entry.get("model", "")) if model_name: result["model"] = model_name - _lift_common_custom_fields( - entry, result, - provider_key=provider_key, - key_env=_clean(entry.get("key_env", "")), - api_mode=_rp()._parse_api_mode(entry.get("api_mode")), - ) + _lift_common_custom_fields(entry, result, provider_key=provider_key, key_env=_clean(entry.get("key_env", "")), + api_mode=_rp()._parse_api_mode(entry.get("api_mode"))) return result return None @@ -200,33 +173,24 @@ def _get_named_custom_provider(requested_provider: str) -> Optional[Dict[str, An requested_norm = _normalize_custom_provider_name(requested_provider or "") if not requested_norm or requested_norm == "auto" or _shadowed_by_builtin(requested_norm): return None - rp = _rp() config = rp.load_config() providers = config.get("providers") - if isinstance(providers, dict): - found = _match_new_style_provider(requested_norm, providers) - if found: - return found - + found = _match_new_style_provider(requested_norm, providers) if isinstance(providers, dict) else None + if found: + return found if isinstance(config.get("custom_providers"), dict): - logger.warning( - "custom_providers in config.yaml is a dict, not a list. " - "Each entry must be prefixed with '-' in YAML. " - "Run 'hermes doctor' for details." - ) + logger.warning("custom_providers in config.yaml is a dict, not a list. " + "Each entry must be prefixed with '-' in YAML. " + "Run 'hermes doctor' for details.") return None custom_providers = rp.get_compatible_custom_providers(config) - if not custom_providers: - return None - return _match_legacy_custom_provider(requested_norm, custom_providers) + return _match_legacy_custom_provider(requested_norm, custom_providers) if custom_providers else None def has_named_custom_provider(requested_provider: str) -> bool: - """True when config defines a ``providers:`` / ``custom_providers:`` entry matching the request. - - Public wrapper so other modules (e.g. the cronjob tool) need not reach into a private helper. - """ + """True when config defines a ``providers:`` / ``custom_providers:`` entry matching the request + (public wrapper so e.g. the cronjob tool need not reach into a private helper).""" try: return _rp()._get_named_custom_provider(requested_provider) is not None except Exception: @@ -244,96 +208,72 @@ def _find_custom_identity(matches: Callable[[Dict[str, Any]], bool]) -> Optional config = rp.load_config() except Exception: return None - providers = config.get("providers") if isinstance(providers, dict): for ep_name, entry in providers.items(): if isinstance(entry, dict) and matches(entry): return custom_provider_slug(str(ep_name), str(ep_name)) - try: custom_providers = rp.get_compatible_custom_providers(config) except Exception: custom_providers = None for entry in custom_providers or []: - if not isinstance(entry, dict): - continue - name = entry.get("name") + name = entry.get("name") if isinstance(entry, dict) else None if isinstance(name, str) and name.strip() and matches(entry): return custom_provider_slug(name, str(entry.get("provider_key", "") or "")) return None def find_custom_provider_identity(base_url: str) -> Optional[str]: - """Map an endpoint URL back to its canonical ``custom:`` menu key. - - Session persistence stores the agent's *resolved* provider, and for every named custom - endpoint that is the literal string ``"custom"`` — the entry name is lost, and the api_key is - deliberately never persisted. - """ + """Map an endpoint URL back to its canonical ``custom:`` menu key. Session persistence + stores the agent's *resolved* provider, which for every named custom endpoint is the literal + string ``"custom"`` — the entry name is lost, and the api_key is deliberately never persisted.""" target = _normalize_base_url_for_match(base_url) if not target: return None return _find_custom_identity(lambda entry: _normalize_base_url_for_match(_entry_url(entry)) == target) -def find_custom_provider_identity_by_model(model: str) -> Optional[str]: - """Map a model id back to the ``custom:`` entry that serves it. +def _model_id_matches(value: Any, target: str) -> bool: + return isinstance(value, str) and value.strip().lower() == target - Companion to :func:`find_custom_provider_identity` for persistence paths where no base_url - survived the round-trip: the session row always stores the model name. - """ + +def find_custom_provider_identity_by_model(model: str) -> Optional[str]: + """Map a model id back to the ``custom:`` entry that serves it — companion to + :func:`find_custom_provider_identity` for persistence paths where no base_url survived the + round-trip (the session row always stores the model name).""" target = str(model or "").strip().lower() if not target: return None def _entry_serves_model(entry: Dict[str, Any]) -> bool: - for key in ("model", "default_model"): - value = entry.get(key) - if isinstance(value, str) and value.strip().lower() == target: - return True + if any(_model_id_matches(entry.get(key), target) for key in ("model", "default_model")): + return True models = entry.get("models") if isinstance(models, dict): return any(str(mid).strip().lower() == target for mid in models) if isinstance(models, list): - for item in models: - if isinstance(item, str) and item.strip().lower() == target: - return True - if isinstance(item, dict): - mid = item.get("id") or item.get("name") - if isinstance(mid, str) and mid.strip().lower() == target: - return True + return any(_model_id_matches(item.get("id") or item.get("name") if isinstance(item, dict) else item, target) + for item in models) return False return _find_custom_identity(_entry_serves_model) -def canonical_custom_identity( - *, - base_url: Optional[str] = None, - config_provider: Optional[str] = None, - model: Optional[str] = None, -) -> Optional[str]: - """Recover a routable ``custom:`` identity for a bare custom provider. - - Every path that persists or restores a session's provider override must run the resolved - provider through this so a bare ``"custom"`` is upgraded back to its durable - ``custom:`` menu key. Recovery sources, in priority order: (1) ``base_url`` reverse - lookup — the one fact that always survives the round-trip when a URL was recorded; (2) - ``model`` reverse lookup (``model``/``default_model``/``models`` catalog); (3) the configured - provider (arg, then ``model.provider``, then ``HERMES_INFERENCE_PROVIDER``) when it names a - real entry. - """ +def canonical_custom_identity(*, base_url: Optional[str] = None, config_provider: Optional[str] = None, + model: Optional[str] = None) -> Optional[str]: + """Recover a routable ``custom:`` identity for a bare custom provider. Every path that + persists or restores a session's provider override must run the resolved provider through this + so a bare ``"custom"`` is upgraded back to its durable menu key. Sources in priority order: + (1) ``base_url`` reverse lookup — the one fact that always survives the round-trip when a URL + was recorded; (2) ``model`` reverse lookup (``model``/``default_model``/``models`` catalog); + (3) the configured provider (arg, ``model.provider``, ``HERMES_INFERENCE_PROVIDER``) when it + names a real entry.""" rp = _rp() - if base_url: - identity = find_custom_provider_identity(base_url) - if identity: - return identity - if model: - identity = find_custom_provider_identity_by_model(model) - if identity: - return identity - + identity = (find_custom_provider_identity(base_url) if base_url else None) or ( + find_custom_provider_identity_by_model(model) if model else None) + if identity: + return identity candidate = str(config_provider or "").strip() if not candidate: try: @@ -342,37 +282,33 @@ def canonical_custom_identity( candidate = "" if not candidate: candidate = os.environ.get("HERMES_INFERENCE_PROVIDER", "").strip() - candidate_norm = _normalize_custom_provider_name(candidate) # A bare/non-routable candidate cannot heal a bare custom override. if not candidate_norm or candidate_norm in {"custom", "auto", "openrouter"}: return None # Only when it resolves to a configured entry — never invent a ``custom:`` resolution - # can't honor. + # can't honor. ``candidate`` may be the entry's DISPLAY NAME, not the durable identity of a + # keyed ``providers:`` entry — re-resolve via its endpoint so every path returns the same + # config-key slug. try: entry = rp._get_named_custom_provider(candidate) - if entry is not None: - # ``candidate`` may be the entry's DISPLAY NAME, not the durable identity of a keyed - # ``providers:`` entry — re-resolve via its endpoint so every path returns the same - # config-key slug. - identity = find_custom_provider_identity(str(entry.get("base_url") or "")) - if identity: - return identity - return candidate_norm if candidate_norm.startswith("custom:") else f"custom:{candidate_norm}" except Exception: - pass - return None + return None + if entry is None: + return None + try: + identity = find_custom_provider_identity(str(entry.get("base_url") or "")) + except Exception: + return None + return identity or custom_provider_slug(candidate_norm) def is_routable_provider(provider: Optional[str]) -> bool: - """Whether a provider name currently resolves to a routable route. - - Empty/None/``auto`` is vacuously routable (agent build falls back to the configured - default). Bare ``custom`` is the resolved billing class shared by every named entry — not a - routable identity; restore paths must heal it (:func:`canonical_custom_identity`) or fall - back. Anything else is routable iff the full chain (built-in -> ``providers:`` -> - ``custom_providers:`` -> models.dev) resolves it. - """ + """Whether a provider name currently resolves to a routable route. Empty/None/``auto`` is + vacuously routable (agent build falls back to the configured default). Bare ``custom`` is the + resolved billing class shared by every named entry — not a routable identity; restore paths + must heal it (:func:`canonical_custom_identity`) or fall back. Anything else is routable iff the + full chain (built-in -> ``providers:`` -> ``custom_providers:`` -> models.dev) resolves it.""" name = str(provider or "").strip() if not name or name.lower() == "auto": return True @@ -380,12 +316,9 @@ def is_routable_provider(provider: Optional[str]) -> bool: return False try: from hermes_cli.providers import resolve_provider_full - rp = _rp() config = rp.load_config() - return resolve_provider_full( - name, config.get("providers"), rp.get_compatible_custom_providers(config) - ) is not None + return resolve_provider_full(name, config.get("providers"), rp.get_compatible_custom_providers(config)) is not None except Exception: return False @@ -394,10 +327,7 @@ def is_routable_provider(provider: Optional[str]) -> bool: def _try_resolve_from_custom_pool( - base_url: str, - provider_label: str, - api_mode_override: Optional[str] = None, - provider_name: Optional[str] = None, + base_url: str, provider_label: str, api_mode_override: Optional[str] = None, provider_name: Optional[str] = None ) -> Optional[Dict[str, Any]]: """Runtime dict from the first credential pool that owns this custom endpoint, else None.""" rp = _rp() @@ -410,12 +340,8 @@ def _try_resolve_from_custom_pool( for pool_key in candidates: try: pool = rp.load_pool(pool_key) - if not pool.has_credentials(): - continue - entry = pool.select() - if entry is None: - continue - pool_api_key = rp._pool_entry_api_key(entry) + entry = pool.select() if pool.has_credentials() else None + pool_api_key = rp._pool_entry_api_key(entry) if entry is not None else "" if not pool_api_key: continue if not rp.has_usable_secret(pool_api_key) and rp._loopback_hostname(base_url_hostname(base_url)): @@ -423,14 +349,8 @@ def _try_resolve_from_custom_pool( # services; has_usable_secret's 4-char floor rejects them. Every other path # substitutes "no-key-required" for a loopback endpoint — this was the one gap. pool_api_key = "no-key-required" - return rp._runtime( - provider_label, - api_mode_override or rp._detect_api_mode_for_url(base_url) or "chat_completions", - base_url, - pool_api_key, - source=f"pool:{pool_key}", - credential_pool=pool, - ) + return rp._runtime(provider_label, api_mode_override or rp._detect_api_mode_for_url(base_url) or "chat_completions", + base_url, pool_api_key, source=f"pool:{pool_key}", credential_pool=pool) except Exception: continue return None @@ -443,16 +363,11 @@ def _custom_provider_request_overrides(custom_provider: Dict[str, Any]) -> Dict[ return {"extra_body": dict(extra_body)} -def _apply_custom_provider_extras( - custom_provider: Dict[str, Any], target_model: Optional[str], result: Dict[str, Any] -) -> None: +def _apply_custom_provider_extras(custom_provider: Dict[str, Any], target_model: Optional[str], result: Dict[str, Any]) -> None: """Copy model / capabilities / max_output_tokens / extra_headers / request_overrides onto a - resolved custom runtime. - - An explicit ``target_model`` wins over the provider's configured default (auxiliary slots / - background-review resolve a concrete model and must not fall back to ``default_model``). - ``extra_headers`` may carry credentials — NEVER log them. - """ + resolved custom runtime. An explicit ``target_model`` wins over the provider's configured + default (auxiliary slots / background-review resolve a concrete model and must not fall back to + ``default_model``). ``extra_headers`` may carry credentials — NEVER log them.""" model_name = target_model or custom_provider.get("model") if model_name: result["model"] = model_name @@ -463,52 +378,45 @@ def _apply_custom_provider_extras( result["extra_headers"] = dict(custom_provider["extra_headers"]) request_overrides = _custom_provider_request_overrides(custom_provider) if request_overrides: - result["request_overrides"] = {**dict(result.get("request_overrides") or {}), **request_overrides} + result["request_overrides"] = {**(result.get("request_overrides") or {}), **request_overrides} def _resolve_llamacpp_runtime(requested_provider: str, explicit_api_key: Optional[str]) -> Dict[str, Any]: """Managed llama.cpp runtime: the supervised (or detected external) server, or a typed error. - No server => say so and stop; falling through to the generic custom path would surface "local server is off" as OpenRouter's baffling "401 Invalid API key". The switch's state picks the - message (server off → point at the switch; else the setup pane). - """ + message (server off → point at the switch; else the setup pane).""" rp = _rp() try: from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint - endpoint = resolve_llamacpp_endpoint() except Exception: # noqa: BLE001 — resolution is best-effort endpoint = None if endpoint: - return rp._runtime( - "custom", - "chat_completions", - endpoint["base_url"], - (explicit_api_key or "").strip() or endpoint["api_key"] or "no-key-required", - source="local-runtime", - requested_provider=requested_provider, - ) + return rp._runtime("custom", "chat_completions", endpoint["base_url"], + (explicit_api_key or "").strip() or endpoint["api_key"] or "no-key-required", source="local-runtime", + requested_provider=requested_provider) try: enabled = bool((rp.load_config().get("local_runtime") or {}).get("enabled")) except Exception: # noqa: BLE001 enabled = False if enabled: - raise ValueError( - "The local model server isn't running. It may still be " - "starting — try again in a moment, or check Settings → " - "Providers → Local models." - ) - raise ValueError( - "The local model server is turned off. Turn it back on in " - "Settings → Providers → Local models, or switch to another " - "model." - ) + raise ValueError("The local model server isn't running. It may still be " + "starting — try again in a moment, or check Settings → " + "Providers → Local models.") + raise ValueError("The local model server is turned off. Turn it back on in " + "Settings → Providers → Local models, or switch to another " + "model.") -def _resolve_direct_alias_runtime( - requested_provider: str, explicit_api_key: Optional[str], explicit_base_url: str -) -> Dict[str, Any]: +def _custom_runtime(rp, base_url: str, api_key: Any, api_mode: Optional[str], **extra: Any) -> Dict[str, Any]: + """``custom`` runtime dict with URL-detected api_mode fallback and the no-auth placeholder.""" + return rp._runtime("custom", api_mode or rp._detect_api_mode_for_url(base_url) or "chat_completions", base_url, + api_key or "no-key-required", **extra) + + +def _resolve_direct_alias_runtime(requested_provider: str, explicit_api_key: Optional[str], + explicit_base_url: str) -> Dict[str, Any]: """Bare ``custom`` + explicit base_url (e.g. a ``model_aliases:`` direct alias).""" rp = _rp() base_url = explicit_base_url.strip().rstrip("/") @@ -521,21 +429,13 @@ def _resolve_direct_alias_runtime( # OLLAMA_API_KEY gets its own gate here: without it a `model_aliases:` entry pointing at # Ollama Cloud resolved no key at all. candidates = [(explicit_api_key or "").strip(), *rp._host_gated_env_key_candidates(base_url, ollama=True)] - api_key = next((c for c in candidates if rp.has_usable_secret(c)), "") or "no-key-required" - return rp._runtime( - "custom", - rp._detect_api_mode_for_url(base_url) or "chat_completions", - base_url, - api_key, - source="direct-alias", - requested_provider=requested_provider, - ) + api_key = next((c for c in candidates if rp.has_usable_secret(c)), "") + return _custom_runtime(rp, base_url, api_key, None, source="direct-alias", requested_provider=requested_provider) def _opencode_family_for_custom(requested_provider: str, base_url: str) -> Optional[str]: """OpenCode family by provider name, else by opencode.ai host (``/zen/go`` => opencode-go).""" from hermes_cli.models import opencode_provider_family - family = opencode_provider_family(requested_provider) if family is not None: return family @@ -547,87 +447,63 @@ def _opencode_family_for_custom(requested_provider: str, base_url: str) -> Optio return None -def _resolve_named_custom_runtime( - *, - requested_provider: str, - explicit_api_key: Optional[str] = None, - explicit_base_url: Optional[str] = None, - target_model: Optional[str] = None, -) -> Optional[Dict[str, Any]]: +def _resolve_named_custom_runtime(*, requested_provider: str, explicit_api_key: Optional[str] = None, + explicit_base_url: Optional[str] = None, + target_model: Optional[str] = None) -> Optional[Dict[str, Any]]: """Runtime for a llamacpp alias, a bare-custom direct alias, or a configured custom entry. - - Aliases resolving to "custom" (ollama, vllm, llamacpp, …) are treated like bare ``custom``. - A llamacpp alias with no explicit base_url resolves to the managed server first; an explicit - base_url always wins. - """ + Aliases resolving to "custom" (ollama, vllm, llamacpp, …) are treated like bare ``custom``. A + llamacpp alias with no explicit base_url resolves to the managed server first; an explicit + base_url always wins.""" rp = _rp() requested_norm = (requested_provider or "").strip().lower() - if requested_norm in ("llamacpp", "llama.cpp", "llama-cpp") and not explicit_base_url: + if requested_norm in _LLAMACPP_ALIASES and not explicit_base_url: return _resolve_llamacpp_runtime(requested_provider, explicit_api_key) if requested_norm and requested_norm != "custom" and rp._resolves_to_custom(requested_norm): requested_norm = "custom" if requested_norm == "custom" and explicit_base_url: return _resolve_direct_alias_runtime(requested_provider, explicit_api_key, explicit_base_url) - custom_provider = rp._get_named_custom_provider(requested_provider) if not custom_provider: return None base_url = ((explicit_base_url or "").strip() or custom_provider.get("base_url", "")).rstrip("/") if not base_url: return None - pool_result = rp._try_resolve_from_custom_pool( - base_url, - "custom", - custom_provider.get("api_mode"), + base_url, "custom", custom_provider.get("api_mode"), provider_name=custom_provider.get("provider_key") or custom_provider.get("name"), ) if pool_result: # The pool doesn't know the custom_providers fields — propagate them here too. _apply_custom_provider_extras(custom_provider, target_model, pool_result) return pool_result - + explicit_key = (explicit_api_key or "").strip() candidates = [ - (explicit_api_key or "").strip(), + explicit_key, _clean(custom_provider.get("api_key", "")), rp._getenv(_clean(custom_provider.get("key_env", "")), "").strip(), *rp._host_gated_env_key_candidates(base_url, ollama=False), ] api_key: Any = next((c for c in candidates if rp.has_usable_secret(c)), "") - # ``key_cmd`` credentials are minted per request (short-lived bearers would go stale # mid-session); both wire clients accept a callable api_key (the Entra ID contract). An # explicit --api-key still wins as the one-off recovery escape hatch. key_cmd = _clean(custom_provider.get("key_cmd", "")) - if key_cmd and not rp.has_usable_secret((explicit_api_key or "").strip()): + if key_cmd and not rp.has_usable_secret(explicit_key): from agent.command_token_source import build_command_token_provider - - token_provider = build_command_token_provider( - key_cmd, str(custom_provider.get("name", requested_provider) or "custom") - ) + token_provider = build_command_token_provider(key_cmd, str(custom_provider.get("name", requested_provider) or "custom")) if token_provider is not None: api_key = token_provider - - result = rp._runtime( - "custom", - custom_provider.get("api_mode") or rp._detect_api_mode_for_url(base_url) or "chat_completions", - base_url, - api_key or "no-key-required", - source=f"custom_provider:{custom_provider.get('name', requested_provider)}", - requested_provider=requested_provider, - ) + result = _custom_runtime(rp, base_url, api_key, custom_provider.get("api_mode"), + source=f"custom_provider:{custom_provider.get('name', requested_provider)}", + requested_provider=requested_provider) _apply_custom_provider_extras(custom_provider, target_model, result) - # OpenCode-family custom providers (opencode-go/zen names, or opencode.ai hosts) serve models # on different API surfaces — a static api_mode 503s for /v1/responses-only models. Re-derive # api_mode from the model and normalize /v1 like the built-in paths. family = _opencode_family_for_custom(requested_provider, base_url) if family is not None and not custom_provider.get("api_mode"): from hermes_cli.models import normalize_opencode_base_url, opencode_model_api_mode - - effective_model = str( - target_model or custom_provider.get("model") or rp._get_model_config().get("default") or "" - ).strip() + effective_model = str(target_model or custom_provider.get("model") or rp._get_model_config().get("default") or "").strip() if effective_model: result["api_mode"] = opencode_model_api_mode(family, effective_model) result["base_url"] = normalize_opencode_base_url(family, result["api_mode"], result["base_url"])