From 0f9815b75681da4c98362ca4d0d53a5b318b9a3d Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:03:07 -0700 Subject: [PATCH 01/19] =?UTF-8?q?refactor(hermes=5Fcli):=20providers.py=20?= =?UTF-8?q?=E2=80=94=20shared=20ProviderDef=20builders,=20packed=20overlay?= =?UTF-8?q?=20table,=20compact=20docs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/providers.py | 640 +++++++++++++--------------------------- 1 file changed, 208 insertions(+), 432 deletions(-) diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py index 306e433c89..32c52e1ed0 100644 --- a/hermes_cli/providers.py +++ b/hermes_cli/providers.py @@ -11,8 +11,7 @@ from utils import base_url_host_matches, base_url_hostname logger = logging.getLogger(__name__) -# -- Hermes overlay ---------------------------------------------------------- -# Hermes-specific metadata that models.dev doesn't provide. +# -- Hermes overlay: metadata models.dev doesn't provide ---------------------- @dataclass(frozen=True) class HermesOverlay: @@ -30,153 +29,76 @@ class HermesOverlay: HERMES_OVERLAYS: Dict[str, HermesOverlay] = { "moa": HermesOverlay(auth_type="virtual", base_url_override="moa://local"), "openrouter": HermesOverlay(is_aggregator=True, base_url_env_var="OPENROUTER_BASE_URL"), - "nous": HermesOverlay( - auth_type="oauth_device_code", - base_url_override="https://inference-api.nousresearch.com/v1", - ), - "openai-codex": HermesOverlay( - transport="codex_responses", - auth_type="oauth_external", - base_url_override="https://chatgpt.com/backend-api/codex", - ), - "openai-api": HermesOverlay( - transport="codex_responses", - base_url_override="https://api.openai.com/v1", - base_url_env_var="OPENAI_BASE_URL", - ), - "xai-oauth": HermesOverlay( - transport="codex_responses", - auth_type="oauth_external", - base_url_override="https://api.x.ai/v1", - base_url_env_var="XAI_BASE_URL", - ), - "qwen-oauth": HermesOverlay( - auth_type="oauth_external", - base_url_override="https://portal.qwen.ai/v1", - base_url_env_var="HERMES_QWEN_BASE_URL", - ), - "lmstudio": HermesOverlay( - extra_env_vars=("LM_API_KEY",), - base_url_override="http://127.0.0.1:1234/v1", - base_url_env_var="LM_BASE_URL", - ), - "copilot-acp": HermesOverlay( - transport="codex_responses", - auth_type="external_process", - base_url_override="acp://copilot", - base_url_env_var="COPILOT_ACP_BASE_URL", - ), + "nous": HermesOverlay(auth_type="oauth_device_code", base_url_override="https://inference-api.nousresearch.com/v1"), + "openai-codex": HermesOverlay(transport="codex_responses", auth_type="oauth_external", + base_url_override="https://chatgpt.com/backend-api/codex"), + "openai-api": HermesOverlay(transport="codex_responses", base_url_override="https://api.openai.com/v1", + base_url_env_var="OPENAI_BASE_URL"), + "xai-oauth": HermesOverlay(transport="codex_responses", auth_type="oauth_external", + base_url_override="https://api.x.ai/v1", base_url_env_var="XAI_BASE_URL"), + "qwen-oauth": HermesOverlay(auth_type="oauth_external", base_url_override="https://portal.qwen.ai/v1", + base_url_env_var="HERMES_QWEN_BASE_URL"), + "lmstudio": HermesOverlay(extra_env_vars=("LM_API_KEY",), base_url_override="http://127.0.0.1:1234/v1", + base_url_env_var="LM_BASE_URL"), + "copilot-acp": HermesOverlay(transport="codex_responses", auth_type="external_process", + base_url_override="acp://copilot", base_url_env_var="COPILOT_ACP_BASE_URL"), "github-copilot": HermesOverlay(extra_env_vars=("COPILOT_GITHUB_TOKEN", "GH_TOKEN")), - "anthropic": HermesOverlay( - transport="anthropic_messages", - extra_env_vars=("ANTHROPIC_TOKEN", "CLAUDE_CODE_OAUTH_TOKEN"), - ), - "zai": HermesOverlay( - extra_env_vars=("GLM_API_KEY", "ZAI_API_KEY", "Z_AI_API_KEY"), - base_url_env_var="GLM_BASE_URL", - ), + "anthropic": HermesOverlay(transport="anthropic_messages", extra_env_vars=("ANTHROPIC_TOKEN", "CLAUDE_CODE_OAUTH_TOKEN")), + "zai": HermesOverlay(extra_env_vars=("GLM_API_KEY", "ZAI_API_KEY", "Z_AI_API_KEY"), base_url_env_var="GLM_BASE_URL"), "kimi-for-coding": HermesOverlay(base_url_env_var="KIMI_BASE_URL"), - "stepfun": HermesOverlay( - extra_env_vars=("STEPFUN_API_KEY",), - base_url_override="https://api.stepfun.ai/step_plan/v1", - base_url_env_var="STEPFUN_BASE_URL", - ), + "stepfun": HermesOverlay(extra_env_vars=("STEPFUN_API_KEY",), + base_url_override="https://api.stepfun.ai/step_plan/v1", + base_url_env_var="STEPFUN_BASE_URL"), "minimax": HermesOverlay(transport="anthropic_messages", base_url_env_var="MINIMAX_BASE_URL"), - "minimax-oauth": HermesOverlay( - transport="anthropic_messages", - auth_type="oauth_external", - base_url_override="https://api.minimax.io/anthropic", - ), - "minimax-cn": HermesOverlay( - transport="anthropic_messages", - base_url_env_var="MINIMAX_CN_BASE_URL", - ), + "minimax-oauth": HermesOverlay(transport="anthropic_messages", auth_type="oauth_external", + base_url_override="https://api.minimax.io/anthropic"), + "minimax-cn": HermesOverlay(transport="anthropic_messages", base_url_env_var="MINIMAX_CN_BASE_URL"), "deepseek": HermesOverlay(base_url_env_var="DEEPSEEK_BASE_URL"), "alibaba": HermesOverlay(base_url_env_var="DASHSCOPE_BASE_URL"), "alibaba-coding-plan": HermesOverlay(base_url_env_var="ALIBABA_CODING_PLAN_BASE_URL"), "vercel": HermesOverlay(is_aggregator=True), "opencode": HermesOverlay(is_aggregator=True, base_url_env_var="OPENCODE_ZEN_BASE_URL"), "opencode-go": HermesOverlay(is_aggregator=True, base_url_env_var="OPENCODE_GO_BASE_URL"), - "opencode-free": HermesOverlay( - is_aggregator=True, - base_url_override="https://opencode.ai/zen/v1", - keyless=True, - ), + "opencode-free": HermesOverlay(is_aggregator=True, base_url_override="https://opencode.ai/zen/v1", keyless=True), "kilo": HermesOverlay(is_aggregator=True, base_url_env_var="KILOCODE_BASE_URL"), "huggingface": HermesOverlay(is_aggregator=True, base_url_env_var="HF_BASE_URL"), "novita": HermesOverlay(is_aggregator=True, base_url_env_var="NOVITA_BASE_URL"), - "xai": HermesOverlay( - transport="codex_responses", - base_url_override="https://api.x.ai/v1", - base_url_env_var="XAI_BASE_URL", - ), - "nvidia": HermesOverlay( - base_url_override="https://integrate.api.nvidia.com/v1", - base_url_env_var="NVIDIA_BASE_URL", - ), + "xai": HermesOverlay(transport="codex_responses", base_url_override="https://api.x.ai/v1", base_url_env_var="XAI_BASE_URL"), + "nvidia": HermesOverlay(base_url_override="https://integrate.api.nvidia.com/v1", base_url_env_var="NVIDIA_BASE_URL"), "xiaomi": HermesOverlay(base_url_env_var="XIAOMI_BASE_URL"), "tencent-tokenhub": HermesOverlay(base_url_env_var="TOKENHUB_BASE_URL"), - "tencent-tokenplan": HermesOverlay( - transport="anthropic_messages", - base_url_override="https://api.lkeap.cloud.tencent.com/plan/anthropic", - base_url_env_var="TOKENPLAN_BASE_URL", - ), - "arcee": HermesOverlay( - base_url_override="https://api.arcee.ai/api/v1", - base_url_env_var="ARCEE_BASE_URL", - ), - "gmi": HermesOverlay( - extra_env_vars=("GMI_API_KEY",), - base_url_override="https://api.gmi-serving.com/v1", - base_url_env_var="GMI_BASE_URL", - ), - "fireworks": HermesOverlay( - extra_env_vars=("FIREWORKS_API_KEY",), - base_url_override="https://api.fireworks.ai/inference/v1", - ), - "actual": HermesOverlay( - transport="codex_responses", - extra_env_vars=("ACTUAL_API_KEY", "ACTUAL_BASE_URL"), - base_url_override="https://api.actual.inc/v1", - base_url_env_var="ACTUAL_BASE_URL", - ), - "upstage": HermesOverlay( - extra_env_vars=("UPSTAGE_API_KEY",), - base_url_override="https://api.upstage.ai/v1", - base_url_env_var="UPSTAGE_BASE_URL", - ), - "nebius-token-factory": HermesOverlay( - extra_env_vars=("NEBIUS_API_KEY", "NEBIUS_TOKEN_FACTORY_API_KEY"), - base_url_override="https://api.tokenfactory.nebius.com/v1", - base_url_env_var="NEBIUS_BASE_URL", - ), - "ollama-cloud": HermesOverlay( - base_url_override="https://ollama.com/v1", - base_url_env_var="OLLAMA_BASE_URL", - ), - # Azure Foundry: supports both OpenAI-style and Anthropic-style endpoints. - # The transport is determined at runtime from config.yaml model.api_mode. - "azure-foundry": HermesOverlay(base_url_env_var="AZURE_FOUNDRY_BASE_URL"), # openai_chat default; api_mode overrides + "tencent-tokenplan": HermesOverlay(transport="anthropic_messages", + base_url_override="https://api.lkeap.cloud.tencent.com/plan/anthropic", + base_url_env_var="TOKENPLAN_BASE_URL"), + "arcee": HermesOverlay(base_url_override="https://api.arcee.ai/api/v1", base_url_env_var="ARCEE_BASE_URL"), + "gmi": HermesOverlay(extra_env_vars=("GMI_API_KEY",), base_url_override="https://api.gmi-serving.com/v1", + base_url_env_var="GMI_BASE_URL"), + "fireworks": HermesOverlay(extra_env_vars=("FIREWORKS_API_KEY",), + base_url_override="https://api.fireworks.ai/inference/v1"), + "actual": HermesOverlay(transport="codex_responses", extra_env_vars=("ACTUAL_API_KEY", "ACTUAL_BASE_URL"), + base_url_override="https://api.actual.inc/v1", base_url_env_var="ACTUAL_BASE_URL"), + "upstage": HermesOverlay(extra_env_vars=("UPSTAGE_API_KEY",), base_url_override="https://api.upstage.ai/v1", + base_url_env_var="UPSTAGE_BASE_URL"), + "nebius-token-factory": HermesOverlay(extra_env_vars=("NEBIUS_API_KEY", "NEBIUS_TOKEN_FACTORY_API_KEY"), + base_url_override="https://api.tokenfactory.nebius.com/v1", + base_url_env_var="NEBIUS_BASE_URL"), + "ollama-cloud": HermesOverlay(base_url_override="https://ollama.com/v1", base_url_env_var="OLLAMA_BASE_URL"), + # Azure Foundry serves OpenAI- and Anthropic-style endpoints; transport comes from model.api_mode. + "azure-foundry": HermesOverlay(base_url_env_var="AZURE_FOUNDRY_BASE_URL"), "bedrock": HermesOverlay(transport="bedrock_converse", auth_type="aws_sdk"), - # Vertex authenticates via OAuth2 (service-account JSON / ADC), not a - # static API key or models.dev entry — resolved specially by - # agent/vertex_adapter.py, like bedrock's aws_sdk. Without an overlay - # entry get_provider("vertex") returns None, which makes - # _preserve_provider_with_base_url() in agent/auxiliary_client.py treat - # a Vertex MoA slot's resolved (base_url, api_key) pair as an unknown - # custom endpoint instead of "vertex" — losing the provider identity - # that _refresh_provider_credentials() needs to re-mint an expired - # OAuth2 token on a 401. + # Vertex is OAuth2 (service-account JSON / ADC), resolved by agent/vertex_adapter.py. Without an + # overlay get_provider("vertex") is None and auxiliary_client._preserve_provider_with_base_url + # would treat a Vertex MoA slot as an unknown custom endpoint, losing the identity + # _refresh_provider_credentials() needs to re-mint an expired token on 401. "vertex": HermesOverlay(auth_type="vertex"), } # -- Resolved provider ------------------------------------------------------- -# The merged result of models.dev + overlay + user config. @dataclass class ProviderDef: - """Complete provider definition — merged from all sources.""" + """Complete provider definition — merged from models.dev + overlay + user config.""" id: str name: str @@ -190,11 +112,8 @@ class ProviderDef: source: str = "" # "models.dev", "hermes", "user-config" -# -- Aliases ------------------------------------------------------------------ -# Maps human-friendly / legacy names to canonical provider IDs. -# Uses models.dev IDs where possible. - -# Aliases grouped by canonical provider id; ``ALIASES`` is the inverted lookup table. +# -- Aliases: human-friendly / legacy names grouped by canonical (models.dev where possible) id; +# ``ALIASES`` is the inverted lookup table. --------------------------------------------------- _ALIAS_GROUPS: Dict[str, Tuple[str, ...]] = { "openrouter": ("openai",), "zai": ("glm", "z-ai", "z.ai", "zhipu"), @@ -226,9 +145,7 @@ _ALIAS_GROUPS: Dict[str, Tuple[str, ...]] = { "fireworks": ("fireworks-ai", "fw"), "upstage": ("solar",), "actual": ("actual-computer", "actualcomputer", "aci"), - "nebius-token-factory": ( - "nebius", "nebius-tokenfactory", "nebius-tf", "token-factory", "tokenfactory", - ), + "nebius-token-factory": ("nebius", "nebius-tokenfactory", "nebius-tf", "token-factory", "tokenfactory"), "lmstudio": ("lmstudio", "lm-studio", "lm_studio"), "custom": ("ollama",), "local": ("vllm", "llamacpp", "llama.cpp", "llama-cpp"), @@ -236,9 +153,7 @@ _ALIAS_GROUPS: Dict[str, Tuple[str, ...]] = { ALIASES: Dict[str, str] = {alias: canon for canon, aliases in _ALIAS_GROUPS.items() for alias in aliases} -# -- Display labels ----------------------------------------------------------- -# Built dynamically from models.dev + overlays. Fallback for providers -# not in the catalog. +# -- Display labels for providers not in the models.dev catalog --------------- _LABEL_OVERRIDES: Dict[str, str] = { "moa": "Mixture of Agents", @@ -281,111 +196,77 @@ def normalize_provider(name: str) -> str: return ALIASES.get(key, key) +def _models_dev_info(canonical: str, allow_network: bool = True): + """models.dev entry or None. Single-arg call on the default path: test sites monkeypatch + ``get_provider_info`` with single-arg lambdas.""" + try: + from agent.models_dev import get_provider_info as _mdev_provider + + return _mdev_provider(canonical) if allow_network else _mdev_provider(canonical, allow_network=False) + except Exception: + return None + + def get_provider(name: str, *, allow_network: bool = True) -> Optional[ProviderDef]: """Look up a built-in provider by id or alias. - Resolution order: 1. Hermes overlays (for providers not in models.dev: nous, openai-codex, etc.) - 2. models.dev catalog + Hermes overlay + Order: models.dev catalog merged with the Hermes overlay; Hermes-only overlay (nous, + openai-codex, …); plugin provider profiles with a concrete endpoint. """ canonical = normalize_provider(name) - - # Try to get models.dev data - try: - from agent.models_dev import get_provider_info as _mdev_provider - # Keep the single-argument call on the default path: test sites - # monkeypatch get_provider_info with single-arg lambdas. - mdev_info = ( - _mdev_provider(canonical) - if allow_network - else _mdev_provider(canonical, allow_network=False) - ) - except Exception: - mdev_info = None - + mdev_info = _models_dev_info(canonical, allow_network) overlay = HERMES_OVERLAYS.get(canonical) if mdev_info is not None: - # Merge models.dev + overlay (defaults when no overlay); env vars = models.dev + hermes extra ov = overlay or HermesOverlay() env_vars = list(mdev_info.env) for ev in ov.extra_env_vars: if ev not in env_vars: env_vars.append(ev) return ProviderDef( - id=canonical, - name=mdev_info.name, - transport=ov.transport, - api_key_env_vars=tuple(env_vars), - base_url=ov.base_url_override or mdev_info.api, - base_url_env_var=ov.base_url_env_var, - is_aggregator=ov.is_aggregator, - auth_type=ov.auth_type, - doc=mdev_info.doc, - source="models.dev", + id=canonical, name=mdev_info.name, transport=ov.transport, api_key_env_vars=tuple(env_vars), + base_url=ov.base_url_override or mdev_info.api, base_url_env_var=ov.base_url_env_var, + is_aggregator=ov.is_aggregator, auth_type=ov.auth_type, doc=mdev_info.doc, source="models.dev", ) if overlay is not None: - # Hermes-only provider (not in models.dev) return ProviderDef( - id=canonical, - name=_LABEL_OVERRIDES.get(canonical, canonical), - transport=overlay.transport, - api_key_env_vars=overlay.extra_env_vars, - base_url=overlay.base_url_override, - base_url_env_var=overlay.base_url_env_var, - is_aggregator=overlay.is_aggregator, - auth_type=overlay.auth_type, - source="hermes", + id=canonical, name=_LABEL_OVERRIDES.get(canonical, canonical), transport=overlay.transport, + api_key_env_vars=overlay.extra_env_vars, base_url=overlay.base_url_override, + base_url_env_var=overlay.base_url_env_var, is_aggregator=overlay.is_aggregator, + auth_type=overlay.auth_type, source="hermes", ) - # Plugin-registered provider profiles (plugins/model-providers//). - # Providers that ship only as plugin profiles (e.g. commandcode, - # tencent-tokenhub) are absent from models.dev and HERMES_OVERLAYS, so - # without this fallback they resolve as "Unknown provider" in /model, - # --provider, and the model-switch path even though the picker lists them - # (CANONICAL_PROVIDERS auto-extends from the same plugin registry). + # Plugin-registered profiles (plugins/model-providers//) absent from models.dev and + # HERMES_OVERLAYS would otherwise be "Unknown provider" in /model, --provider and model-switch + # even though the picker lists them. Only profiles with a concrete endpoint resolve here: + # placeholder profiles like ``custom`` (aliases ollama/local/vllm) ship an empty base_url and + # are completed by config.yaml custom_providers — resolving them would preempt + # resolve_provider_full's custom step and collapse keyed ``custom:`` ids to bare custom. try: from providers import get_provider_profile as _profile _prof = _profile(canonical) - # Only profiles with a concrete endpoint resolve here. Placeholder - # profiles like ``custom`` (aliases: ollama/local/vllm) ship with an - # empty base_url and are completed by config.yaml custom_providers — - # resolving them here would preempt resolve_provider_full's - # custom-provider step and collapse keyed IDs - # (``custom:local-...``) back to a bare, endpoint-less ``custom``. if _prof is not None and (_prof.base_url or "").strip(): _api_mode_to_transport = {v: k for k, v in TRANSPORT_TO_API_MODE.items()} - _transport = _api_mode_to_transport.get(_prof.api_mode, "openai_chat") return ProviderDef( - id=canonical, - name=_prof.display_name or _prof.name or canonical, - transport=_transport, - api_key_env_vars=tuple(_prof.env_vars or ()), - base_url=_prof.base_url or "", - auth_type=_prof.auth_type or "api_key", - source="plugin-profile", + id=canonical, name=_prof.display_name or _prof.name or canonical, + transport=_api_mode_to_transport.get(_prof.api_mode, "openai_chat"), + api_key_env_vars=tuple(_prof.env_vars or ()), base_url=_prof.base_url or "", + auth_type=_prof.auth_type or "api_key", source="plugin-profile", ) except Exception: pass - return None def get_label(provider_id: str) -> str: - """Get a human-readable display name for a provider.""" + """Human-readable display name: label override, else models.dev name, else the id.""" canonical = normalize_provider(provider_id) - - # Check label overrides first if canonical in _LABEL_OVERRIDES: return _LABEL_OVERRIDES[canonical] - - # Try models.dev pdef = get_provider(canonical) - if pdef: - return pdef.name - - return canonical + return pdef.name if pdef else canonical def is_aggregator(provider: str) -> bool: @@ -397,30 +278,19 @@ def is_aggregator(provider: str) -> bool: return pdef.is_aggregator if pdef else False -# Flat-namespace resellers (e.g. opencode-go, opencode-zen) are flagged -# ``is_aggregator=True`` because their live ``/v1/models`` returns bare model -# IDs ("deepseek-v4-flash") rather than ``vendor/model`` routing slugs — the -# model-switch resolver relies on that flag to search their flat catalog -# (see model_switch.py step d). But they are NOT routing aggregators: every -# model they list is a first-party model served under their own subscription, -# not a passthrough route to another provider's endpoint. The picker dedup -# (build_models_payload) must treat them differently from true routers like -# OpenRouter — a reseller's first-party "minimax-m3" must never be stripped -# just because a user's custom proxy also happens to serve a same-named model. -_FLAT_NAMESPACE_RESELLERS: frozenset[str] = frozenset({ - # Use normalized provider IDs: normalize_provider("opencode-zen") -> "opencode". - "opencode-go", - "opencode", -}) +# Flat-namespace resellers (opencode-go, opencode-zen) are flagged ``is_aggregator=True`` because +# their live ``/v1/models`` returns bare model IDs ("deepseek-v4-flash") rather than +# ``vendor/model`` routing slugs — model_switch searches their flat catalog on that flag. But they +# are NOT routing aggregators: every listed model is first-party under their own subscription, so +# picker dedup (build_models_payload) must not strip a reseller's "minimax-m3" just because a +# user's custom proxy serves a same-named model. Normalized ids: "opencode-zen" -> "opencode". +_FLAT_NAMESPACE_RESELLERS: frozenset[str] = frozenset({"opencode-go", "opencode"}) def is_routing_aggregator(provider: str) -> bool: - """True only for TRUE routing aggregators (OpenRouter, named ``custom:*`` proxies). - - Unlike ``is_aggregator``, excludes flat-namespace resellers (opencode-go/zen) whose catalog is - first-party. Use for "would selecting this model silently re-route away from the intended - provider?" -- i.e. picker dedup; reseller rows must not be deduped against user proxies. - """ + """True only for TRUE routing aggregators (OpenRouter, named ``custom:*`` proxies) — excludes + flat-namespace resellers whose catalog is first-party. Use for "would selecting this model + silently re-route away from the intended provider?" (picker dedup).""" provider_norm = normalize_provider(provider or "") if provider_norm in _FLAT_NAMESPACE_RESELLERS: return False @@ -432,150 +302,106 @@ def is_official_openai_host(base_url: str) -> bool: Hostname-parsed matching only — never substring — so lookalike hosts (``api.openai.com.attacker.test``) and path-segment spoofs (``proxy.test/api.openai.com/v1``) - are rejected. A genuine ``*.api.openai.com`` subdomain requires control of openai.com DNS, so - the dot-suffix match does not reopen the #32243 spoofing hole. + are rejected. A genuine ``*.api.openai.com`` subdomain requires control of openai.com DNS. """ return base_url_host_matches(base_url, "api.openai.com") +# Exact hostnames that are Responses-API-native: api.meta.ai only achieves prompt-cache hits on +# Responses with prompt_cache_retention (chat/completions stays cache-cold); api.router.com (Ramp +# Router) keeps reasoning validation/summaries and prompt caching on /v1/responses and serves +# /v1/chat/completions as a minimal shim. +_RESPONSES_NATIVE_HOSTS: frozenset[str] = frozenset({"api.meta.ai", "api.router.com"}) + + def host_mandated_api_mode(base_url: str = "") -> Optional[str]: """Return the wire protocol a specific endpoint *requires*, or None. - Some hosts only accept one API mode and reject the others outright: - api.openai.com only - accepts the Responses API for its (reasoning) models when tools + reasoning are in play - (chat/completions 400s). - - These are *mandatory* — a session carrying a stale api_mode (e.g. a /model switch that kept the - previous provider's ``chat_completions``) must be overridden to the host's required mode, not - merely filled in when empty. + Some hosts accept exactly one API mode (api.openai.com 400s chat/completions for reasoning + models with tools). These are *mandatory*: a session carrying a stale api_mode (a /model switch + that kept the previous provider's ``chat_completions``) must be overridden, not merely filled + in when empty. Exact-hostname matching only — never substring — so lookalike hosts and + path-segment spoofs are not treated as the real endpoint. """ if not base_url: return None url_lower = base_url.rstrip("/").lower() hostname = base_url_hostname(base_url) - # Exact-hostname matching only — never bare substring — so lookalike hosts - # (api.openai.com.attacker.test) and path-segment spoofs - # (proxy.test/api.openai.com/v1) are NOT treated as the real endpoint. (#32243) if hostname == "api.kimi.com" and "/coding" in url_lower: return "anthropic_messages" if hostname == "api.anthropic.com" or url_lower.endswith("/anthropic"): return "anthropic_messages" - # Official OpenAI host family: canonical + data-residency regional hosts - # (us./eu.api.openai.com) all mandate the Responses API for reasoning - # models with tools. Shared predicate keeps this lane in lockstep with - # catalog filtering and listing authority. - if is_official_openai_host(base_url): - return "codex_responses" - if hostname in _RESPONSES_NATIVE_HOSTS: + # Official OpenAI host family (canonical + us./eu. data-residency hosts) mandates Responses; + # the shared predicate keeps this in lockstep with catalog filtering and listing authority. + if is_official_openai_host(base_url) or hostname in _RESPONSES_NATIVE_HOSTS: return "codex_responses" if hostname.startswith("bedrock-runtime.") and base_url_host_matches(base_url, "amazonaws.com"): return "bedrock_converse" return None -# Exact hostnames (#32243) that are Responses-API-native: -# - api.meta.ai: Meta Model API only achieves prompt-cache hits on the Responses API with -# prompt_cache_retention; chat/completions stays cache-cold (0% vs 93-99% measured). -# - api.router.com: Ramp Router keeps reasoning-effort validation, reasoning summaries and prompt -# caching on /v1/responses; /v1/chat/completions is a minimal shim (docs.router.com/api/endpoint). -_RESPONSES_NATIVE_HOSTS: frozenset[str] = frozenset({"api.meta.ai", "api.router.com"}) - - def nous_api_mode(model: str = "") -> str: - """Resolve the wire protocol for a Nous Portal model. - - Portal serves its ``anthropic/*`` catalog on a native Anthropic Messages route - (``/v1/messages``) alongside the OpenAI-compatible ``/v1/chat/completions`` used by every other - model it proxies. - - When *model* is empty/unknown, defaults to ``chat_completions`` — the historical Nous transport - — so callers that don't yet know the model stay on the safer OpenAI-compatible path. - """ + """Wire protocol for a Nous Portal model: Portal serves its ``anthropic/*`` catalog on a native + Messages route alongside OpenAI-compatible chat/completions for everything else. Empty/unknown + model defaults to ``chat_completions`` (the historical Nous transport) as the safer path.""" if str(model or "").strip().lower().startswith("anthropic/"): return "anthropic_messages" return "chat_completions" def determine_api_mode(provider: str, base_url: str = "", model: str = "") -> str: - """Determine the API mode (wire protocol) for a provider/endpoint. - - Resolution order: 1. Host-mandated mode (special endpoints that only accept one protocol). 2. - Nous Portal dual-wire (model-derived; overlay alone is openai_chat). 3. Known provider → - transport → TRANSPORT_TO_API_MODE. 4. Direct provider checks (bedrock). 5. Default: - 'chat_completions'. - """ + """API mode (wire protocol) for a provider/endpoint: host-mandated mode, then Nous dual-wire + (model-derived — the overlay alone says openai_chat and would pin Claude on the wrong wire), + then the known provider's transport, then bedrock, else ``chat_completions``.""" mandated = host_mandated_api_mode(base_url) if mandated is not None: return mandated - - # Nous is dual-wire: anthropic/* → Messages, everything else → - # chat_completions. The Hermes overlay still advertises openai_chat - # (the majority of the Portal catalog), so the transport lookup below - # would pin Claude on the wrong wire without this carve-out. - provider_norm = (provider or "").strip().lower() - if provider_norm in {"nous", "nous-portal", "nousresearch"}: + if (provider or "").strip().lower() in {"nous", "nous-portal", "nousresearch"}: return nous_api_mode(model) - pdef = get_provider(provider) if pdef is not None: return TRANSPORT_TO_API_MODE.get(pdef.transport, "chat_completions") - - # Direct provider checks for providers not in HERMES_OVERLAYS if provider == "bedrock": return "bedrock_converse" - return "chat_completions" # -- Provider from user config ------------------------------------------------ +def _user_pdef(pid: str, name: str, base_url: str, key_env: str, transport: str = "openai_chat") -> ProviderDef: + """``source="user-config"`` ProviderDef shared by ``providers:`` and ``custom_providers:`` entries.""" + return ProviderDef( + id=pid, name=name, transport=transport, api_key_env_vars=(key_env,) if key_env else (), + base_url=base_url, is_aggregator=False, auth_type="api_key", source="user-config", + ) + + def resolve_user_provider(name: str, user_config: Dict[str, Any]) -> Optional[ProviderDef]: """Resolve a provider from the user's config.yaml ``providers:`` section.""" if not user_config or not isinstance(user_config, dict): return None - entry = user_config.get(name) if not isinstance(entry, dict): return None - - # Extract fields - display_name = entry.get("name", "") or name - api_url = entry.get("api", "") or entry.get("url", "") or entry.get("base_url", "") or "" - key_env = entry.get("key_env") or entry.get("api_key_env") or "" - transport = entry.get("transport", "openai_chat") or "openai_chat" - - env_vars: List[str] = [] - if key_env: - env_vars.append(key_env) - - return ProviderDef( - id=name, - name=display_name, - transport=transport, - api_key_env_vars=tuple(env_vars), - base_url=api_url, - is_aggregator=False, - auth_type="api_key", - source="user-config", + return _user_pdef( + name, + entry.get("name", "") or name, + entry.get("api", "") or entry.get("url", "") or entry.get("base_url", "") or "", + entry.get("key_env") or entry.get("api_key_env") or "", + entry.get("transport", "openai_chat") or "openai_chat", ) def custom_provider_slug(display_name: str, provider_key: str = "") -> str: - """Build the stable ``custom:`` identity for a configured provider. - - Keyed ``providers:`` entries use their config key so the identity survives display-name - changes; legacy ``custom_providers:`` entries have no key, so their normalized display name - remains the identity. - """ + """Stable ``custom:`` identity for a configured provider: keyed ``providers:`` entries use their + config key (survives display-name changes); legacy ``custom_providers:`` entries have no key, + so their normalized display name is the identity.""" identity = str(provider_key or "").strip() or str(display_name or "").strip() normalized = identity.lower().replace(" ", "-") return normalized if normalized.startswith("custom:") else f"custom:{normalized}" -def custom_provider_aliases( - display_name: str, - provider_key: str = "", -) -> frozenset[str]: +def custom_provider_aliases(display_name: str, provider_key: str = "") -> frozenset[str]: """Return every current and legacy identity accepted for one endpoint.""" aliases: set[str] = set() for value in (display_name, provider_key): @@ -591,175 +417,125 @@ def custom_provider_aliases( return frozenset(aliases) -def resolve_custom_provider( - name: str, - custom_providers: Optional[List[Dict[str, Any]]], -) -> Optional[ProviderDef]: - """Resolve a provider from the user's config.yaml ``custom_providers`` list.""" +def resolve_custom_provider(name: str, custom_providers: Optional[List[Dict[str, Any]]]) -> Optional[ProviderDef]: + """Resolve a provider from the user's config.yaml ``custom_providers`` list. + + A stored bare ``"custom"`` (corrupt state from a prior model-switch bug) falls back to the first + valid entry so existing configs self-heal. + """ if not custom_providers or not isinstance(custom_providers, list): return None - requested = (name or "").strip().lower() if not requested: return None - - # If the stored provider is the bare string "custom" (corrupt state - # from a prior model-switch bug), fall back to the first custom - # provider entry so existing configs self-heal. (GH #17478) - bare_custom_fallback = requested == "custom" first_valid: Optional[ProviderDef] = None - for entry in custom_providers: if not isinstance(entry, dict): continue - display_name = (entry.get("name") or "").strip() - api_url = ( - entry.get("base_url", "") - or entry.get("url", "") - or entry.get("api", "") - or "" - ).strip() + api_url = (entry.get("base_url", "") or entry.get("url", "") or entry.get("api", "") or "").strip() if not display_name or not api_url: continue - - key_env = (entry.get("key_env") or "").strip() provider_key = (entry.get("provider_key") or "").strip() - pdef = ProviderDef( - id=custom_provider_slug(display_name, provider_key), - name=display_name, - transport="openai_chat", - api_key_env_vars=(key_env,) if key_env else (), - base_url=api_url, - is_aggregator=False, - auth_type="api_key", - source="user-config", + pdef = _user_pdef( + custom_provider_slug(display_name, provider_key), display_name, api_url, (entry.get("key_env") or "").strip() ) - - # Stash the first valid entry for bare-"custom" fallback if first_valid is None: first_valid = pdef - if requested in custom_provider_aliases(display_name, provider_key): return pdef - - # Self-heal: bare "custom" matched nothing — return first valid entry - if bare_custom_fallback and first_valid: + if requested == "custom" and first_valid: return first_valid - return None +def _lossy_alias_registry_pdef(raw: str, canonical: str) -> Optional[ProviderDef]: + """Exact Hermes registry ids win over LOSSY alias collapsing (kimi-coding-cn must stay distinct + from kimi-coding instead of collapsing through the shared models.dev alias "kimi-for-coding"). + A collapse is lossy only when MULTIPLE registry providers normalize to the same canonical name; + single-entry rewrites ("copilot" -> "github-copilot") are correct routing and keep resolving + through the built-in chain so overlay transports apply.""" + try: + from hermes_cli.auth import PROVIDER_REGISTRY as _AUTH_PROVIDER_REGISTRY + + _pcfg = _AUTH_PROVIDER_REGISTRY.get(raw) + if _pcfg is None: + return None + if sum(1 for _rid in _AUTH_PROVIDER_REGISTRY if normalize_provider(_rid) == canonical) > 1: + return ProviderDef( + id=_pcfg.id, name=_pcfg.name, transport="openai_chat", + api_key_env_vars=tuple(_pcfg.api_key_env_vars or ()), base_url=_pcfg.inference_base_url or "", + source="hermes-auth-registry", + ) + except Exception: + pass + return None + + +def _llamacpp_pdef() -> Optional[ProviderDef]: + """The llamacpp aliases are a real provider whenever the managed server (or a detected external + one) resolves — reachability is the credential. Without this rung model-switch rejected the very + provider the Local Models 'Use' flow writes to config.""" + try: + from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint + + endpoint = resolve_llamacpp_endpoint(wait_for_boot_s=0) + except Exception: + endpoint = None + if not endpoint: + return None + return ProviderDef( + id="llamacpp", name="Local", transport="openai_chat", api_key_env_vars=(), base_url=endpoint["base_url"], + source="local-runtime", + ) + + def resolve_provider_full( name: str, user_providers: Optional[Dict[str, Any]] = None, custom_providers: Optional[List[Dict[str, Any]]] = None, ) -> Optional[ProviderDef]: - """Full resolution chain: built-in → models.dev → user config.""" + """Full resolution chain: user ``providers.`` -> lossy-alias registry id -> built-in + (models.dev + overlays) -> user providers (canonical, then raw) -> ``custom_providers`` -> + managed llamacpp -> models.dev directly. + + User-defined ``providers.`` is tried FIRST on the raw (pre-alias) name: a configured + ``providers.openai`` pointing at api.openai.com must not be hijacked by the legacy + "openai" -> "openrouter" alias. + """ canonical = normalize_provider(name) raw = name.strip().lower() - # 0. User-defined config providers win over the built-in alias table. - # A user who declares ``providers.`` in config.yaml has stated - # explicit intent for that name — it must not be hijacked by a legacy - # vendor alias (e.g. bare "openai" → "openrouter"). Resolve the raw - # name against user config FIRST so a configured ``providers.openai`` - # (pointing at api.openai.com) beats the alias that would otherwise - # silently route to OpenRouter. Only the raw (pre-alias) name is tried - # here; canonical/alias resolution still happens below. if user_providers: user_pdef = resolve_user_provider(raw, user_providers) if user_pdef is not None: return user_pdef - - # 0.5 Exact Hermes provider IDs must win over LOSSY alias collapsing. - # Example: kimi-coding-cn should stay distinct from kimi-coding instead of - # normalizing through the shared models.dev alias "kimi-for-coding". - # A collapse is lossy only when MULTIPLE distinct registry providers - # normalize to the same canonical name — resolving through the alias - # would then lose which one the caller meant. Single-entry rewrites - # (e.g. "copilot" → "github-copilot") are correct routing and must keep - # resolving through the built-in chain below so overlay transports apply. if canonical != raw: - try: - from hermes_cli.auth import PROVIDER_REGISTRY as _AUTH_PROVIDER_REGISTRY - _pcfg = _AUTH_PROVIDER_REGISTRY.get(raw) - if _pcfg is not None: - _collapsed_siblings = [ - _rid - for _rid in _AUTH_PROVIDER_REGISTRY - if normalize_provider(_rid) == canonical - ] - if len(_collapsed_siblings) > 1: - return ProviderDef( - id=_pcfg.id, - name=_pcfg.name, - transport="openai_chat", - api_key_env_vars=tuple(_pcfg.api_key_env_vars or ()), - base_url=_pcfg.inference_base_url or "", - source="hermes-auth-registry", - ) - except Exception: - pass - - # 1. Built-in (models.dev + overlays) + pdef = _lossy_alias_registry_pdef(raw, canonical) + if pdef is not None: + return pdef pdef = get_provider(canonical) if pdef is not None: return pdef - - # 2. User-defined providers from config if user_providers: - # Try canonical name - user_pdef = resolve_user_provider(canonical, user_providers) - if user_pdef is not None: - return user_pdef - # Try original name (in case alias didn't match) - user_pdef = resolve_user_provider(raw, user_providers) - if user_pdef is not None: - return user_pdef - - # 2b. Saved custom providers from config + for candidate in (canonical, raw): + user_pdef = resolve_user_provider(candidate, user_providers) + if user_pdef is not None: + return user_pdef custom_pdef = resolve_custom_provider(name, custom_providers) if custom_pdef is not None: return custom_pdef - - # 2c. Managed local runtime: the llamacpp aliases are a real provider - # whenever the managed server (or a detected external one) resolves — - # no credential and no providers: entry required, the credential is - # reachability. Without this rung the model-switch path rejected the - # very provider the Local Models 'Use' flow writes to config - # ("Unknown provider 'llamacpp'" from the desktop dropdown). if raw in ("llamacpp", "llama.cpp", "llama-cpp"): - try: - from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint - - endpoint = resolve_llamacpp_endpoint(wait_for_boot_s=0) - except Exception: - endpoint = None - if endpoint: - return ProviderDef( - id="llamacpp", - name="Local", - transport="openai_chat", - api_key_env_vars=(), - base_url=endpoint["base_url"], - source="local-runtime", - ) - - # 3. Try models.dev directly (for providers not in our ALIASES) + pdef = _llamacpp_pdef() + if pdef is not None: + return pdef try: - from agent.models_dev import get_provider_info as _mdev_provider - mdev_info = _mdev_provider(canonical) + mdev_info = _models_dev_info(canonical) if mdev_info is not None: return ProviderDef( - id=canonical, - name=mdev_info.name, - transport="openai_chat", - api_key_env_vars=mdev_info.env, - base_url=mdev_info.api, - source="models.dev", + id=canonical, name=mdev_info.name, transport="openai_chat", api_key_env_vars=mdev_info.env, + base_url=mdev_info.api, source="models.dev", ) except Exception: pass - return None From 6ae394025f060def7dc62c10dcebd75d6644933c Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:13:18 -0700 Subject: [PATCH 02/19] =?UTF-8?q?refactor(hermes=5Fcli):=20runtime=5Fprovi?= =?UTF-8?q?der=20=E2=80=94=20pool-entry=20mode=20helper,=20shared=20anthro?= =?UTF-8?q?pic/azure-key/effective-model=20helpers,=20minimax+openrouter?= =?UTF-8?q?=20tail=20helpers?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/runtime_provider.py | 251 +++++++++++++++++---------------- 1 file changed, 126 insertions(+), 125 deletions(-) diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index 38e6fe461c..eb80f77a21 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -1,7 +1,7 @@ """Shared runtime provider resolution for CLI, gateway, cron, and helpers. -Layout: this module owns the resolution ORDER (:func:`resolve_runtime_provider`), the api_mode / -base_url helpers and the pool / OAuth / explicit paths. Custom-provider lookup lives in +This module owns the resolution ORDER (:func:`resolve_runtime_provider`), the api_mode / base_url +helpers and the pool / OAuth / explicit paths. Custom-provider lookup lives in :mod:`hermes_cli.runtime_provider_custom`; Azure Foundry, OpenRouter/bare-custom, Bedrock and external-process builders in :mod:`hermes_cli.runtime_provider_backends`. Both are re-exported here so ``hermes_cli.runtime_provider.`` imports and test patches keep working. @@ -53,32 +53,24 @@ from hermes_cli.providers import is_official_openai_host from utils import base_url_host_matches, base_url_hostname, env_int +# Late-bound delegates, deliberately NOT module-level from-imports: this module is often imported +# lazily, so its first import can happen while a test has ``hermes_cli.config.load_config`` patched +# — a from-import would bind the MagicMock permanently and poison every later caller. def load_config(): - """Late-bound delegate to :func:`hermes_cli.config.load_config`. - - Deliberately NOT a module-level from-import: this module is often imported lazily, so its - first import can happen while a test has ``hermes_cli.config.load_config`` patched — a - from-import would bind the MagicMock permanently and poison every later caller. - """ return _config_mod.load_config() def get_compatible_custom_providers(config=None): - """Late-bound delegate — see :func:`load_config` for why.""" return _config_mod.get_compatible_custom_providers(config) def normalize_extra_headers(value): - """Late-bound delegate — see :func:`load_config` for why.""" return _config_mod.normalize_extra_headers(value) def _getenv(name: str, default: str = "") -> str: - """Profile-scoped ``os.getenv`` for credential/provider reads. - - Identical to ``os.getenv`` when multiplexing is off; scope-aware (fail-closed on an unscoped - read) when on. Genuinely-global vars are handled inside ``get_secret``. - """ + """Profile-scoped ``os.getenv`` for credential/provider reads: identical to ``os.getenv`` when + multiplexing is off; scope-aware (fail-closed on an unscoped read) when on.""" val = _get_secret(name, default) return val if val is not None else default @@ -98,10 +90,10 @@ def _resolves_to_custom(name: str) -> bool: def _config_base_url_trustworthy_for_bare_custom(cfg_base_url: str, cfg_provider: str) -> bool: """Whether ``model.base_url`` may back bare ``custom`` runtime resolution. - The model picker can select Custom while ``model.provider`` still reflects a previous - provider. Non-loopback URLs are rejected unless the YAML provider is already ``custom`` or a - local-server alias (ollama/vllm/llamacpp — else a legit LAN ollama endpoint silently falls - through to OpenRouter), so a stale OpenRouter/Z.ai base_url cannot hijack local sessions. + The model picker can select Custom while ``model.provider`` still reflects a previous provider. + Non-loopback URLs are rejected unless the YAML provider is already ``custom`` or a local-server + alias (ollama/vllm/llamacpp — else a legit LAN ollama endpoint silently falls through to + OpenRouter), so a stale OpenRouter/Z.ai base_url cannot hijack local sessions. """ cfg_provider_norm = (cfg_provider or "").strip().lower() bu = (cfg_base_url or "").strip() @@ -127,15 +119,9 @@ _HOST_MANDATED_API_MODES = { "api.anthropic.com": "anthropic_messages", } -_VALID_API_MODES = { - "chat_completions", - "codex_responses", - "anthropic_messages", - "bedrock_converse", - # Opt-in: hand the whole turn to a `codex app-server` subprocess (Codex's own tool runtime). - # Gated on `model.openai_runtime == "codex_app_server"` AND provider in {openai, openai-codex}. - "codex_app_server", -} +# codex_app_server is opt-in: hand the whole turn to a `codex app-server` subprocess (Codex's own +# tool runtime), gated on `model.openai_runtime == "codex_app_server"` AND provider in {openai, openai-codex}. +_VALID_API_MODES = {"chat_completions", "codex_responses", "anthropic_messages", "bedrock_converse", "codex_app_server"} def _detect_api_mode_for_url(base_url: str) -> Optional[str]: @@ -176,8 +162,8 @@ def _parse_api_mode(raw: Any) -> Optional[str]: def _fallback_api_mode(provider: str, base_url: str, model: str = "") -> str: """api_mode when no explicit/persisted mode applies: URL detection (host-mandated wire shapes) first, then the transport the provider overlay declares via ``providers.determine_api_mode`` - (which was never consulted before — ``openai-api`` pointed at us.api.openai.com 400'd on every - tool call), then ``chat_completions``.""" + (``openai-api`` pointed at us.api.openai.com 400'd on every tool call without it), then + ``chat_completions``.""" detected = _detect_api_mode_for_url(base_url) if detected: return detected @@ -212,13 +198,27 @@ def _provider_supports_explicit_api_mode(provider: Optional[str], configured_pro return normalized_configured == normalized_provider -def _copilot_runtime_api_mode(model_cfg: Dict[str, Any], api_key: str, *, target_model: Optional[str] = None) -> str: +def _configured_api_mode(provider: str, model_cfg: Dict[str, Any]) -> Optional[str]: + """Persisted ``model.api_mode`` when valid and recorded for this provider, else None.""" configured_mode = _parse_api_mode(model_cfg.get("api_mode")) - if configured_mode and _provider_supports_explicit_api_mode("copilot", _cfg_provider(model_cfg)): + if configured_mode and _provider_supports_explicit_api_mode(provider, _cfg_provider(model_cfg)): + return configured_mode + return None + + +def _effective_model(model_cfg: Dict[str, Any], target_model: Optional[str]) -> str: + """The caller's target model (e.g. /model switch) beats the persisted default, else api_mode + is computed from a stale default.""" + return target_model or model_cfg.get("default") or "" + + +def _copilot_runtime_api_mode(model_cfg: Dict[str, Any], api_key: str, *, target_model: Optional[str] = None) -> str: + configured_mode = _configured_api_mode("copilot", model_cfg) + if configured_mode: return configured_mode # Use the model being resolved, not the persisted default: a Claude MoA slot inheriting # codex_responses from a GPT-5 default fails with "model ... does not support Responses API". - model_name = str(target_model or model_cfg.get("default") or "").strip() + model_name = str(_effective_model(model_cfg, target_model)).strip() if not model_name: return "chat_completions" try: @@ -258,10 +258,7 @@ def _configured_or_fallback_api_mode( from hermes_cli.models import opencode_model_api_mode return opencode_model_api_mode(provider, effective_model) - configured_mode = _parse_api_mode(model_cfg.get("api_mode")) - if configured_mode and _provider_supports_explicit_api_mode(provider, _cfg_provider(model_cfg)): - return configured_mode - return _fallback_api_mode(provider, base_url, effective_model) + return _configured_api_mode(provider, model_cfg) or _fallback_api_mode(provider, base_url, effective_model) def _api_key_provider_api_mode( @@ -333,6 +330,15 @@ def _anthropic_cfg_base_url(model_cfg: Dict[str, Any]) -> str: return cfg_base_url if _anthropic_base_url_override_ok(cfg_base_url) else "" +def _anthropic_token_or_raise() -> str: + from agent.anthropic_adapter import resolve_anthropic_token + + token = resolve_anthropic_token() + if not token: + raise AuthError(_NO_ANTHROPIC_CREDENTIALS_MSG) + return token + + def _host_derived_api_key(base_url: str) -> str: """``_API_KEY`` from the env, vendor = registrable hostname label (``api.deepseek.com`` → ``deepseek``). Lookalike hosts pick the ATTACKER's label (api.deepseek.com.attacker.test → @@ -408,21 +414,16 @@ def _nous_api_mode(model: str) -> str: return nous_api_mode(model) -def _normalize_opencode_runtime_base_url(provider: str, api_mode: str, base_url: str) -> str: - """OpenCode base URLs end with /v1 for OpenAI-compatible models, but the Anthropic SDK prepends - its own /v1/messages: strip /v1 for anthropic_messages, re-append otherwise.""" +def _finalize_base_url(provider: str, api_mode: str, base_url: str) -> str: + """Shared tail for pool-entry and api-key paths: OpenCode /v1 rule (OpenCode URLs end with /v1 + for OpenAI-compatible models but the Anthropic SDK prepends its own /v1/messages — strip for + anthropic_messages, re-append otherwise), then LM Studio normalization.""" from hermes_cli.models import opencode_provider_family - if opencode_provider_family(provider) is None: - return base_url - from hermes_cli.models import normalize_opencode_base_url + if opencode_provider_family(provider) is not None: + from hermes_cli.models import normalize_opencode_base_url - return normalize_opencode_base_url(provider, api_mode, base_url) - - -def _finalize_base_url(provider: str, api_mode: str, base_url: str) -> str: - """Shared tail for pool-entry and api-key paths: OpenCode /v1 rule, then LM Studio normalization.""" - base_url = _normalize_opencode_runtime_base_url(provider, api_mode, base_url) + base_url = normalize_opencode_base_url(provider, api_mode, base_url) if provider == "lmstudio": base_url = auth_mod._normalize_lmstudio_runtime_base_url(base_url) return base_url @@ -540,6 +541,35 @@ _POOL_ENTRY_SIMPLE_MODES: Dict[str, tuple] = { } +def _pool_entry_mode_and_url(provider, entry, model_cfg, effective_model, base_url) -> tuple: + """(api_mode, base_url) for a pool entry of ``provider``.""" + if provider in _POOL_ENTRY_SIMPLE_MODES: + api_mode, default_url = _POOL_ENTRY_SIMPLE_MODES[provider] + return api_mode, base_url or (default_url() if callable(default_url) else default_url) + if provider == "anthropic": + return "anthropic_messages", _anthropic_cfg_base_url(model_cfg) or base_url or _ANTHROPIC_DEFAULT_BASE_URL + if provider == "nous": + return _nous_api_mode(effective_model), _nous_inference_base_url_override() or base_url + if provider == "copilot": + api_mode = _copilot_runtime_api_mode(model_cfg, getattr(entry, "runtime_api_key", ""), target_model=effective_model) + return api_mode, base_url or PROVIDER_REGISTRY["copilot"].inference_base_url + if provider == "azure-foundry": + api_mode = "chat_completions" + if _cfg_provider(model_cfg) == "azure-foundry": + base_url = _config_base_url_for_provider(model_cfg, "azure-foundry") or base_url + api_mode = _parse_api_mode(model_cfg.get("api_mode")) or api_mode + api_mode = _azure_inferred_api_mode(effective_model, api_mode) + if api_mode == "anthropic_messages": + base_url = re.sub(r"/v1/?$", "", base_url) + return api_mode, base_url + # Honour model.base_url only when the pool entry carries no explicit base_url (i.e. it fell + # back to the registry default). Env var overrides win. + pconfig = PROVIDER_REGISTRY.get(provider) + if pconfig and base_url.rstrip("/") == pconfig.inference_base_url.rstrip("/"): + base_url = _config_base_url_for_provider(model_cfg, provider) or base_url + return _configured_or_fallback_api_mode(provider, model_cfg, base_url, effective_model, opencode_by_model=True), base_url + + def _resolve_runtime_from_pool_entry( *, provider: str, @@ -550,43 +580,13 @@ def _resolve_runtime_from_pool_entry( target_model: Optional[str] = None, ) -> Dict[str, Any]: model_cfg = model_cfg or _get_model_config() - # The caller's target model (e.g. /model switch) beats the persisted default, else api_mode is - # computed from a stale default. - effective_model = target_model or model_cfg.get("default") or "" - base_url = _pool_entry_base_url(entry).rstrip("/") - api_key = _pool_entry_api_key(entry) - if provider in _POOL_ENTRY_SIMPLE_MODES: - api_mode, default_url = _POOL_ENTRY_SIMPLE_MODES[provider] - base_url = base_url or (default_url() if callable(default_url) else default_url) - elif provider == "anthropic": - api_mode = "anthropic_messages" - base_url = _anthropic_cfg_base_url(model_cfg) or base_url or _ANTHROPIC_DEFAULT_BASE_URL - elif provider == "nous": - api_mode = _nous_api_mode(effective_model) - base_url = _nous_inference_base_url_override() or base_url - elif provider == "copilot": - api_mode = _copilot_runtime_api_mode(model_cfg, getattr(entry, "runtime_api_key", ""), target_model=effective_model) - base_url = base_url or PROVIDER_REGISTRY["copilot"].inference_base_url - elif provider == "azure-foundry": - api_mode = "chat_completions" - if _cfg_provider(model_cfg) == "azure-foundry": - base_url = _config_base_url_for_provider(model_cfg, "azure-foundry") or base_url - api_mode = _parse_api_mode(model_cfg.get("api_mode")) or api_mode - api_mode = _azure_inferred_api_mode(effective_model, api_mode) - if api_mode == "anthropic_messages": - base_url = re.sub(r"/v1/?$", "", base_url) - else: - # Honour model.base_url only when the pool entry carries no explicit base_url (i.e. it - # fell back to the registry default). Env var overrides win. - pconfig = PROVIDER_REGISTRY.get(provider) - if pconfig and base_url.rstrip("/") == pconfig.inference_base_url.rstrip("/"): - base_url = _config_base_url_for_provider(model_cfg, provider) or base_url - api_mode = _configured_or_fallback_api_mode(provider, model_cfg, base_url, effective_model, opencode_by_model=True) - + api_mode, base_url = _pool_entry_mode_and_url( + provider, entry, model_cfg, _effective_model(model_cfg, target_model), _pool_entry_base_url(entry).rstrip("/") + ) base_url = _finalize_base_url(provider, api_mode, base_url) api_mode = _maybe_apply_codex_app_server_runtime(provider=provider, api_mode=api_mode, model_cfg=model_cfg) return _runtime( - provider, api_mode, base_url, api_key, + provider, api_mode, base_url, _pool_entry_api_key(entry), source=getattr(entry, "source", "pool"), credential_pool=pool, requested_provider=requested_provider, ) @@ -658,12 +658,7 @@ def _resolve_from_pool( def _explicit_anthropic(requested_provider, model_cfg, api_key, base_url, target_model): base_url = base_url or _anthropic_cfg_base_url(model_cfg) or _ANTHROPIC_DEFAULT_BASE_URL - if not api_key: - from agent.anthropic_adapter import resolve_anthropic_token - - api_key = resolve_anthropic_token() - if not api_key: - raise AuthError(_NO_ANTHROPIC_CREDENTIALS_MSG) + api_key = api_key or _anthropic_token_or_raise() return _runtime("anthropic", "anthropic_messages", base_url, api_key, source="explicit", requested_provider=requested_provider) @@ -700,7 +695,7 @@ def _explicit_nous(requested_provider, model_cfg, api_key, explicit_base_url, ta expires_at = creds.get("expires_at") base_url = explicit_base_url or creds.get("base_url", "").rstrip("/") or base_url return _runtime( - "nous", _nous_api_mode(target_model or model_cfg.get("default") or ""), base_url, api_key, + "nous", _nous_api_mode(_effective_model(model_cfg, target_model)), base_url, api_key, source="explicit", expires_at=expires_at, requested_provider=requested_provider, ) @@ -821,7 +816,7 @@ def _resolve_oauth_runtime(provider, requested_provider, model_cfg, target_model return None api_mode = spec.api_mode if callable(api_mode): - api_mode = api_mode(target_model or model_cfg.get("default") or "") + api_mode = api_mode(_effective_model(model_cfg, target_model)) return _runtime( provider, api_mode, (creds.get("base_url") or "").rstrip("/") or spec.default_base_url, @@ -832,31 +827,47 @@ def _resolve_oauth_runtime(provider, requested_provider, model_cfg, target_model ) +def _minimax_oauth_runtime(provider, requested_provider) -> Optional[Dict[str, Any]]: + pconfig = PROVIDER_REGISTRY.get(provider) + if not (pconfig and pconfig.auth_type == "oauth_minimax"): + return None + from hermes_cli.auth import resolve_minimax_oauth_runtime_credentials + + creds = resolve_minimax_oauth_runtime_credentials() + return _runtime( + provider, "anthropic_messages", creds["base_url"], creds["api_key"], + source=creds.get("source", "oauth"), requested_provider=requested_provider, + ) + + # ── env/config paths for anthropic and registry api_key providers ────────────────────────── +def _azure_anthropic_env_key(model_cfg: Dict[str, Any]) -> str: + """Azure Anthropic key: `key_env` / `api_key_env` hints on the model config, then an inline + api_key (multi-profile setups), then the historical fixed names.""" + for hint_key in ("key_env", "api_key_env"): + env_var = str(model_cfg.get(hint_key) or "").strip() + if env_var: + token = _getenv(env_var, "").strip() + if token: + return token + return ( + str(model_cfg.get("api_key") or "").strip() + or _getenv("AZURE_ANTHROPIC_KEY", "").strip() + or _getenv("ANTHROPIC_API_KEY", "").strip() + ) + + def _anthropic_env_runtime(requested_provider: str, model_cfg: Dict[str, Any]) -> Dict[str, Any]: """Native Anthropic (Messages API) from env/auth store; ``model.base_url`` honoured only when the configured provider is anthropic (else a Codex endpoint would leak into Anthropic requests).""" cfg_base_url = _anthropic_cfg_base_url(model_cfg) base_url = cfg_base_url or _ANTHROPIC_DEFAULT_BASE_URL # Microsoft Foundry endpoints reject Claude Code OAuth tokens, which resolve_anthropic_token() - # would return first — use the env key directly: `key_env` / `api_key_env` hints on the model - # config, then an inline api_key (multi-profile setups), then the historical fixed names. + # would return first — use the env key directly. if base_url_host_matches(base_url, "azure.com") or (cfg_base_url and base_url_host_matches(cfg_base_url, "azure.com")): - token = "" - for hint_key in ("key_env", "api_key_env"): - env_var = str(model_cfg.get(hint_key) or "").strip() - if env_var: - token = _getenv(env_var, "").strip() - if token: - break - token = ( - token - or str(model_cfg.get("api_key") or "").strip() - or _getenv("AZURE_ANTHROPIC_KEY", "").strip() - or _getenv("ANTHROPIC_API_KEY", "").strip() - ) + token = _azure_anthropic_env_key(model_cfg) if not token: raise AuthError( "No Azure Anthropic API key found. Set AZURE_ANTHROPIC_KEY or " @@ -864,11 +875,7 @@ def _anthropic_env_runtime(requested_provider: str, model_cfg: Dict[str, Any]) - "config.yaml model section at a custom env var." ) else: - from agent.anthropic_adapter import resolve_anthropic_token - - token = resolve_anthropic_token() - if not token: - raise AuthError(_NO_ANTHROPIC_CREDENTIALS_MSG) + token = _anthropic_token_or_raise() return _runtime("anthropic", "anthropic_messages", base_url, token, source="env", requested_provider=requested_provider) @@ -987,6 +994,10 @@ def _local_endpoint_bypass(requested_provider: str, explicit_api_key, explicit_b return None if any(base_url_host_matches(cfg_base_url, host) for host in _LOCAL_BYPASS_CLOUD_HOSTS): return None + return _openrouter_fallback(requested_provider, explicit_api_key, explicit_base_url) + + +def _openrouter_fallback(requested_provider, explicit_api_key, explicit_base_url) -> Dict[str, Any]: runtime = _resolve_openrouter_runtime( requested_provider=requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url ) @@ -1076,15 +1087,9 @@ def resolve_runtime_provider( return runtime if provider == "minimax-oauth": - pconfig = PROVIDER_REGISTRY.get(provider) - if pconfig and pconfig.auth_type == "oauth_minimax": - from hermes_cli.auth import resolve_minimax_oauth_runtime_credentials - - creds = resolve_minimax_oauth_runtime_credentials() - return _runtime( - provider, "anthropic_messages", creds["base_url"], creds["api_key"], - source=creds.get("source", "oauth"), requested_provider=requested_provider, - ) + runtime = _minimax_oauth_runtime(provider, requested_provider) + if runtime: + return runtime if _is_external_process_provider(provider): return _resolve_external_process_runtime(provider, requested_provider) @@ -1099,11 +1104,7 @@ def resolve_runtime_provider( if pconfig and pconfig.auth_type == "api_key": return _api_key_provider_runtime(provider, pconfig, requested_provider, model_cfg, target_model) - runtime = _resolve_openrouter_runtime( - requested_provider=requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url - ) - runtime["requested_provider"] = requested_provider - return runtime + return _openrouter_fallback(requested_provider, explicit_api_key, explicit_base_url) def format_runtime_provider_error(error: Exception) -> str: From b8c7add3a9a6c95303942c7397a139a86a6bcd4b Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:21:10 -0700 Subject: [PATCH 03/19] =?UTF-8?q?refactor(hermes=5Fcli):=20runtime=5Fprovi?= =?UTF-8?q?der=5Fcustom/backends=20=E2=80=94=20shared=20=5Fcustom=5Fruntim?= =?UTF-8?q?e=20builder,=20azure=20key=20helper,=20compact=20docs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/runtime_provider_backends.py | 93 +++++------- hermes_cli/runtime_provider_custom.py | 194 +++++++++--------------- 2 files changed, 112 insertions(+), 175 deletions(-) diff --git a/hermes_cli/runtime_provider_backends.py b/hermes_cli/runtime_provider_backends.py index d9c662f200..f9ddf0afc6 100644 --- a/hermes_cli/runtime_provider_backends.py +++ b/hermes_cli/runtime_provider_backends.py @@ -1,9 +1,9 @@ -"""Provider-specific runtime builders for :mod:`hermes_cli.runtime_provider`. +"""Provider-specific runtime builders for :mod:`hermes_cli.runtime_provider`: Azure Foundry, the +OpenRouter / bare-custom fallback resolver, Bedrock, and external-process providers. -Azure Foundry, the OpenRouter / bare-custom fallback resolver, Bedrock, and external-process -providers. Origin-internal collaborators are resolved on the origin module at call time via -:func:`_rp` so test patches on ``hermes_cli.runtime_provider.*`` (``_get_model_config``, -``load_config``, ``has_usable_secret``, ``_try_resolve_from_custom_pool``, …) still apply. +Origin-internal collaborators are resolved on the origin module at call time via :func:`_rp` so +test patches on ``hermes_cli.runtime_provider.*`` (``_get_model_config``, ``load_config``, +``has_usable_secret``, ``_try_resolve_from_custom_pool``, …) still apply. """ from __future__ import annotations @@ -22,11 +22,6 @@ def _rp(): return origin -def _strip_v1(base_url: str) -> str: - """Anthropic SDK appends /v1/messages itself — drop an inherited trailing /v1.""" - return re.sub(r"/v1/?$", "", base_url) - - # ── Azure Foundry ────────────────────────────────────────────────────────────────────────── @@ -35,11 +30,7 @@ def _azure_entra_credentials(cfg_entra: Dict[str, Any]) -> Any: ``build_anthropic_client`` injects the bearer via an httpx hook).""" AuthError = _rp().AuthError try: - from agent.azure_identity_adapter import ( - SCOPE_AI_AZURE_DEFAULT, - EntraIdentityConfig, - build_token_provider, - ) + from agent.azure_identity_adapter import SCOPE_AI_AZURE_DEFAULT, EntraIdentityConfig, build_token_provider except Exception as exc: raise AuthError( "Azure Foundry Entra ID auth requires the 'azure-identity' " @@ -53,6 +44,27 @@ def _azure_entra_credentials(cfg_entra: Dict[str, Any]) -> Any: raise AuthError(str(exc)) from exc +def _azure_foundry_api_key(rp, explicit_api_key: str) -> str: + if explicit_api_key: + return explicit_api_key + try: + from hermes_cli.config import get_env_value + + api_key = get_env_value("AZURE_FOUNDRY_API_KEY") or "" + except Exception: + api_key = "" + api_key = api_key or rp._getenv("AZURE_FOUNDRY_API_KEY", "").strip() + if not api_key: + raise rp.AuthError( + "Azure Foundry requires an API key. Set AZURE_FOUNDRY_API_KEY in " + "~/.hermes/.env or run 'hermes model' to configure. To use " + "keyless Microsoft Entra ID auth instead, set " + "model.auth_mode: entra_id in config.yaml (or pick " + "'Microsoft Entra ID' in 'hermes model')." + ) + return api_key + + def _resolve_azure_foundry_runtime( *, requested_provider: str, @@ -63,7 +75,7 @@ def _resolve_azure_foundry_runtime( ) -> Dict[str, Any]: """Azure Foundry: ``model.base_url`` + ``model.api_mode`` (or explicit overrides), API key from ``.env``/env or a per-request Entra ID token, trailing ``/v1`` stripped for Anthropic-style - endpoints.""" + endpoints (the Anthropic SDK appends /v1/messages itself).""" rp = _rp() explicit_api_key = str(explicit_api_key or "").strip() explicit_base_url_clean = str(explicit_base_url or "").strip().rstrip("/") @@ -88,7 +100,7 @@ def _resolve_azure_foundry_runtime( "the AZURE_FOUNDRY_BASE_URL environment variable." ) if cfg_api_mode == "anthropic_messages": - base_url = _strip_v1(base_url) + base_url = re.sub(r"/v1/?$", "", base_url) if cfg_auth_mode == "entra_id": if explicit_api_key: @@ -105,28 +117,9 @@ def _resolve_azure_foundry_runtime( "azure-foundry", cfg_api_mode, base_url, api_key, auth_mode=auth_mode, entra=clean_entra, source=source, requested_provider=requested_provider, ) - - api_key = explicit_api_key - if not api_key: - try: - from hermes_cli.config import get_env_value - - api_key = get_env_value("AZURE_FOUNDRY_API_KEY") or "" - except Exception: - api_key = "" - api_key = api_key or rp._getenv("AZURE_FOUNDRY_API_KEY", "").strip() - if not api_key: - raise rp.AuthError( - "Azure Foundry requires an API key. Set AZURE_FOUNDRY_API_KEY in " - "~/.hermes/.env or run 'hermes model' to configure. To use " - "keyless Microsoft Entra ID auth instead, set " - "model.auth_mode: entra_id in config.yaml (or pick " - "'Microsoft Entra ID' in 'hermes model')." - ) return rp._runtime( - "azure-foundry", cfg_api_mode, base_url, api_key, - auth_mode="api_key", - source="explicit" if (explicit_api_key or explicit_base_url) else "config", + "azure-foundry", cfg_api_mode, base_url, _azure_foundry_api_key(rp, explicit_api_key), + auth_mode="api_key", source="explicit" if (explicit_api_key or explicit_base_url) else "config", requested_provider=requested_provider, ) @@ -135,10 +128,7 @@ def _resolve_azure_foundry_runtime( def _resolve_openrouter_runtime( - *, - requested_provider: str, - explicit_api_key: Optional[str] = None, - explicit_base_url: Optional[str] = None, + *, requested_provider: str, explicit_api_key: Optional[str] = None, explicit_base_url: Optional[str] = None ) -> Dict[str, Any]: """Terminal resolver: OpenRouter, or a bare/aliased ``custom`` endpoint. @@ -152,8 +142,7 @@ def _resolve_openrouter_runtime( cfg_base_url = model_cfg.get("base_url") if isinstance(model_cfg.get("base_url"), str) else "" cfg_provider = model_cfg.get("provider") if isinstance(model_cfg.get("provider"), str) else "" cfg_api_key = next( - (v.strip() for v in (model_cfg.get("api_key"), model_cfg.get("api")) if isinstance(v, str) and v.strip()), - "", + (v.strip() for v in (model_cfg.get("api_key"), model_cfg.get("api")) if isinstance(v, str) and v.strip()), "" ) requested_norm = (requested_provider or "").strip().lower() cfg_provider = cfg_provider.strip().lower() @@ -163,7 +152,6 @@ def _resolve_openrouter_runtime( env_openrouter_base_url = rp._getenv("OPENROUTER_BASE_URL", "").strip() env_custom_base_url = rp._getenv("CUSTOM_BASE_URL", "").strip() - use_config_base_url = bool(cfg_base_url.strip()) and not explicit_base_url and ( (requested_norm == "auto" and cfg_provider in ("", "auto")) or (requested_norm == "custom" and rp._config_base_url_trustworthy_for_bare_custom(cfg_base_url, cfg_provider)) @@ -194,20 +182,19 @@ def _resolve_openrouter_runtime( ] api_key = next((str(c or "").strip() for c in candidates if rp.has_usable_secret(c)), "") source = "explicit" if (explicit_api_key or explicit_base_url) else "env/config" + cfg_api_mode = rp._parse_api_mode(model_cfg.get("api_mode")) # Explicit "custom" stays "custom" rather than relabeling to "openrouter". if requested_norm != "custom": return rp._runtime( - "openrouter", - rp._parse_api_mode(model_cfg.get("api_mode")) or rp._detect_api_mode_for_url(base_url) or "chat_completions", + "openrouter", cfg_api_mode or rp._detect_api_mode_for_url(base_url) or "chat_completions", base_url, api_key, source=source, ) if base_url: # provider_name makes pool lookup prefer name match over base_url (fixes credential # mix-ups when multiple custom providers share a base_url). pool_result = rp._try_resolve_from_custom_pool( - base_url, "custom", rp._parse_api_mode(model_cfg.get("api_mode")), - provider_name=requested_provider if requested_norm != "custom" else None, + base_url, "custom", cfg_api_mode, provider_name=requested_provider if requested_norm != "custom" else None ) if pool_result: return pool_result @@ -274,12 +261,8 @@ def _resolve_bedrock_runtime(requested_provider: str, model_cfg: Dict[str, Any], if is_openai_bedrock_model(current_model): bearer = resolve_bedrock_bearer_token() runtime.update( - api_mode="codex_responses", - base_url=bedrock_openai_base_url(region), - api_key=bearer or "aws-sdk", - source="AWS_BEARER_TOKEN_BEDROCK" if bearer else auth_source, - model=current_model, - bedrock_openai=True, + api_mode="codex_responses", base_url=bedrock_openai_base_url(region), api_key=bearer or "aws-sdk", + source="AWS_BEARER_TOKEN_BEDROCK" if bearer else auth_source, model=current_model, bedrock_openai=True, ) elif is_anthropic_bedrock_model(current_model) and not has_bearer_token: runtime.update(api_mode="anthropic_messages", bedrock_anthropic=True) diff --git a/hermes_cli/runtime_provider_custom.py b/hermes_cli/runtime_provider_custom.py index 1de8c6771d..51be816abe 100644 --- a/hermes_cli/runtime_provider_custom.py +++ b/hermes_cli/runtime_provider_custom.py @@ -1,11 +1,11 @@ """Custom-provider resolution: ``providers:`` / ``custom_providers:`` lookup, identity recovery, custom credential pools, and the named-custom runtime builder. -Extracted from :mod:`hermes_cli.runtime_provider`; every public/private name here is -re-exported there. Origin-internal collaborators (``load_config``, ``_get_model_config``, -``load_pool``, ``has_usable_secret``, ``custom_provider_pool_key_candidates``, …) are looked up -on the origin module AT CALL TIME via :func:`_rp` so ``monkeypatch.setattr(runtime_provider, -name, …)`` in tests keeps working for moved bodies. +Extracted from :mod:`hermes_cli.runtime_provider`; every name here is re-exported there. +Origin-internal collaborators (``load_config``, ``_get_model_config``, ``load_pool``, +``has_usable_secret``, ``custom_provider_pool_key_candidates``, …) are looked up on the origin +module AT CALL TIME via :func:`_rp` so ``monkeypatch.setattr(runtime_provider, name, …)`` keeps +working for moved bodies. """ from __future__ import annotations @@ -19,6 +19,8 @@ from utils import base_url_hostname logger = logging.getLogger("hermes_cli.runtime_provider") +_LLAMACPP_ALIASES = ("llamacpp", "llama.cpp", "llama-cpp") + def _rp(): """Origin module, late-bound so test patches on ``hermes_cli.runtime_provider.*`` apply.""" @@ -65,11 +67,9 @@ def _lift_model_capabilities(entry: Dict[str, Any], model: Optional[str], result def _lift_max_output_tokens(entry: Dict[str, Any], result: Dict[str, Any]) -> None: - """``max_output_tokens`` or ``max_tokens`` on a provider entry pins its own output limit. - - Gateway/CLI map it onto ``AIAgent.max_tokens`` only when top-level ``model.max_tokens`` is - unset, so the documented global key still wins. - """ + """``max_output_tokens`` or ``max_tokens`` on a provider entry pins its own output limit; + gateway/CLI map it onto ``AIAgent.max_tokens`` only when top-level ``model.max_tokens`` is + unset, so the documented global key still wins.""" for key in ("max_output_tokens", "max_tokens"): value = entry.get(key) if isinstance(value, int) and value > 0: @@ -85,12 +85,7 @@ def _lift_extra_headers(entry: Dict[str, Any], result: Dict[str, Any]) -> None: def _lift_common_custom_fields( - entry: Dict[str, Any], - result: Dict[str, Any], - *, - provider_key: str, - key_env: str, - api_mode: Optional[str], + entry: Dict[str, Any], result: Dict[str, Any], *, provider_key: str, key_env: str, api_mode: Optional[str] ) -> None: """Copy the optional fields shared by ``providers:`` and legacy ``custom_providers:`` entries.""" if key_env: @@ -115,11 +110,11 @@ def _lift_common_custom_fields( def _shadowed_by_builtin(requested_norm: str) -> bool: """Raw names map to custom providers only when they are not canonical built-ins. - Explicit ``custom:`` keys always target the saved entry, and bare ``custom`` is - exempt: a user may literally name a ``providers:`` entry "custom" (returning None before - the config scan made such cron jobs fail with ``auth_unavailable``). Defer to the built-in - only when the raw name IS the canonical provider (``nous``); an entry matching merely an - alias (``kimi`` → ``kimi-coding``) is the user's target. + Explicit ``custom:`` keys always target the saved entry, and bare ``custom`` is exempt: a + user may literally name a ``providers:`` entry "custom" (returning None before the config scan + made such cron jobs fail with ``auth_unavailable``). Defer to the built-in only when the raw + name IS the canonical provider (``nous``); an entry matching merely an alias (``kimi`` → + ``kimi-coding``) is the user's target. """ if requested_norm == "custom" or requested_norm.startswith("custom:"): return False @@ -162,9 +157,7 @@ def _match_new_style_provider(requested_norm: str, providers: Dict[str, Any]) -> # v12 migration writes ``transport``; hand-edited configs may still use ``api_mode``. # Accept both or migrated configs silently downgrade to chat_completions. _lift_common_custom_fields( - entry, result, - provider_key=_clean(ep_name), - key_env=key_env, + entry, result, provider_key=_clean(ep_name), key_env=key_env, api_mode=rp._parse_api_mode(entry.get("api_mode") or entry.get("transport")), ) return result @@ -187,9 +180,7 @@ def _match_legacy_custom_provider(requested_norm: str, custom_providers) -> Opti if model_name: result["model"] = model_name _lift_common_custom_fields( - entry, result, - provider_key=provider_key, - key_env=_clean(entry.get("key_env", "")), + entry, result, provider_key=provider_key, key_env=_clean(entry.get("key_env", "")), api_mode=_rp()._parse_api_mode(entry.get("api_mode")), ) return result @@ -200,7 +191,6 @@ def _get_named_custom_provider(requested_provider: str) -> Optional[Dict[str, An requested_norm = _normalize_custom_provider_name(requested_provider or "") if not requested_norm or requested_norm == "auto" or _shadowed_by_builtin(requested_norm): return None - rp = _rp() config = rp.load_config() providers = config.get("providers") @@ -208,7 +198,6 @@ def _get_named_custom_provider(requested_provider: str) -> Optional[Dict[str, An found = _match_new_style_provider(requested_norm, providers) if found: return found - if isinstance(config.get("custom_providers"), dict): logger.warning( "custom_providers in config.yaml is a dict, not a list. " @@ -223,10 +212,8 @@ def _get_named_custom_provider(requested_provider: str) -> Optional[Dict[str, An def has_named_custom_provider(requested_provider: str) -> bool: - """True when config defines a ``providers:`` / ``custom_providers:`` entry matching the request. - - Public wrapper so other modules (e.g. the cronjob tool) need not reach into a private helper. - """ + """True when config defines a ``providers:`` / ``custom_providers:`` entry matching the request + (public wrapper so e.g. the cronjob tool need not reach into a private helper).""" try: return _rp()._get_named_custom_provider(requested_provider) is not None except Exception: @@ -244,13 +231,11 @@ def _find_custom_identity(matches: Callable[[Dict[str, Any]], bool]) -> Optional config = rp.load_config() except Exception: return None - providers = config.get("providers") if isinstance(providers, dict): for ep_name, entry in providers.items(): if isinstance(entry, dict) and matches(entry): return custom_provider_slug(str(ep_name), str(ep_name)) - try: custom_providers = rp.get_compatible_custom_providers(config) except Exception: @@ -265,64 +250,54 @@ def _find_custom_identity(matches: Callable[[Dict[str, Any]], bool]) -> Optional def find_custom_provider_identity(base_url: str) -> Optional[str]: - """Map an endpoint URL back to its canonical ``custom:`` menu key. - - Session persistence stores the agent's *resolved* provider, and for every named custom - endpoint that is the literal string ``"custom"`` — the entry name is lost, and the api_key is - deliberately never persisted. - """ + """Map an endpoint URL back to its canonical ``custom:`` menu key. Session persistence + stores the agent's *resolved* provider, which for every named custom endpoint is the literal + string ``"custom"`` — the entry name is lost, and the api_key is deliberately never persisted.""" target = _normalize_base_url_for_match(base_url) if not target: return None return _find_custom_identity(lambda entry: _normalize_base_url_for_match(_entry_url(entry)) == target) -def find_custom_provider_identity_by_model(model: str) -> Optional[str]: - """Map a model id back to the ``custom:`` entry that serves it. +def _model_id_matches(value: Any, target: str) -> bool: + return isinstance(value, str) and value.strip().lower() == target - Companion to :func:`find_custom_provider_identity` for persistence paths where no base_url - survived the round-trip: the session row always stores the model name. - """ + +def find_custom_provider_identity_by_model(model: str) -> Optional[str]: + """Map a model id back to the ``custom:`` entry that serves it — companion to + :func:`find_custom_provider_identity` for persistence paths where no base_url survived the + round-trip (the session row always stores the model name).""" target = str(model or "").strip().lower() if not target: return None def _entry_serves_model(entry: Dict[str, Any]) -> bool: - for key in ("model", "default_model"): - value = entry.get(key) - if isinstance(value, str) and value.strip().lower() == target: - return True + if any(_model_id_matches(entry.get(key), target) for key in ("model", "default_model")): + return True models = entry.get("models") if isinstance(models, dict): return any(str(mid).strip().lower() == target for mid in models) if isinstance(models, list): - for item in models: - if isinstance(item, str) and item.strip().lower() == target: - return True - if isinstance(item, dict): - mid = item.get("id") or item.get("name") - if isinstance(mid, str) and mid.strip().lower() == target: - return True + return any( + _model_id_matches(item.get("id") or item.get("name") if isinstance(item, dict) else item, target) + for item in models + ) return False return _find_custom_identity(_entry_serves_model) def canonical_custom_identity( - *, - base_url: Optional[str] = None, - config_provider: Optional[str] = None, - model: Optional[str] = None, + *, base_url: Optional[str] = None, config_provider: Optional[str] = None, model: Optional[str] = None ) -> Optional[str]: """Recover a routable ``custom:`` identity for a bare custom provider. Every path that persists or restores a session's provider override must run the resolved - provider through this so a bare ``"custom"`` is upgraded back to its durable - ``custom:`` menu key. Recovery sources, in priority order: (1) ``base_url`` reverse - lookup — the one fact that always survives the round-trip when a URL was recorded; (2) - ``model`` reverse lookup (``model``/``default_model``/``models`` catalog); (3) the configured - provider (arg, then ``model.provider``, then ``HERMES_INFERENCE_PROVIDER``) when it names a - real entry. + provider through this so a bare ``"custom"`` is upgraded back to its durable ``custom:`` + menu key. Sources in priority order: (1) ``base_url`` reverse lookup — the one fact that always + survives the round-trip when a URL was recorded; (2) ``model`` reverse lookup + (``model``/``default_model``/``models`` catalog); (3) the configured provider (arg, then + ``model.provider``, then ``HERMES_INFERENCE_PROVIDER``) when it names a real entry. """ rp = _rp() if base_url: @@ -333,7 +308,6 @@ def canonical_custom_identity( identity = find_custom_provider_identity_by_model(model) if identity: return identity - candidate = str(config_provider or "").strip() if not candidate: try: @@ -342,7 +316,6 @@ def canonical_custom_identity( candidate = "" if not candidate: candidate = os.environ.get("HERMES_INFERENCE_PROVIDER", "").strip() - candidate_norm = _normalize_custom_provider_name(candidate) # A bare/non-routable candidate cannot heal a bare custom override. if not candidate_norm or candidate_norm in {"custom", "auto", "openrouter"}: @@ -367,11 +340,11 @@ def canonical_custom_identity( def is_routable_provider(provider: Optional[str]) -> bool: """Whether a provider name currently resolves to a routable route. - Empty/None/``auto`` is vacuously routable (agent build falls back to the configured - default). Bare ``custom`` is the resolved billing class shared by every named entry — not a - routable identity; restore paths must heal it (:func:`canonical_custom_identity`) or fall - back. Anything else is routable iff the full chain (built-in -> ``providers:`` -> - ``custom_providers:`` -> models.dev) resolves it. + Empty/None/``auto`` is vacuously routable (agent build falls back to the configured default). + Bare ``custom`` is the resolved billing class shared by every named entry — not a routable + identity; restore paths must heal it (:func:`canonical_custom_identity`) or fall back. Anything + else is routable iff the full chain (built-in -> ``providers:`` -> ``custom_providers:`` -> + models.dev) resolves it. """ name = str(provider or "").strip() if not name or name.lower() == "auto": @@ -383,9 +356,7 @@ def is_routable_provider(provider: Optional[str]) -> bool: rp = _rp() config = rp.load_config() - return resolve_provider_full( - name, config.get("providers"), rp.get_compatible_custom_providers(config) - ) is not None + return resolve_provider_full(name, config.get("providers"), rp.get_compatible_custom_providers(config)) is not None except Exception: return False @@ -394,10 +365,7 @@ def is_routable_provider(provider: Optional[str]) -> bool: def _try_resolve_from_custom_pool( - base_url: str, - provider_label: str, - api_mode_override: Optional[str] = None, - provider_name: Optional[str] = None, + base_url: str, provider_label: str, api_mode_override: Optional[str] = None, provider_name: Optional[str] = None ) -> Optional[Dict[str, Any]]: """Runtime dict from the first credential pool that owns this custom endpoint, else None.""" rp = _rp() @@ -424,12 +392,8 @@ def _try_resolve_from_custom_pool( # substitutes "no-key-required" for a loopback endpoint — this was the one gap. pool_api_key = "no-key-required" return rp._runtime( - provider_label, - api_mode_override or rp._detect_api_mode_for_url(base_url) or "chat_completions", - base_url, - pool_api_key, - source=f"pool:{pool_key}", - credential_pool=pool, + provider_label, api_mode_override or rp._detect_api_mode_for_url(base_url) or "chat_completions", + base_url, pool_api_key, source=f"pool:{pool_key}", credential_pool=pool, ) except Exception: continue @@ -447,12 +411,9 @@ def _apply_custom_provider_extras( custom_provider: Dict[str, Any], target_model: Optional[str], result: Dict[str, Any] ) -> None: """Copy model / capabilities / max_output_tokens / extra_headers / request_overrides onto a - resolved custom runtime. - - An explicit ``target_model`` wins over the provider's configured default (auxiliary slots / - background-review resolve a concrete model and must not fall back to ``default_model``). - ``extra_headers`` may carry credentials — NEVER log them. - """ + resolved custom runtime. An explicit ``target_model`` wins over the provider's configured + default (auxiliary slots / background-review resolve a concrete model and must not fall back to + ``default_model``). ``extra_headers`` may carry credentials — NEVER log them.""" model_name = target_model or custom_provider.get("model") if model_name: result["model"] = model_name @@ -482,12 +443,9 @@ def _resolve_llamacpp_runtime(requested_provider: str, explicit_api_key: Optiona endpoint = None if endpoint: return rp._runtime( - "custom", - "chat_completions", - endpoint["base_url"], + "custom", "chat_completions", endpoint["base_url"], (explicit_api_key or "").strip() or endpoint["api_key"] or "no-key-required", - source="local-runtime", - requested_provider=requested_provider, + source="local-runtime", requested_provider=requested_provider, ) try: enabled = bool((rp.load_config().get("local_runtime") or {}).get("enabled")) @@ -506,6 +464,14 @@ def _resolve_llamacpp_runtime(requested_provider: str, explicit_api_key: Optiona ) +def _custom_runtime(rp, base_url: str, api_key: Any, api_mode: Optional[str], **extra: Any) -> Dict[str, Any]: + """``custom`` runtime dict with URL-detected api_mode fallback and the no-auth placeholder.""" + return rp._runtime( + "custom", api_mode or rp._detect_api_mode_for_url(base_url) or "chat_completions", base_url, + api_key or "no-key-required", **extra, + ) + + def _resolve_direct_alias_runtime( requested_provider: str, explicit_api_key: Optional[str], explicit_base_url: str ) -> Dict[str, Any]: @@ -521,15 +487,8 @@ def _resolve_direct_alias_runtime( # OLLAMA_API_KEY gets its own gate here: without it a `model_aliases:` entry pointing at # Ollama Cloud resolved no key at all. candidates = [(explicit_api_key or "").strip(), *rp._host_gated_env_key_candidates(base_url, ollama=True)] - api_key = next((c for c in candidates if rp.has_usable_secret(c)), "") or "no-key-required" - return rp._runtime( - "custom", - rp._detect_api_mode_for_url(base_url) or "chat_completions", - base_url, - api_key, - source="direct-alias", - requested_provider=requested_provider, - ) + api_key = next((c for c in candidates if rp.has_usable_secret(c)), "") + return _custom_runtime(rp, base_url, api_key, None, source="direct-alias", requested_provider=requested_provider) def _opencode_family_for_custom(requested_provider: str, base_url: str) -> Optional[str]: @@ -562,7 +521,7 @@ def _resolve_named_custom_runtime( """ rp = _rp() requested_norm = (requested_provider or "").strip().lower() - if requested_norm in ("llamacpp", "llama.cpp", "llama-cpp") and not explicit_base_url: + if requested_norm in _LLAMACPP_ALIASES and not explicit_base_url: return _resolve_llamacpp_runtime(requested_provider, explicit_api_key) if requested_norm and requested_norm != "custom" and rp._resolves_to_custom(requested_norm): requested_norm = "custom" @@ -577,9 +536,7 @@ def _resolve_named_custom_runtime( return None pool_result = rp._try_resolve_from_custom_pool( - base_url, - "custom", - custom_provider.get("api_mode"), + base_url, "custom", custom_provider.get("api_mode"), provider_name=custom_provider.get("provider_key") or custom_provider.get("name"), ) if pool_result: @@ -587,8 +544,9 @@ def _resolve_named_custom_runtime( _apply_custom_provider_extras(custom_provider, target_model, pool_result) return pool_result + explicit_key = (explicit_api_key or "").strip() candidates = [ - (explicit_api_key or "").strip(), + explicit_key, _clean(custom_provider.get("api_key", "")), rp._getenv(_clean(custom_provider.get("key_env", "")), "").strip(), *rp._host_gated_env_key_candidates(base_url, ollama=False), @@ -599,7 +557,7 @@ def _resolve_named_custom_runtime( # mid-session); both wire clients accept a callable api_key (the Entra ID contract). An # explicit --api-key still wins as the one-off recovery escape hatch. key_cmd = _clean(custom_provider.get("key_cmd", "")) - if key_cmd and not rp.has_usable_secret((explicit_api_key or "").strip()): + if key_cmd and not rp.has_usable_secret(explicit_key): from agent.command_token_source import build_command_token_provider token_provider = build_command_token_provider( @@ -608,13 +566,9 @@ def _resolve_named_custom_runtime( if token_provider is not None: api_key = token_provider - result = rp._runtime( - "custom", - custom_provider.get("api_mode") or rp._detect_api_mode_for_url(base_url) or "chat_completions", - base_url, - api_key or "no-key-required", - source=f"custom_provider:{custom_provider.get('name', requested_provider)}", - requested_provider=requested_provider, + result = _custom_runtime( + rp, base_url, api_key, custom_provider.get("api_mode"), + source=f"custom_provider:{custom_provider.get('name', requested_provider)}", requested_provider=requested_provider, ) _apply_custom_provider_extras(custom_provider, target_model, result) From b69a423835bc7229bda0d9362f1a952cdf679cca Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:24:52 -0700 Subject: [PATCH 04/19] =?UTF-8?q?refactor(hermes=5Fcli):=20provider=20clus?= =?UTF-8?q?ter=20=E2=80=94=20fold/pack=20short=20multi-line=20statements?= =?UTF-8?q?=20(AST-identical)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/providers.py | 6 ++-- hermes_cli/route_identity.py | 32 +++++---------------- hermes_cli/runtime_provider.py | 37 +++++++------------------ hermes_cli/runtime_provider_backends.py | 17 +++--------- hermes_cli/runtime_provider_custom.py | 5 +--- 5 files changed, 24 insertions(+), 73 deletions(-) diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py index 32c52e1ed0..8ec9df5700 100644 --- a/hermes_cli/providers.py +++ b/hermes_cli/providers.py @@ -384,8 +384,7 @@ def resolve_user_provider(name: str, user_config: Dict[str, Any]) -> Optional[Pr if not isinstance(entry, dict): return None return _user_pdef( - name, - entry.get("name", "") or name, + name, entry.get("name", "") or name, entry.get("api", "") or entry.get("url", "") or entry.get("base_url", "") or "", entry.get("key_env") or entry.get("api_key_env") or "", entry.get("transport", "openai_chat") or "openai_chat", @@ -491,8 +490,7 @@ def _llamacpp_pdef() -> Optional[ProviderDef]: def resolve_provider_full( - name: str, - user_providers: Optional[Dict[str, Any]] = None, + name: str, user_providers: Optional[Dict[str, Any]] = None, custom_providers: Optional[List[Dict[str, Any]]] = None, ) -> Optional[ProviderDef]: """Full resolution chain: user ``providers.`` -> lossy-alias registry id -> built-in diff --git a/hermes_cli/route_identity.py b/hermes_cli/route_identity.py index 9e7d45194b..90d4e0c9cf 100644 --- a/hermes_cli/route_identity.py +++ b/hermes_cli/route_identity.py @@ -48,12 +48,8 @@ def normalize_route_base_url(base_url: Any) -> str: def should_clear_context_pin( - configured_model: Any, - active_model: Any, - configured_base_url: Any, - active_base_url: Any, - configured_provider: Any, - active_provider: Any, + configured_model: Any, active_model: Any, configured_base_url: Any, active_base_url: Any, + configured_provider: Any, active_provider: Any, ) -> bool: """True when a configured ``model.context_length`` pin no longer matches its runtime route. @@ -66,23 +62,14 @@ def should_clear_context_pin( try: from agent.agent_init import _context_route_mismatch - return _context_route_mismatch( - configured_base_url, - active_base_url, - configured_provider, - active_provider, - ) + return _context_route_mismatch(configured_base_url, active_base_url, configured_provider, active_provider) except Exception: return True async def should_clear_context_pin_async( - configured_model: Any, - active_model: Any, - configured_base_url: Any, - active_base_url: Any, - configured_provider: Any, - active_provider: Any, + configured_model: Any, active_model: Any, configured_base_url: Any, active_base_url: Any, + configured_provider: Any, active_provider: Any, ) -> bool: """Async wrapper for ``should_clear_context_pin``. @@ -93,11 +80,6 @@ async def should_clear_context_pin_async( import asyncio return await asyncio.to_thread( - should_clear_context_pin, - configured_model, - active_model, - configured_base_url, - active_base_url, - configured_provider, - active_provider, + should_clear_context_pin, configured_model, active_model, configured_base_url, active_base_url, + configured_provider, active_provider, ) diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index eb80f77a21..52ecd5ab61 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -179,8 +179,7 @@ def _resolve_plain_custom_api_mode(model_cfg: Dict[str, Any], base_url: str) -> detected_mode = _detect_api_mode_for_url(base_url) if configured_mode == "codex_responses" and detected_mode != "codex_responses": logger.info( - "Ignoring persisted custom api_mode=codex_responses for non-OpenAI endpoint %s", - base_url or "(unknown)", + "Ignoring persisted custom api_mode=codex_responses for non-OpenAI endpoint %s", base_url or "(unknown)", ) configured_mode = None return configured_mode or detected_mode or "chat_completions" @@ -571,13 +570,8 @@ def _pool_entry_mode_and_url(provider, entry, model_cfg, effective_model, base_u def _resolve_runtime_from_pool_entry( - *, - provider: str, - entry: PooledCredential, - requested_provider: str, - model_cfg: Optional[Dict[str, Any]] = None, - pool: Optional[CredentialPool] = None, - target_model: Optional[str] = None, + *, provider: str, entry: PooledCredential, requested_provider: str, model_cfg: Optional[Dict[str, Any]] = None, + pool: Optional[CredentialPool] = None, target_model: Optional[str] = None, ) -> Dict[str, Any]: model_cfg = model_cfg or _get_model_config() api_mode, base_url = _pool_entry_mode_and_url( @@ -741,13 +735,8 @@ _EXPLICIT_RESOLVERS: Dict[str, Callable[..., Dict[str, Any]]] = { def _resolve_explicit_runtime( - *, - provider: str, - requested_provider: str, - model_cfg: Dict[str, Any], - explicit_api_key: Optional[str] = None, - explicit_base_url: Optional[str] = None, - target_model: Optional[str] = None, + *, provider: str, requested_provider: str, model_cfg: Dict[str, Any], explicit_api_key: Optional[str] = None, + explicit_base_url: Optional[str] = None, target_model: Optional[str] = None, ) -> Optional[Dict[str, Any]]: explicit_api_key = str(explicit_api_key or "").strip() explicit_base_url = str(explicit_base_url or "").strip().rstrip("/") @@ -818,12 +807,9 @@ def _resolve_oauth_runtime(provider, requested_provider, model_cfg, target_model if callable(api_mode): api_mode = api_mode(_effective_model(model_cfg, target_model)) return _runtime( - provider, api_mode, - (creds.get("base_url") or "").rstrip("/") or spec.default_base_url, - creds.get("api_key", ""), - source=creds.get("source", spec.default_source), - **{spec.expiry_key: creds.get(spec.expiry_key)}, - requested_provider=requested_provider, + provider, api_mode, (creds.get("base_url") or "").rstrip("/") or spec.default_base_url, + creds.get("api_key", ""), source=creds.get("source", spec.default_source), + **{spec.expiry_key: creds.get(spec.expiry_key)}, requested_provider=requested_provider, ) @@ -1021,11 +1007,8 @@ def _opencode_free_runtime(provider, requested_provider, model_cfg, target_model def resolve_runtime_provider( - *, - requested: Optional[str] = None, - explicit_api_key: Optional[str] = None, - explicit_base_url: Optional[str] = None, - target_model: Optional[str] = None, + *, requested: Optional[str] = None, explicit_api_key: Optional[str] = None, + explicit_base_url: Optional[str] = None, target_model: Optional[str] = None, ) -> Dict[str, Any]: """Resolve runtime provider credentials for agent execution. diff --git a/hermes_cli/runtime_provider_backends.py b/hermes_cli/runtime_provider_backends.py index f9ddf0afc6..28acb1e22e 100644 --- a/hermes_cli/runtime_provider_backends.py +++ b/hermes_cli/runtime_provider_backends.py @@ -66,12 +66,8 @@ def _azure_foundry_api_key(rp, explicit_api_key: str) -> str: def _resolve_azure_foundry_runtime( - *, - requested_provider: str, - model_cfg: Dict[str, Any], - explicit_api_key: Optional[str] = None, - explicit_base_url: Optional[str] = None, - target_model: Optional[str] = None, + *, requested_provider: str, model_cfg: Dict[str, Any], explicit_api_key: Optional[str] = None, + explicit_base_url: Optional[str] = None, target_model: Optional[str] = None, ) -> Dict[str, Any]: """Azure Foundry: ``model.base_url`` + ``model.api_mode`` (or explicit overrides), API key from ``.env``/env or a per-request Entra ID token, trailing ``/v1`` stripped for Anthropic-style @@ -224,13 +220,8 @@ def _resolve_bedrock_runtime(requested_provider: str, model_cfg: Dict[str, Any], AWS_BEARER_TOKEN_BEDROCK auth is unsupported by AnthropicBedrock (SigV4 only), so bearer users go through Converse regardless of model.""" from agent.bedrock_adapter import ( - bedrock_openai_base_url, - has_aws_credentials, - is_anthropic_bedrock_model, - is_openai_bedrock_model, - resolve_aws_auth_env_var, - resolve_bedrock_bearer_token, - resolve_bedrock_runtime_region, + bedrock_openai_base_url, has_aws_credentials, is_anthropic_bedrock_model, is_openai_bedrock_model, + resolve_aws_auth_env_var, resolve_bedrock_bearer_token, resolve_bedrock_runtime_region, ) from hermes_cli.config import load_config # direct (not the origin delegate), as before diff --git a/hermes_cli/runtime_provider_custom.py b/hermes_cli/runtime_provider_custom.py index 51be816abe..d5a556fde3 100644 --- a/hermes_cli/runtime_provider_custom.py +++ b/hermes_cli/runtime_provider_custom.py @@ -507,10 +507,7 @@ def _opencode_family_for_custom(requested_provider: str, base_url: str) -> Optio def _resolve_named_custom_runtime( - *, - requested_provider: str, - explicit_api_key: Optional[str] = None, - explicit_base_url: Optional[str] = None, + *, requested_provider: str, explicit_api_key: Optional[str] = None, explicit_base_url: Optional[str] = None, target_model: Optional[str] = None, ) -> Optional[Dict[str, Any]]: """Runtime for a llamacpp alias, a bare-custom direct alias, or a configured custom entry. From c73a0c3ef001c39a5b58acae0b93a548025752b3 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:33:02 -0700 Subject: [PATCH 05/19] =?UTF-8?q?refactor(hermes=5Fcli):=20route=5Fidentit?= =?UTF-8?q?y/provider=5Fcatalog=20=E2=80=94=20*args=20async=20wrapper,=20u?= =?UTF-8?q?rl-var=20predicate,=20compact=20docs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/provider_catalog.py | 99 ++++++++++++---------------------- hermes_cli/route_identity.py | 23 +++----- 2 files changed, 41 insertions(+), 81 deletions(-) diff --git a/hermes_cli/provider_catalog.py b/hermes_cli/provider_catalog.py index d80e3dbe49..eda5631fa3 100644 --- a/hermes_cli/provider_catalog.py +++ b/hermes_cli/provider_catalog.py @@ -1,35 +1,26 @@ """Unified provider catalog — one source of truth for the provider universe. The provider list shown by ``hermes model`` (CLI/TUI) and the desktop Settings → Providers tabs -(Accounts + API keys) **must be the same set**. Every provider added after those lists were written -silently went missing from the GUI — e.g. - -* ``auth_type`` / ``api_key_env_vars`` / ``base_url_env_var`` from -:data:`hermes_cli.auth.PROVIDER_REGISTRY` (credential truth), and * ``display_name`` / -``description`` / ``signup_url`` from the provider's :class:`providers.base.ProviderProfile` when -one exists, falling back to the ``CANONICAL_PROVIDERS`` entry's ``label`` / ``tui_desc`` and the -``OPTIONAL_ENV_VARS`` signup URL otherwise (many profiles leave these blank, and four canonical -providers have no profile at all — lmstudio, openai-api, tencent-tokenhub, xai-oauth — so the -fallbacks are load-bearing). +(Accounts + API keys) **must be the same set**; providers added after those lists were written +silently went missing from the GUI. ``auth_type`` / ``api_key_env_vars`` / ``base_url_env_var`` +come from :data:`hermes_cli.auth.PROVIDER_REGISTRY` (credential truth); ``display_name`` / +``description`` / ``signup_url`` from the provider's :class:`providers.base.ProviderProfile`, falling +back to the ``CANONICAL_PROVIDERS`` entry's ``label`` / ``tui_desc`` and the ``OPTIONAL_ENV_VARS`` +signup URL (many profiles leave these blank, and lmstudio, openai-api, tencent-tokenhub, xai-oauth +have no profile at all — the fallbacks are load-bearing). """ from __future__ import annotations from dataclasses import dataclass -# Auth types that authenticate via an account / sign-in flow rather than a -# pasted API key. These route to the desktop "Accounts" tab; everything else -# (api_key, and aws_sdk which is configured via AWS_REGION/AWS_PROFILE) routes -# to the "API keys" tab. Mirrors the auth_type strings used in -# hermes_cli.auth.PROVIDER_REGISTRY and providers.base.ProviderProfile. +# Auth types that authenticate via an account / sign-in flow rather than a pasted API key; these +# route to the desktop "Accounts" tab, everything else (api_key, and aws_sdk configured via +# AWS_REGION/AWS_PROFILE) to "API keys". Mirrors the auth_type strings in PROVIDER_REGISTRY and +# ProviderProfile: external_process = copilot-acp (spawns `copilot --acp --stdio`), copilot = GitHub +# Copilot token / gh auth. _ACCOUNTS_AUTH_TYPES: frozenset[str] = frozenset( - { - "oauth_device_code", - "oauth_external", - "oauth_minimax", - "external_process", # copilot-acp: spawns `copilot --acp --stdio` - "copilot", # GitHub Copilot token / gh auth - } + {"oauth_device_code", "oauth_external", "oauth_minimax", "external_process", "copilot"} ) @@ -54,19 +45,19 @@ def tab_for_auth_type(auth_type: str) -> str: return "accounts" if auth_type in _ACCOUNTS_AUTH_TYPES else "keys" +def _is_url_var(name: str) -> bool: + return name.endswith("_BASE_URL") or name.endswith("_URL") + + def _split_env_vars(env_vars: tuple[str, ...]) -> tuple[tuple[str, ...], str]: """Split a profile's ``env_vars`` into (api_key_vars, base_url_var).""" - keys = tuple(v for v in env_vars if not (v.endswith("_BASE_URL") or v.endswith("_URL"))) - base = next((v for v in env_vars if v.endswith("_BASE_URL") or v.endswith("_URL")), "") - return keys, base + return tuple(v for v in env_vars if not _is_url_var(v)), next((v for v in env_vars if _is_url_var(v)), "") def _safe_import(module: str, attr: str, default): - """Import ``attr`` from ``module``; return ``default`` on ANY failure. - - This module is on the import path of the web server and the CLI, and a - provider-plugin import error must never blank the whole catalog. - """ + """Import ``attr`` from ``module``; ``default`` on ANY failure — this module is on the import + path of the web server and the CLI, and a provider-plugin import error must never blank the + whole catalog.""" try: return getattr(__import__(module, fromlist=[attr]), attr) except Exception: @@ -74,19 +65,15 @@ def _safe_import(module: str, attr: str, default): def provider_catalog() -> list[ProviderDescriptor]: - """Return one descriptor per provider in the ``hermes model`` universe. - - Membership is :data:`CANONICAL_PROVIDERS` (auto-extended by provider plugins). Auth/env come - from ``PROVIDER_REGISTRY``; display metadata from ``ProviderProfile`` with canonical/env - fallbacks so providers without a profile still resolve sensibly. - """ + """One descriptor per provider in the ``hermes model`` universe (:data:`CANONICAL_PROVIDERS`, + auto-extended by provider plugins). Auth/env from ``PROVIDER_REGISTRY``; display metadata from + ``ProviderProfile`` with canonical/env fallbacks so profile-less providers still resolve.""" from hermes_cli.models import CANONICAL_PROVIDERS PROVIDER_REGISTRY = _safe_import("hermes_cli.auth", "PROVIDER_REGISTRY", {}) OPTIONAL_ENV_VARS = _safe_import("hermes_cli.config", "OPTIONAL_ENV_VARS", {}) - # Hermes overlays carry auth_type for providers that have no registry/profile - # entry of their own — notably the ``moa`` virtual provider (auth_type - # "virtual"), which has no real credential and no network endpoint. + # Overlays carry auth_type for providers with no registry/profile entry — notably the ``moa`` + # virtual provider (auth_type "virtual"), which has no credential and no network endpoint. HERMES_OVERLAYS = _safe_import("hermes_cli.providers", "HERMES_OVERLAYS", {}) try: from providers import list_providers @@ -101,47 +88,31 @@ def provider_catalog() -> list[ProviderDescriptor]: cfg = PROVIDER_REGISTRY.get(slug) prof = profiles.get(slug) overlay = HERMES_OVERLAYS.get(slug) - - # auth_type: registry is authoritative; fall back to profile, then the - # Hermes overlay (e.g. moa → "virtual"), then api_key. + # auth_type: registry is authoritative; then profile, then overlay (moa → "virtual"), then api_key. auth_type = ( (cfg.auth_type if cfg else "") or (prof.auth_type if prof else "") or (overlay.auth_type if overlay else "") or "api_key" ) - - # Credential env vars: registry first (it already normalizes these), - # else derive from the profile's env_vars tuple. + # Credential env vars: registry first (already normalized), else derived from the profile. if cfg and cfg.api_key_env_vars: - api_key_vars = tuple(cfg.api_key_env_vars) - base_url_var = cfg.base_url_env_var or "" + api_key_vars, base_url_var = tuple(cfg.api_key_env_vars), cfg.base_url_env_var or "" elif prof and prof.env_vars: api_key_vars, base_url_var = _split_env_vars(tuple(prof.env_vars)) else: api_key_vars, base_url_var = (), "" - label = (prof.display_name if prof else "") or entry.label or slug - description = (prof.description if prof else "") or entry.tui_desc or label signup_url = (prof.signup_url if prof else "") or "" if not signup_url and api_key_vars: - info = OPTIONAL_ENV_VARS.get(api_key_vars[0]) or {} - signup_url = info.get("url") or "" - + signup_url = (OPTIONAL_ENV_VARS.get(api_key_vars[0]) or {}).get("url") or "" out.append( ProviderDescriptor( - slug=slug, - label=label, - description=description, - auth_type=auth_type, - tab=tab_for_auth_type(auth_type), - api_key_env_vars=api_key_vars, - base_url_env_var=base_url_var, - signup_url=signup_url, - order=order, - # Keyless providers (e.g. opencode-free) are served - # anonymously: there is no credential to configure, so the - # GUI renders no key card and contract tests exempt them. + slug=slug, label=label, description=(prof.description if prof else "") or entry.tui_desc or label, + auth_type=auth_type, tab=tab_for_auth_type(auth_type), api_key_env_vars=api_key_vars, + base_url_env_var=base_url_var, signup_url=signup_url, order=order, + # Keyless providers (opencode-free) are served anonymously: no key card in the GUI, + # and contract tests exempt them. keyless=bool(overlay.keyless) if overlay else False, ) ) diff --git a/hermes_cli/route_identity.py b/hermes_cli/route_identity.py index 90d4e0c9cf..65c5be808b 100644 --- a/hermes_cli/route_identity.py +++ b/hermes_cli/route_identity.py @@ -52,10 +52,8 @@ def should_clear_context_pin( configured_provider: Any, active_provider: Any, ) -> bool: """True when a configured ``model.context_length`` pin no longer matches its runtime route. - Fail-closed: any error during route comparison returns ``True`` (drop the pin) so a stale window - never silently inflates the compression threshold. - """ + never silently inflates the compression threshold.""" configured_model = str(configured_model or "").strip() if configured_model and configured_model != str(active_model or "").strip(): return True @@ -67,19 +65,10 @@ def should_clear_context_pin( return True -async def should_clear_context_pin_async( - configured_model: Any, active_model: Any, configured_base_url: Any, active_base_url: Any, - configured_provider: Any, active_provider: Any, -) -> bool: - """Async wrapper for ``should_clear_context_pin``. - - Offloads the route comparison to a worker thread so async gateway handlers never run it on the - event loop — the resolution chain is cache-only (``allow_network=False``) but can still do cold- - start disk I/O. Shares all logic with the sync version — no code duplication. - """ +async def should_clear_context_pin_async(*args: Any) -> bool: + """``should_clear_context_pin`` on a worker thread so async gateway handlers never run it on the + event loop — the resolution chain is cache-only (``allow_network=False``) but can still do + cold-start disk I/O.""" import asyncio - return await asyncio.to_thread( - should_clear_context_pin, configured_model, active_model, configured_base_url, active_base_url, - configured_provider, active_provider, - ) + return await asyncio.to_thread(should_clear_context_pin, *args) From 588bec52631edcce4598f758ac5561e02f50a5f5 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:45:55 -0700 Subject: [PATCH 06/19] =?UTF-8?q?refactor(hermes=5Fcli):=20runtime=5Fprovi?= =?UTF-8?q?der=20=E2=80=94=20module-attribute=20access=20replaces=2012=20l?= =?UTF-8?q?azy=20re-imports,=20drop=20thin=20nous/ttl=20wrappers,=20pack?= =?UTF-8?q?=20tables?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/runtime_provider.py | 175 +++++++++------------------------ 1 file changed, 49 insertions(+), 126 deletions(-) diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index 52ecd5ab61..202cd2647f 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -4,8 +4,7 @@ This module owns the resolution ORDER (:func:`resolve_runtime_provider`), the ap helpers and the pool / OAuth / explicit paths. Custom-provider lookup lives in :mod:`hermes_cli.runtime_provider_custom`; Azure Foundry, OpenRouter/bare-custom, Bedrock and external-process builders in :mod:`hermes_cli.runtime_provider_backends`. Both are re-exported -here so ``hermes_cli.runtime_provider.`` imports and test patches keep working. -""" +here so ``hermes_cli.runtime_provider.`` imports and test patches keep working.""" from __future__ import annotations @@ -18,38 +17,23 @@ from urllib.parse import urlparse logger = logging.getLogger(__name__) from hermes_cli import auth as auth_mod -from agent.credential_pool import ( - CredentialPool, - PooledCredential, - credential_pool_matches_provider, - custom_provider_pool_key_candidates, # noqa: F401 — read via origin by runtime_provider_custom (patchable) +from agent.credential_pool import ( # custom_provider_pool_key_candidates is read via origin by runtime_provider_custom + CredentialPool, PooledCredential, credential_pool_matches_provider, custom_provider_pool_key_candidates, # noqa: F401 load_pool, ) from agent.secret_scope import get_secret as _get_secret -from hermes_cli.auth import ( - ACTUAL_LOCAL_NOAUTH_PLACEHOLDER, - AuthError, - DEFAULT_CODEX_BASE_URL, - DEFAULT_QWEN_BASE_URL, - DEFAULT_XAI_OAUTH_BASE_URL, - PROVIDER_REGISTRY, - _agent_key_is_usable, - _nous_inference_env_override, - format_auth_error, - resolve_provider, - resolve_nous_runtime_credentials, - resolve_codex_runtime_credentials, - resolve_xai_oauth_runtime_credentials, - resolve_qwen_runtime_credentials, - resolve_api_key_provider_credentials, - resolve_external_process_provider_credentials, # noqa: F401 — read via origin by runtime_provider_backends - has_usable_secret, - is_actual_local_base_url, - normalize_actual_base_url, +from hermes_cli.auth import ( # resolve_external_process_provider_credentials is read via origin by runtime_provider_backends + ACTUAL_LOCAL_NOAUTH_PLACEHOLDER, AuthError, DEFAULT_CODEX_BASE_URL, DEFAULT_QWEN_BASE_URL, DEFAULT_XAI_OAUTH_BASE_URL, + PROVIDER_REGISTRY, _agent_key_is_usable, _nous_inference_env_override, format_auth_error, resolve_provider, + resolve_nous_runtime_credentials, resolve_codex_runtime_credentials, resolve_xai_oauth_runtime_credentials, + resolve_qwen_runtime_credentials, resolve_api_key_provider_credentials, + resolve_external_process_provider_credentials, # noqa: F401 + has_usable_secret, is_actual_local_base_url, normalize_actual_base_url, ) from hermes_cli import config as _config_mod +from hermes_cli import models as _models # attribute access keeps ``hermes_cli.models.`` patches effective from hermes_constants import OPENROUTER_BASE_URL -from hermes_cli.providers import is_official_openai_host +from hermes_cli.providers import determine_api_mode, is_official_openai_host, nous_api_mode from utils import base_url_host_matches, base_url_hostname, env_int @@ -93,8 +77,7 @@ def _config_base_url_trustworthy_for_bare_custom(cfg_base_url: str, cfg_provider The model picker can select Custom while ``model.provider`` still reflects a previous provider. Non-loopback URLs are rejected unless the YAML provider is already ``custom`` or a local-server alias (ollama/vllm/llamacpp — else a legit LAN ollama endpoint silently falls through to - OpenRouter), so a stale OpenRouter/Z.ai base_url cannot hijack local sessions. - """ + OpenRouter), so a stale OpenRouter/Z.ai base_url cannot hijack local sessions.""" cfg_provider_norm = (cfg_provider or "").strip().lower() bu = (cfg_base_url or "").strip() if not bu: @@ -112,11 +95,8 @@ def _config_base_url_trustworthy_for_bare_custom(cfg_base_url: str, cfg_provider # so the runtime resolver stays in lockstep: api.meta.ai — prompt caching only on Responses; # api.router.com — /v1/chat/completions is a minimal shim; api.anthropic.com — native Messages. _HOST_MANDATED_API_MODES = { - "api.x.ai": "codex_responses", - "api.meta.ai": "codex_responses", - "api.actual.inc": "codex_responses", - "api.router.com": "codex_responses", - "api.anthropic.com": "anthropic_messages", + "api.x.ai": "codex_responses", "api.meta.ai": "codex_responses", "api.actual.inc": "codex_responses", + "api.router.com": "codex_responses", "api.anthropic.com": "anthropic_messages", } # codex_app_server is opt-in: hand the whole turn to a `codex app-server` subprocess (Codex's own @@ -129,8 +109,7 @@ def _detect_api_mode_for_url(base_url: str) -> Optional[str]: Exact-hostname matches reject lookalike subdomains (api.anthropic.com.attacker.test) and path-segment spoofing (proxy.test/api.anthropic.com/v1). Official OpenAI hosts (incl. the - data-residency us./eu. regional hosts) need Responses for GPT-5.x tool calls with reasoning. - """ + data-residency us./eu. regional hosts) need Responses for GPT-5.x tool calls with reasoning.""" normalized = (base_url or "").strip().lower().rstrip("/") hostname = base_url_hostname(base_url) mandated = _HOST_MANDATED_API_MODES.get(hostname) @@ -151,9 +130,7 @@ def _parse_api_mode(raw: Any) -> Optional[str]: ``anthropic``, ``responses``, …) are canonicalized first so old configs keep their transport instead of silently falling through to hostname-based detection.""" if isinstance(raw, str): - from hermes_cli.config import _canonical_api_mode - - normalized = _canonical_api_mode(raw).lower() + normalized = _config_mod._canonical_api_mode(raw).lower() if normalized in _VALID_API_MODES: return normalized return None @@ -164,12 +141,7 @@ def _fallback_api_mode(provider: str, base_url: str, model: str = "") -> str: first, then the transport the provider overlay declares via ``providers.determine_api_mode`` (``openai-api`` pointed at us.api.openai.com 400'd on every tool call without it), then ``chat_completions``.""" - detected = _detect_api_mode_for_url(base_url) - if detected: - return detected - from hermes_cli.providers import determine_api_mode - - return determine_api_mode(provider, base_url, model) or "chat_completions" + return _detect_api_mode_for_url(base_url) or determine_api_mode(provider, base_url, model) or "chat_completions" def _resolve_plain_custom_api_mode(model_cfg: Dict[str, Any], base_url: str) -> str: @@ -221,9 +193,7 @@ def _copilot_runtime_api_mode(model_cfg: Dict[str, Any], api_key: str, *, target if not model_name: return "chat_completions" try: - from hermes_cli.models import copilot_model_api_mode - - return copilot_model_api_mode(model_name, api_key=api_key) + return _models.copilot_model_api_mode(model_name, api_key=api_key) except Exception: return "chat_completions" @@ -234,9 +204,7 @@ def _azure_inferred_api_mode(effective_model: str, api_mode: str) -> str: if not effective_model or api_mode == "anthropic_messages": return api_mode try: - from hermes_cli.models import azure_foundry_model_api_mode - - inferred = azure_foundry_model_api_mode(effective_model) + inferred = _models.azure_foundry_model_api_mode(effective_model) except Exception: inferred = None return inferred or api_mode @@ -248,15 +216,9 @@ def _configured_or_fallback_api_mode( """Persisted ``model.api_mode`` when it belongs to this provider, else URL/transport fallback. OpenCode Zen/Go serve both anthropic_messages and chat_completions models, so (when - ``opencode_by_model``) their mode is always re-derived from the effective model. - """ - if opencode_by_model: - from hermes_cli.models import opencode_provider_family - - if opencode_provider_family(provider) is not None: - from hermes_cli.models import opencode_model_api_mode - - return opencode_model_api_mode(provider, effective_model) + ``opencode_by_model``) their mode is always re-derived from the effective model.""" + if opencode_by_model and _models.opencode_provider_family(provider) is not None: + return _models.opencode_model_api_mode(provider, effective_model) return _configured_api_mode(provider, model_cfg) or _fallback_api_mode(provider, base_url, effective_model) @@ -362,8 +324,7 @@ def _host_gated_env_key_candidates(base_url: str, *, ollama: bool) -> list: Sending OPENAI/OPENROUTER/OLLAMA keys to an unrelated endpoint leaks credentials (GHSA-76xc-57q6-vm5m); match on HOST, not substring. ``_host_derived_api_key`` skips OLLAMA, - so callers that want it opt in via ``ollama``. - """ + so callers that want it opt in via ``ollama``.""" is_openai = base_url_host_matches(base_url, "openai.com") or base_url_host_matches(base_url, "openai.azure.com") candidates = [] if ollama: @@ -393,12 +354,6 @@ def _registry_base_url(provider: str) -> str: return pconfig.inference_base_url if pconfig else "" -def _nous_inference_base_url_override() -> str: - """Trusted ``NOUS_INFERENCE_BASE_URL`` override (bypasses the network host allowlist); one - normalization path via ``auth._nous_inference_env_override``.""" - return _nous_inference_env_override() or "" - - def _nous_min_key_ttl() -> int: return max(60, env_int("HERMES_NOUS_MIN_KEY_TTL_SECONDS", 1800)) @@ -407,22 +362,12 @@ def _resolve_nous_creds() -> Dict[str, Any]: return resolve_nous_runtime_credentials(timeout_seconds=float(_getenv("HERMES_NOUS_TIMEOUT_SECONDS", "15"))) -def _nous_api_mode(model: str) -> str: - from hermes_cli.providers import nous_api_mode - - return nous_api_mode(model) - - def _finalize_base_url(provider: str, api_mode: str, base_url: str) -> str: """Shared tail for pool-entry and api-key paths: OpenCode /v1 rule (OpenCode URLs end with /v1 for OpenAI-compatible models but the Anthropic SDK prepends its own /v1/messages — strip for anthropic_messages, re-append otherwise), then LM Studio normalization.""" - from hermes_cli.models import opencode_provider_family - - if opencode_provider_family(provider) is not None: - from hermes_cli.models import normalize_opencode_base_url - - base_url = normalize_opencode_base_url(provider, api_mode, base_url) + if _models.opencode_provider_family(provider) is not None: + base_url = _models.normalize_opencode_base_url(provider, api_mode, base_url) if provider == "lmstudio": base_url = auth_mod._normalize_lmstudio_runtime_base_url(base_url) return base_url @@ -465,9 +410,7 @@ def _get_model_config() -> Dict[str, Any]: cfg["default"] = cfg["model"] _default = cfg.get("default") if isinstance(_default, dict): - from hermes_cli.config import split_model_config_default - - cfg_model, cfg_provider = split_model_config_default(_default) + cfg_model, cfg_provider = _config_mod.split_model_config_default(_default) cfg_provider = cfg_provider or str(model_cfg.get("provider") or "") cfg["default"] = cfg_model if cfg_provider and not cfg.get("provider"): @@ -531,12 +474,9 @@ from hermes_cli.runtime_provider_backends import ( # noqa: E402,F401 # are valid only against the Anthropic Messages endpoint, so a stale model.api_mode from a prior # OpenAI-compatible provider is never honoured for it (it would 404 on /chat/completions). _POOL_ENTRY_SIMPLE_MODES: Dict[str, tuple] = { - "openai-codex": ("codex_responses", DEFAULT_CODEX_BASE_URL), - "xai-oauth": ("codex_responses", DEFAULT_XAI_OAUTH_BASE_URL), - "qwen-oauth": ("chat_completions", DEFAULT_QWEN_BASE_URL), - "minimax-oauth": ("anthropic_messages", lambda: _registry_base_url("minimax-oauth")), - "openrouter": ("chat_completions", OPENROUTER_BASE_URL), - "xai": ("codex_responses", ""), + "openai-codex": ("codex_responses", DEFAULT_CODEX_BASE_URL), "xai-oauth": ("codex_responses", DEFAULT_XAI_OAUTH_BASE_URL), + "qwen-oauth": ("chat_completions", DEFAULT_QWEN_BASE_URL), "openrouter": ("chat_completions", OPENROUTER_BASE_URL), + "minimax-oauth": ("anthropic_messages", lambda: _registry_base_url("minimax-oauth")), "xai": ("codex_responses", ""), } @@ -548,7 +488,7 @@ def _pool_entry_mode_and_url(provider, entry, model_cfg, effective_model, base_u if provider == "anthropic": return "anthropic_messages", _anthropic_cfg_base_url(model_cfg) or base_url or _ANTHROPIC_DEFAULT_BASE_URL if provider == "nous": - return _nous_api_mode(effective_model), _nous_inference_base_url_override() or base_url + return nous_api_mode(effective_model), (_nous_inference_env_override() or "") or base_url if provider == "copilot": api_mode = _copilot_runtime_api_mode(model_cfg, getattr(entry, "runtime_api_key", ""), target_model=effective_model) return api_mode, base_url or PROVIDER_REGISTRY["copilot"].inference_base_url @@ -674,7 +614,7 @@ def _explicit_nous(requested_provider, model_cfg, api_key, explicit_base_url, ta state = auth_mod.get_provider_auth_state("nous") or {} base_url = ( explicit_base_url - or _nous_inference_base_url_override() + or (_nous_inference_env_override() or "") or str(state.get("inference_base_url") or auth_mod.DEFAULT_NOUS_INFERENCE_URL).strip().rstrip("/") ) # The agent_key compatibility field is used for inference only when it holds a NAS invoke JWT; @@ -689,7 +629,7 @@ def _explicit_nous(requested_provider, model_cfg, api_key, explicit_base_url, ta expires_at = creds.get("expires_at") base_url = explicit_base_url or creds.get("base_url", "").rstrip("/") or base_url return _runtime( - "nous", _nous_api_mode(_effective_model(model_cfg, target_model)), base_url, api_key, + "nous", nous_api_mode(_effective_model(model_cfg, target_model)), base_url, api_key, source="explicit", expires_at=expires_at, requested_provider=requested_provider, ) @@ -727,9 +667,7 @@ def _explicit_api_key_provider(provider, pconfig, requested_provider, model_cfg, # Providers with a dedicated explicit-credential builder; everything else goes through the # registry ``api_key`` path (or None when the provider takes no explicit creds). _EXPLICIT_RESOLVERS: Dict[str, Callable[..., Dict[str, Any]]] = { - "anthropic": _explicit_anthropic, - "openai-codex": _explicit_codex, - "nous": _explicit_nous, + "anthropic": _explicit_anthropic, "openai-codex": _explicit_codex, "nous": _explicit_nous, "azure-foundry": _explicit_azure_foundry, } @@ -771,23 +709,15 @@ class _OAuthRuntimeSpec: # ``resolve`` entries are late-bound lambdas so tests can monkeypatch the module-level # ``resolve_*_runtime_credentials`` names. _OAUTH_RUNTIME_PROVIDERS: Dict[str, _OAuthRuntimeSpec] = { - "nous": _OAuthRuntimeSpec( - _resolve_nous_creds, _nous_api_mode, "portal", "expires_at", - "Auto-detected Nous provider but credentials failed", - ), - "openai-codex": _OAuthRuntimeSpec( - lambda: resolve_codex_runtime_credentials(), "codex_responses", "hermes-auth-store", "last_refresh", - "Auto-detected Codex provider but credentials failed", - ), - "xai-oauth": _OAuthRuntimeSpec( - lambda: resolve_xai_oauth_runtime_credentials(), "codex_responses", "hermes-auth-store", "last_refresh", - "Auto-detected xAI OAuth provider but credentials failed", - default_base_url=DEFAULT_XAI_OAUTH_BASE_URL, - ), - "qwen-oauth": _OAuthRuntimeSpec( - lambda: resolve_qwen_runtime_credentials(), "chat_completions", "qwen-cli", "expires_at_ms", - "Qwen OAuth credentials failed", - ), + "nous": _OAuthRuntimeSpec(_resolve_nous_creds, nous_api_mode, "portal", "expires_at", + "Auto-detected Nous provider but credentials failed"), + "openai-codex": _OAuthRuntimeSpec(lambda: resolve_codex_runtime_credentials(), "codex_responses", "hermes-auth-store", + "last_refresh", "Auto-detected Codex provider but credentials failed"), + "xai-oauth": _OAuthRuntimeSpec(lambda: resolve_xai_oauth_runtime_credentials(), "codex_responses", "hermes-auth-store", + "last_refresh", "Auto-detected xAI OAuth provider but credentials failed", + default_base_url=DEFAULT_XAI_OAUTH_BASE_URL), + "qwen-oauth": _OAuthRuntimeSpec(lambda: resolve_qwen_runtime_credentials(), "chat_completions", "qwen-cli", + "expires_at_ms", "Qwen OAuth credentials failed"), } @@ -817,9 +747,7 @@ def _minimax_oauth_runtime(provider, requested_provider) -> Optional[Dict[str, A pconfig = PROVIDER_REGISTRY.get(provider) if not (pconfig and pconfig.auth_type == "oauth_minimax"): return None - from hermes_cli.auth import resolve_minimax_oauth_runtime_credentials - - creds = resolve_minimax_oauth_runtime_credentials() + creds = auth_mod.resolve_minimax_oauth_runtime_credentials() return _runtime( provider, "anthropic_messages", creds["base_url"], creds["api_key"], source=creds.get("source", "oauth"), requested_provider=requested_provider, @@ -905,12 +833,10 @@ _LOCAL_BYPASS_CLOUD_HOSTS = ("openrouter.ai", "anthropic.com", "openai.com") def _raise_if_provider_disabled(requested_provider: str) -> None: """Honour ``providers..enabled: false`` for built-ins too (the custom lookup gate only covers custom blocks); a typed error lets the fallback chain advance.""" - from hermes_cli.config import is_provider_enabled, load_config - - full_cfg = load_config() + full_cfg = _config_mod.load_config() provs_cfg = full_cfg.get("providers") if isinstance(full_cfg, dict) else None block = provs_cfg.get(requested_provider) if isinstance(provs_cfg, dict) else None - if isinstance(block, dict) and not is_provider_enabled(block): + if isinstance(block, dict) and not _config_mod.is_provider_enabled(block): raise ValueError( f"provider {requested_provider!r} is disabled in config " f"(providers.{requested_provider}.enabled: false)" @@ -995,12 +921,10 @@ def _opencode_free_runtime(provider, requested_provider, model_cfg, target_model """OpenCode Zen free tier (*-free slugs) is served ANONYMOUSLY on the Zen relay only: unknown bearers 401 and the Go relay rejects free models, so free slugs route through the keyless Zen runtime BEFORE the pool / explicit / api_key paths.""" - from hermes_cli.models import opencode_provider_family, opencode_zen_free_runtime - - if opencode_provider_family(provider) is None: + if _models.opencode_provider_family(provider) is None: return None model = str(target_model or model_cfg.get("default") or model_cfg.get("model") or "").strip() - free_runtime = opencode_zen_free_runtime(provider, model) + free_runtime = _models.opencode_zen_free_runtime(provider, model) if free_runtime is not None: free_runtime["requested_provider"] = requested_provider return free_runtime @@ -1024,8 +948,7 @@ def resolve_runtime_provider( 8. OpenRouter / bare-custom fallback target_model: overrides model_cfg["default"] when computing provider-specific api_mode - (e.g. OpenCode Zen/Go where different models route through different API surfaces). - """ + (e.g. OpenCode Zen/Go where different models route through different API surfaces).""" requested_provider = resolve_requested_provider(requested) _raise_if_provider_disabled(requested_provider) From 6740d264f1df300c74fe0e0c163eba9be269782f Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:49:11 -0700 Subject: [PATCH 07/19] =?UTF-8?q?refactor(hermes=5Fcli):=20provider=20clus?= =?UTF-8?q?ter=20=E2=80=94=20drop=20intra-function=20separator=20blanks=20?= =?UTF-8?q?(whitespace-only)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/provider_catalog.py | 3 --- hermes_cli/providers.py | 8 -------- hermes_cli/route_identity.py | 5 ----- hermes_cli/runtime_provider.py | 17 ----------------- hermes_cli/runtime_provider_backends.py | 12 ------------ hermes_cli/runtime_provider_custom.py | 13 ------------- 6 files changed, 58 deletions(-) diff --git a/hermes_cli/provider_catalog.py b/hermes_cli/provider_catalog.py index eda5631fa3..3b6bdd6c81 100644 --- a/hermes_cli/provider_catalog.py +++ b/hermes_cli/provider_catalog.py @@ -69,7 +69,6 @@ def provider_catalog() -> list[ProviderDescriptor]: auto-extended by provider plugins). Auth/env from ``PROVIDER_REGISTRY``; display metadata from ``ProviderProfile`` with canonical/env fallbacks so profile-less providers still resolve.""" from hermes_cli.models import CANONICAL_PROVIDERS - PROVIDER_REGISTRY = _safe_import("hermes_cli.auth", "PROVIDER_REGISTRY", {}) OPTIONAL_ENV_VARS = _safe_import("hermes_cli.config", "OPTIONAL_ENV_VARS", {}) # Overlays carry auth_type for providers with no registry/profile entry — notably the ``moa`` @@ -77,11 +76,9 @@ def provider_catalog() -> list[ProviderDescriptor]: HERMES_OVERLAYS = _safe_import("hermes_cli.providers", "HERMES_OVERLAYS", {}) try: from providers import list_providers - profiles = {p.name: p for p in list_providers()} except Exception: profiles = {} - out: list[ProviderDescriptor] = [] for order, entry in enumerate(CANONICAL_PROVIDERS): slug = entry.slug diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py index 8ec9df5700..9115c89f4d 100644 --- a/hermes_cli/providers.py +++ b/hermes_cli/providers.py @@ -201,7 +201,6 @@ def _models_dev_info(canonical: str, allow_network: bool = True): ``get_provider_info`` with single-arg lambdas.""" try: from agent.models_dev import get_provider_info as _mdev_provider - return _mdev_provider(canonical) if allow_network else _mdev_provider(canonical, allow_network=False) except Exception: return None @@ -216,7 +215,6 @@ def get_provider(name: str, *, allow_network: bool = True) -> Optional[ProviderD canonical = normalize_provider(name) mdev_info = _models_dev_info(canonical, allow_network) overlay = HERMES_OVERLAYS.get(canonical) - if mdev_info is not None: ov = overlay or HermesOverlay() env_vars = list(mdev_info.env) @@ -228,7 +226,6 @@ def get_provider(name: str, *, allow_network: bool = True) -> Optional[ProviderD base_url=ov.base_url_override or mdev_info.api, base_url_env_var=ov.base_url_env_var, is_aggregator=ov.is_aggregator, auth_type=ov.auth_type, doc=mdev_info.doc, source="models.dev", ) - if overlay is not None: return ProviderDef( id=canonical, name=_LABEL_OVERRIDES.get(canonical, canonical), transport=overlay.transport, @@ -236,7 +233,6 @@ def get_provider(name: str, *, allow_network: bool = True) -> Optional[ProviderD base_url_env_var=overlay.base_url_env_var, is_aggregator=overlay.is_aggregator, auth_type=overlay.auth_type, source="hermes", ) - # Plugin-registered profiles (plugins/model-providers//) absent from models.dev and # HERMES_OVERLAYS would otherwise be "Unknown provider" in /model, --provider and model-switch # even though the picker lists them. Only profiles with a concrete endpoint resolve here: @@ -245,7 +241,6 @@ def get_provider(name: str, *, allow_network: bool = True) -> Optional[ProviderD # resolve_provider_full's custom step and collapse keyed ``custom:`` ids to bare custom. try: from providers import get_provider_profile as _profile - _prof = _profile(canonical) if _prof is not None and (_prof.base_url or "").strip(): _api_mode_to_transport = {v: k for k, v in TRANSPORT_TO_API_MODE.items()} @@ -456,7 +451,6 @@ def _lossy_alias_registry_pdef(raw: str, canonical: str) -> Optional[ProviderDef through the built-in chain so overlay transports apply.""" try: from hermes_cli.auth import PROVIDER_REGISTRY as _AUTH_PROVIDER_REGISTRY - _pcfg = _AUTH_PROVIDER_REGISTRY.get(raw) if _pcfg is None: return None @@ -477,7 +471,6 @@ def _llamacpp_pdef() -> Optional[ProviderDef]: provider the Local Models 'Use' flow writes to config.""" try: from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint - endpoint = resolve_llamacpp_endpoint(wait_for_boot_s=0) except Exception: endpoint = None @@ -503,7 +496,6 @@ def resolve_provider_full( """ canonical = normalize_provider(name) raw = name.strip().lower() - if user_providers: user_pdef = resolve_user_provider(raw, user_providers) if user_pdef is not None: diff --git a/hermes_cli/route_identity.py b/hermes_cli/route_identity.py index 65c5be808b..3732874d72 100644 --- a/hermes_cli/route_identity.py +++ b/hermes_cli/route_identity.py @@ -28,7 +28,6 @@ def normalize_route_base_url(base_url: Any) -> str: port = parsed.port except (TypeError, ValueError): return raw - route_host = parsed.netloc.rsplit("@", 1)[-1] if route_host.startswith("[") or ":" in host: host = f"[{host}]" @@ -36,11 +35,9 @@ def normalize_route_base_url(base_url: Any) -> str: host = f"{host}:{port}" if "@" in parsed.netloc: host = f"{parsed.netloc.rsplit('@', 1)[0]}@{host}" - path = parsed.path if path.endswith("/") and not had_query_delimiter: path = path[:-1] - normalized = urlunsplit((scheme, host, path, parsed.query, "")) if had_query_delimiter and not parsed.query: normalized += "?" @@ -59,7 +56,6 @@ def should_clear_context_pin( return True try: from agent.agent_init import _context_route_mismatch - return _context_route_mismatch(configured_base_url, active_base_url, configured_provider, active_provider) except Exception: return True @@ -70,5 +66,4 @@ async def should_clear_context_pin_async(*args: Any) -> bool: event loop — the resolution chain is cache-only (``allow_network=False``) but can still do cold-start disk I/O.""" import asyncio - return await asyncio.to_thread(should_clear_context_pin, *args) diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index 202cd2647f..47c5cd23ee 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -293,7 +293,6 @@ def _anthropic_cfg_base_url(model_cfg: Dict[str, Any]) -> str: def _anthropic_token_or_raise() -> str: from agent.anthropic_adapter import resolve_anthropic_token - token = resolve_anthropic_token() if not token: raise AuthError(_NO_ANTHROPIC_CREDENTIALS_MSG) @@ -382,7 +381,6 @@ def _auto_detect_local_model(base_url: str) -> str: return "" try: import requests - url = base_url.rstrip("/") if not url.endswith("/v1"): url += "/v1" @@ -848,7 +846,6 @@ def _resolve_vertex_runtime(requested_provider: str) -> Dict[str, Any]: treated as a static API key; a short-lived token is minted per call, and mid-session expiry is recovered on 401 by run_agent._try_refresh_vertex_client_credentials().""" from agent.vertex_adapter import get_vertex_config - token, base_url = get_vertex_config() if not token or not base_url: raise AuthError( @@ -951,11 +948,9 @@ def resolve_runtime_provider( (e.g. OpenCode Zen/Go where different models route through different API surfaces).""" requested_provider = resolve_requested_provider(requested) _raise_if_provider_disabled(requested_provider) - runtime = _resolve_requested_shortcuts(requested_provider, explicit_api_key, explicit_base_url, target_model) if runtime: return runtime - runtime = _resolve_named_custom_runtime( requested_provider=requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, target_model=target_model, @@ -963,53 +958,41 @@ def resolve_runtime_provider( if runtime: runtime["requested_provider"] = requested_provider return runtime - if not explicit_base_url and not explicit_api_key: runtime = _local_endpoint_bypass(requested_provider, explicit_api_key, explicit_base_url) if runtime: return runtime - provider = resolve_provider(requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url) model_cfg = _get_model_config() - runtime = _opencode_free_runtime(provider, requested_provider, model_cfg, target_model) if runtime is not None: return runtime - runtime = _resolve_explicit_runtime( provider=provider, requested_provider=requested_provider, model_cfg=model_cfg, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, target_model=target_model, ) if runtime: return runtime - runtime = _resolve_from_pool(provider, requested_provider, model_cfg, explicit_api_key, explicit_base_url, target_model) if runtime: return runtime - if provider in _OAUTH_RUNTIME_PROVIDERS: runtime = _resolve_oauth_runtime(provider, requested_provider, model_cfg, target_model) if runtime: return runtime - if provider == "minimax-oauth": runtime = _minimax_oauth_runtime(provider, requested_provider) if runtime: return runtime - if _is_external_process_provider(provider): return _resolve_external_process_runtime(provider, requested_provider) - if provider == "anthropic": return _anthropic_env_runtime(requested_provider, model_cfg) - if provider == "bedrock": return _resolve_bedrock_runtime(requested_provider, model_cfg, target_model) - pconfig = PROVIDER_REGISTRY.get(provider) if pconfig and pconfig.auth_type == "api_key": return _api_key_provider_runtime(provider, pconfig, requested_provider, model_cfg, target_model) - return _openrouter_fallback(requested_provider, explicit_api_key, explicit_base_url) diff --git a/hermes_cli/runtime_provider_backends.py b/hermes_cli/runtime_provider_backends.py index 28acb1e22e..aba924a26e 100644 --- a/hermes_cli/runtime_provider_backends.py +++ b/hermes_cli/runtime_provider_backends.py @@ -18,7 +18,6 @@ from utils import base_url_host_matches def _rp(): import hermes_cli.runtime_provider as origin - return origin @@ -49,7 +48,6 @@ def _azure_foundry_api_key(rp, explicit_api_key: str) -> str: return explicit_api_key try: from hermes_cli.config import get_env_value - api_key = get_env_value("AZURE_FOUNDRY_API_KEY") or "" except Exception: api_key = "" @@ -75,7 +73,6 @@ def _resolve_azure_foundry_runtime( rp = _rp() explicit_api_key = str(explicit_api_key or "").strip() explicit_base_url_clean = str(explicit_base_url or "").strip().rstrip("/") - cfg_base_url, cfg_api_mode, cfg_auth_mode, cfg_entra = "", "chat_completions", "api_key", {} if rp._cfg_provider(model_cfg) == "azure-foundry": cfg_base_url = rp._config_base_url_for_provider(model_cfg, "azure-foundry") @@ -83,11 +80,9 @@ def _resolve_azure_foundry_runtime( cfg_auth_mode = str(model_cfg.get("auth_mode") or "api_key").strip().lower() or "api_key" if isinstance(model_cfg.get("entra"), dict): cfg_entra = model_cfg["entra"] - # GPT-5.x / codex / o1-o4 deployments are Responses-API-only on Foundry. effective_model = str(target_model or model_cfg.get("default") or "").strip() cfg_api_mode = rp._azure_inferred_api_mode(effective_model, cfg_api_mode) - env_base_url = rp._getenv("AZURE_FOUNDRY_BASE_URL", "").strip().rstrip("/") base_url = explicit_base_url_clean or cfg_base_url or env_base_url if not base_url: @@ -97,7 +92,6 @@ def _resolve_azure_foundry_runtime( ) if cfg_api_mode == "anthropic_messages": base_url = re.sub(r"/v1/?$", "", base_url) - if cfg_auth_mode == "entra_id": if explicit_api_key: # --api-key on the CLI while config says entra_id: honour the explicit string @@ -145,7 +139,6 @@ def _resolve_openrouter_runtime( # Aliases resolving to "custom" (ollama, vllm, …) follow bare-custom trust + routing rules. if requested_norm and requested_norm != "custom" and rp._resolves_to_custom(requested_norm): requested_norm = "custom" - env_openrouter_base_url = rp._getenv("OPENROUTER_BASE_URL", "").strip() env_custom_base_url = rp._getenv("CUSTOM_BASE_URL", "").strip() use_config_base_url = bool(cfg_base_url.strip()) and not explicit_base_url and ( @@ -159,7 +152,6 @@ def _resolve_openrouter_runtime( or env_openrouter_base_url or OPENROUTER_BASE_URL ).rstrip("/") - is_openrouter_url = base_url_host_matches(base_url, "openrouter.ai") # Explicitly-configured OpenRouter mirrors (OPENROUTER_BASE_URL + provider=openrouter) still # count as OpenRouter for key selection. @@ -179,7 +171,6 @@ def _resolve_openrouter_runtime( api_key = next((str(c or "").strip() for c in candidates if rp.has_usable_secret(c)), "") source = "explicit" if (explicit_api_key or explicit_base_url) else "env/config" cfg_api_mode = rp._parse_api_mode(model_cfg.get("api_mode")) - # Explicit "custom" stays "custom" rather than relabeling to "openrouter". if requested_norm != "custom": return rp._runtime( @@ -224,7 +215,6 @@ def _resolve_bedrock_runtime(requested_provider: str, model_cfg: Dict[str, Any], resolve_aws_auth_env_var, resolve_bedrock_bearer_token, resolve_bedrock_runtime_region, ) from hermes_cli.config import load_config # direct (not the origin delegate), as before - rp = _rp() # Explicitly selected bedrock trusts boto3's credential chain (IMDS, ECS/Lambda roles, SSO) # which the env-var check can't detect. @@ -273,7 +263,6 @@ def _is_external_process_provider(provider: str) -> bool: return False try: from hermes_cli.auth import PROVIDER_REGISTRY - pconfig = PROVIDER_REGISTRY.get(name) if pconfig is not None: return pconfig.auth_type == "external_process" @@ -281,7 +270,6 @@ def _is_external_process_provider(provider: str) -> bool: pass try: from providers import get_provider_profile - profile = get_provider_profile(name) except Exception: return False diff --git a/hermes_cli/runtime_provider_custom.py b/hermes_cli/runtime_provider_custom.py index d5a556fde3..0dd10f34b8 100644 --- a/hermes_cli/runtime_provider_custom.py +++ b/hermes_cli/runtime_provider_custom.py @@ -25,7 +25,6 @@ _LLAMACPP_ALIASES = ("llamacpp", "llama.cpp", "llama-cpp") def _rp(): """Origin module, late-bound so test patches on ``hermes_cli.runtime_provider.*`` apply.""" import hermes_cli.runtime_provider as origin - return origin @@ -129,7 +128,6 @@ def _shadowed_by_builtin(requested_norm: str) -> bool: def _match_new_style_provider(requested_norm: str, providers: Dict[str, Any]) -> Optional[Dict[str, Any]]: """Scan ``providers:`` (new-style, keyed) for ``requested_norm``.""" from hermes_cli.config import is_provider_enabled - rp = _rp() for ep_name, entry in providers.items(): # ``providers..enabled: false`` entries stay in config but are invisible here. @@ -353,7 +351,6 @@ def is_routable_provider(provider: Optional[str]) -> bool: return False try: from hermes_cli.providers import resolve_provider_full - rp = _rp() config = rp.load_config() return resolve_provider_full(name, config.get("providers"), rp.get_compatible_custom_providers(config)) is not None @@ -437,7 +434,6 @@ def _resolve_llamacpp_runtime(requested_provider: str, explicit_api_key: Optiona rp = _rp() try: from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint - endpoint = resolve_llamacpp_endpoint() except Exception: # noqa: BLE001 — resolution is best-effort endpoint = None @@ -494,7 +490,6 @@ def _resolve_direct_alias_runtime( def _opencode_family_for_custom(requested_provider: str, base_url: str) -> Optional[str]: """OpenCode family by provider name, else by opencode.ai host (``/zen/go`` => opencode-go).""" from hermes_cli.models import opencode_provider_family - family = opencode_provider_family(requested_provider) if family is not None: return family @@ -524,14 +519,12 @@ def _resolve_named_custom_runtime( requested_norm = "custom" if requested_norm == "custom" and explicit_base_url: return _resolve_direct_alias_runtime(requested_provider, explicit_api_key, explicit_base_url) - custom_provider = rp._get_named_custom_provider(requested_provider) if not custom_provider: return None base_url = ((explicit_base_url or "").strip() or custom_provider.get("base_url", "")).rstrip("/") if not base_url: return None - pool_result = rp._try_resolve_from_custom_pool( base_url, "custom", custom_provider.get("api_mode"), provider_name=custom_provider.get("provider_key") or custom_provider.get("name"), @@ -540,7 +533,6 @@ def _resolve_named_custom_runtime( # The pool doesn't know the custom_providers fields — propagate them here too. _apply_custom_provider_extras(custom_provider, target_model, pool_result) return pool_result - explicit_key = (explicit_api_key or "").strip() candidates = [ explicit_key, @@ -549,33 +541,28 @@ def _resolve_named_custom_runtime( *rp._host_gated_env_key_candidates(base_url, ollama=False), ] api_key: Any = next((c for c in candidates if rp.has_usable_secret(c)), "") - # ``key_cmd`` credentials are minted per request (short-lived bearers would go stale # mid-session); both wire clients accept a callable api_key (the Entra ID contract). An # explicit --api-key still wins as the one-off recovery escape hatch. key_cmd = _clean(custom_provider.get("key_cmd", "")) if key_cmd and not rp.has_usable_secret(explicit_key): from agent.command_token_source import build_command_token_provider - token_provider = build_command_token_provider( key_cmd, str(custom_provider.get("name", requested_provider) or "custom") ) if token_provider is not None: api_key = token_provider - result = _custom_runtime( rp, base_url, api_key, custom_provider.get("api_mode"), source=f"custom_provider:{custom_provider.get('name', requested_provider)}", requested_provider=requested_provider, ) _apply_custom_provider_extras(custom_provider, target_model, result) - # OpenCode-family custom providers (opencode-go/zen names, or opencode.ai hosts) serve models # on different API surfaces — a static api_mode 503s for /v1/responses-only models. Re-derive # api_mode from the model and normalize /v1 like the built-in paths. family = _opencode_family_for_custom(requested_provider, base_url) if family is not None and not custom_provider.get("api_mode"): from hermes_cli.models import normalize_opencode_base_url, opencode_model_api_mode - effective_model = str( target_model or custom_provider.get("model") or rp._get_model_config().get("default") or "" ).strip() From d211f8562cc6340e0ddc058d1d6791fc090c287c Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:54:02 -0700 Subject: [PATCH 08/19] =?UTF-8?q?refactor(hermes=5Fcli):=20runtime=5Fprovi?= =?UTF-8?q?der=20=E2=80=94=20=5Fcreds=5Ffallback=20+=20=5Factual=5Flocal?= =?UTF-8?q?=5Fkey=20helpers,=20lazy=20rung=20generator=20for=20ladder=20ta?= =?UTF-8?q?il;=20pack=20alias/label=20tables?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/providers.py | 69 +++++++++-------------------- hermes_cli/runtime_provider.py | 80 ++++++++++++++++++---------------- 2 files changed, 62 insertions(+), 87 deletions(-) diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py index 9115c89f4d..06b3a6dc26 100644 --- a/hermes_cli/providers.py +++ b/hermes_cli/providers.py @@ -115,39 +115,25 @@ class ProviderDef: # -- Aliases: human-friendly / legacy names grouped by canonical (models.dev where possible) id; # ``ALIASES`` is the inverted lookup table. --------------------------------------------------- _ALIAS_GROUPS: Dict[str, Tuple[str, ...]] = { - "openrouter": ("openai",), - "zai": ("glm", "z-ai", "z.ai", "zhipu"), - "xai": ("x-ai", "x.ai", "grok"), + "openrouter": ("openai",), "zai": ("glm", "z-ai", "z.ai", "zhipu"), "xai": ("x-ai", "x.ai", "grok"), "xai-oauth": ("grok-oauth", "xai-oauth", "x-ai-oauth", "xai-grok-oauth"), "nvidia": ("nim", "nvidia-nim", "build-nvidia", "nemotron"), "kimi-for-coding": ("kimi", "kimi-coding", "kimi-coding-cn", "moonshot"), - "stepfun": ("step", "stepfun-coding-plan"), - "minimax-cn": ("minimax-china", "minimax_cn"), - "anthropic": ("claude", "claude-code"), - "github-copilot": ("copilot", "github"), - "copilot-acp": ("github-copilot-acp",), - "vercel": ("ai-gateway", "aigateway", "vercel-ai-gateway"), - "opencode": ("opencode-zen", "zen"), - "opencode-go": ("go", "opencode-go-sub"), - "opencode-free": ("free", "opencode_free"), - "kilo": ("kilocode", "kilo-code", "kilo-gateway"), - "deepseek": ("deep-seek",), - "alibaba": ("dashscope", "aliyun", "qwen", "alibaba-cloud"), + "stepfun": ("step", "stepfun-coding-plan"), "minimax-cn": ("minimax-china", "minimax_cn"), + "anthropic": ("claude", "claude-code"), "github-copilot": ("copilot", "github"), + "copilot-acp": ("github-copilot-acp",), "vercel": ("ai-gateway", "aigateway", "vercel-ai-gateway"), + "opencode": ("opencode-zen", "zen"), "opencode-go": ("go", "opencode-go-sub"), + "opencode-free": ("free", "opencode_free"), "kilo": ("kilocode", "kilo-code", "kilo-gateway"), + "deepseek": ("deep-seek",), "alibaba": ("dashscope", "aliyun", "qwen", "alibaba-cloud"), "alibaba-coding-plan": ("alibaba_coding", "alibaba-coding", "alibaba_coding_plan"), - "huggingface": ("hf", "hugging-face", "huggingface-hub"), - "novita": ("novita-ai", "novitaai"), - "xiaomi": ("mimo", "xiaomi-mimo"), - "tencent-tokenhub": ("tencent", "tokenhub", "tencent-cloud", "tencentmaas"), + "huggingface": ("hf", "hugging-face", "huggingface-hub"), "novita": ("novita-ai", "novitaai"), + "xiaomi": ("mimo", "xiaomi-mimo"), "tencent-tokenhub": ("tencent", "tokenhub", "tencent-cloud", "tencentmaas"), "tencent-tokenplan": ("tokenplan", "tencent-lkeap"), - "bedrock": ("aws", "aws-bedrock", "amazon-bedrock", "amazon"), - "arcee": ("arcee-ai", "arceeai"), - "gmi": ("gmi-cloud", "gmicloud"), - "fireworks": ("fireworks-ai", "fw"), - "upstage": ("solar",), + "bedrock": ("aws", "aws-bedrock", "amazon-bedrock", "amazon"), "arcee": ("arcee-ai", "arceeai"), + "gmi": ("gmi-cloud", "gmicloud"), "fireworks": ("fireworks-ai", "fw"), "upstage": ("solar",), "actual": ("actual-computer", "actualcomputer", "aci"), "nebius-token-factory": ("nebius", "nebius-tokenfactory", "nebius-tf", "token-factory", "tokenfactory"), - "lmstudio": ("lmstudio", "lm-studio", "lm_studio"), - "custom": ("ollama",), + "lmstudio": ("lmstudio", "lm-studio", "lm_studio"), "custom": ("ollama",), "local": ("vllm", "llamacpp", "llama.cpp", "llama-cpp"), } ALIASES: Dict[str, str] = {alias: canon for canon, aliases in _ALIAS_GROUPS.items() for alias in aliases} @@ -156,35 +142,20 @@ ALIASES: Dict[str, str] = {alias: canon for canon, aliases in _ALIAS_GROUPS.item # -- Display labels for providers not in the models.dev catalog --------------- _LABEL_OVERRIDES: Dict[str, str] = { - "moa": "Mixture of Agents", - "nous": "Nous Portal", - "openai-codex": "ChatGPT or Codex Subscription", - "copilot-acp": "GitHub Copilot ACP", - "stepfun": "StepFun Step Plan", - "xiaomi": "Xiaomi MiMo", - "gmi": "GMI Cloud", - "upstage": "Upstage Solar", - "actual": "Actual Computer", - "tencent-tokenhub": "Tencent TokenHub", - "nebius-token-factory": "Nebius Token Factory", - "tencent-tokenplan": "Tencent TokenPlan", - "lmstudio": "LM Studio", - "local": "Local endpoint", - "bedrock": "AWS Bedrock", - "vertex": "Google Vertex AI", - "ollama-cloud": "Ollama Cloud", - "xai-oauth": "xAI Grok OAuth (SuperGrok / Premium+)", - "opencode-free": "OpenCode Free", + "moa": "Mixture of Agents", "nous": "Nous Portal", "openai-codex": "ChatGPT or Codex Subscription", + "copilot-acp": "GitHub Copilot ACP", "stepfun": "StepFun Step Plan", "xiaomi": "Xiaomi MiMo", "gmi": "GMI Cloud", + "upstage": "Upstage Solar", "actual": "Actual Computer", "tencent-tokenhub": "Tencent TokenHub", + "nebius-token-factory": "Nebius Token Factory", "tencent-tokenplan": "Tencent TokenPlan", "lmstudio": "LM Studio", + "local": "Local endpoint", "bedrock": "AWS Bedrock", "vertex": "Google Vertex AI", "ollama-cloud": "Ollama Cloud", + "xai-oauth": "xAI Grok OAuth (SuperGrok / Premium+)", "opencode-free": "OpenCode Free", } # -- Transport → API mode mapping --------------------------------------------- TRANSPORT_TO_API_MODE: Dict[str, str] = { - "openai_chat": "chat_completions", - "anthropic_messages": "anthropic_messages", - "codex_responses": "codex_responses", - "bedrock_converse": "bedrock_converse", + "openai_chat": "chat_completions", "anthropic_messages": "anthropic_messages", + "codex_responses": "codex_responses", "bedrock_converse": "bedrock_converse", } diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index 47c5cd23ee..3784c69d68 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -594,14 +594,20 @@ def _explicit_anthropic(requested_provider, model_cfg, api_key, base_url, target return _runtime("anthropic", "anthropic_messages", base_url, api_key, source="explicit", requested_provider=requested_provider) +def _creds_fallback(api_key, explicit_base_url, base_url, expiry, expiry_key, resolve): + """When no explicit key was given, take api_key / expiry / base_url from stored credentials + (an explicit --base-url still wins over the stored one).""" + if api_key: + return api_key, base_url, expiry + creds = resolve() + return creds.get("api_key", ""), explicit_base_url or creds.get("base_url", "").rstrip("/") or base_url, creds.get(expiry_key) + + def _explicit_codex(requested_provider, model_cfg, api_key, explicit_base_url, target_model): - base_url = explicit_base_url or DEFAULT_CODEX_BASE_URL - last_refresh = None - if not api_key: - creds = resolve_codex_runtime_credentials() - api_key = creds.get("api_key", "") - last_refresh = creds.get("last_refresh") - base_url = explicit_base_url or creds.get("base_url", "").rstrip("/") or base_url + api_key, base_url, last_refresh = _creds_fallback( + api_key, explicit_base_url, explicit_base_url or DEFAULT_CODEX_BASE_URL, None, "last_refresh", + resolve_codex_runtime_credentials, + ) return _runtime( "openai-codex", "codex_responses", base_url, api_key, source="explicit", last_refresh=last_refresh, requested_provider=requested_provider, @@ -620,12 +626,10 @@ def _explicit_nous(requested_provider, model_cfg, api_key, explicit_base_url, ta api_key = api_key or ( str(state.get("agent_key") or "").strip() if _agent_key_is_usable(state, _nous_min_key_ttl()) else "" ) - expires_at = state.get("agent_key_expires_at") or state.get("expires_at") - if not api_key: - creds = _resolve_nous_creds() - api_key = creds.get("api_key", "") - expires_at = creds.get("expires_at") - base_url = explicit_base_url or creds.get("base_url", "").rstrip("/") or base_url + api_key, base_url, expires_at = _creds_fallback( + api_key, explicit_base_url, base_url, state.get("agent_key_expires_at") or state.get("expires_at"), "expires_at", + _resolve_nous_creds, + ) return _runtime( "nous", nous_api_mode(_effective_model(model_cfg, target_model)), base_url, api_key, source="explicit", expires_at=expires_at, requested_provider=requested_provider, @@ -638,6 +642,13 @@ def _explicit_azure_foundry(requested_provider, model_cfg, api_key, base_url, ta ) +def _actual_local_key(provider: str, api_key: str, base_url: str) -> str: + """Actual Computer's loopback daemon speaks a no-auth local API — substitute the placeholder key.""" + if provider == "actual" and not api_key and is_actual_local_base_url(base_url): + return ACTUAL_LOCAL_NOAUTH_PLACEHOLDER + return api_key + + def _explicit_api_key_provider(provider, pconfig, requested_provider, model_cfg, api_key, base_url, target_model): if not base_url: if provider in {"kimi-coding", "kimi-coding-cn"}: @@ -657,8 +668,7 @@ def _explicit_api_key_provider(provider, pconfig, requested_provider, model_cfg, api_mode = _api_key_provider_api_mode( provider, model_cfg, api_key, base_url, target_model or model_cfg.get("default", ""), opencode_by_model=False ) - if provider == "actual" and not api_key and is_actual_local_base_url(base_url): - api_key = ACTUAL_LOCAL_NOAUTH_PLACEHOLDER + api_key = _actual_local_key(provider, api_key, base_url) return _runtime(provider, api_mode, base_url.rstrip("/"), api_key, source="explicit", requested_provider=requested_provider) @@ -816,9 +826,7 @@ def _api_key_provider_runtime(provider, pconfig, requested_provider, model_cfg, provider, model_cfg, creds.get("api_key", ""), base_url, target_model or model_cfg.get("default", ""), opencode_by_model=True ) base_url = _finalize_base_url(provider, api_mode, base_url) - api_key = creds.get("api_key", "") - if provider == "actual" and not api_key and is_actual_local_base_url(base_url): - api_key = ACTUAL_LOCAL_NOAUTH_PLACEHOLDER + api_key = _actual_local_key(provider, creds.get("api_key", ""), base_url) return _runtime(provider, api_mode, base_url, api_key, source=creds.get("source", "env"), requested_provider=requested_provider) @@ -963,37 +971,33 @@ def resolve_runtime_provider( if runtime: return runtime provider = resolve_provider(requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url) + return next(r for r in _provider_rungs(provider, requested_provider, explicit_api_key, explicit_base_url, target_model) if r) + + +def _provider_rungs(provider, requested_provider, explicit_api_key, explicit_base_url, target_model): + """Rungs 5-8 of the ladder, yielded lazily so each is evaluated only when the previous one + returned nothing; the last rung (OpenRouter / bare-custom fallback) always yields a runtime.""" model_cfg = _get_model_config() - runtime = _opencode_free_runtime(provider, requested_provider, model_cfg, target_model) - if runtime is not None: - return runtime - runtime = _resolve_explicit_runtime( + yield _opencode_free_runtime(provider, requested_provider, model_cfg, target_model) + yield _resolve_explicit_runtime( provider=provider, requested_provider=requested_provider, model_cfg=model_cfg, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, target_model=target_model, ) - if runtime: - return runtime - runtime = _resolve_from_pool(provider, requested_provider, model_cfg, explicit_api_key, explicit_base_url, target_model) - if runtime: - return runtime + yield _resolve_from_pool(provider, requested_provider, model_cfg, explicit_api_key, explicit_base_url, target_model) if provider in _OAUTH_RUNTIME_PROVIDERS: - runtime = _resolve_oauth_runtime(provider, requested_provider, model_cfg, target_model) - if runtime: - return runtime + yield _resolve_oauth_runtime(provider, requested_provider, model_cfg, target_model) if provider == "minimax-oauth": - runtime = _minimax_oauth_runtime(provider, requested_provider) - if runtime: - return runtime + yield _minimax_oauth_runtime(provider, requested_provider) if _is_external_process_provider(provider): - return _resolve_external_process_runtime(provider, requested_provider) + yield _resolve_external_process_runtime(provider, requested_provider) if provider == "anthropic": - return _anthropic_env_runtime(requested_provider, model_cfg) + yield _anthropic_env_runtime(requested_provider, model_cfg) if provider == "bedrock": - return _resolve_bedrock_runtime(requested_provider, model_cfg, target_model) + yield _resolve_bedrock_runtime(requested_provider, model_cfg, target_model) pconfig = PROVIDER_REGISTRY.get(provider) if pconfig and pconfig.auth_type == "api_key": - return _api_key_provider_runtime(provider, pconfig, requested_provider, model_cfg, target_model) - return _openrouter_fallback(requested_provider, explicit_api_key, explicit_base_url) + yield _api_key_provider_runtime(provider, pconfig, requested_provider, model_cfg, target_model) + yield _openrouter_fallback(requested_provider, explicit_api_key, explicit_base_url) def format_runtime_provider_error(error: Exception) -> str: From e789d79cfe119d3fa7a7cd525269d55ed2f75bbe Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 22:14:12 -0700 Subject: [PATCH 09/19] =?UTF-8?q?refactor(hermes=5Fcli):=20provider=20clus?= =?UTF-8?q?ter=20=E2=80=94=20shared=20=5Foverlay=5Fpdef,=20pool-select=20g?= =?UTF-8?q?uard=20collapse,=20docstring=20compaction=20(WHY=20kept)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/providers.py | 71 +++++++++++------------- hermes_cli/runtime_provider.py | 61 +++++++++----------- hermes_cli/runtime_provider_backends.py | 22 +++----- hermes_cli/runtime_provider_custom.py | 74 ++++++++++--------------- 4 files changed, 96 insertions(+), 132 deletions(-) diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py index 06b3a6dc26..eccefc684b 100644 --- a/hermes_cli/providers.py +++ b/hermes_cli/providers.py @@ -177,12 +177,17 @@ def _models_dev_info(canonical: str, allow_network: bool = True): return None -def get_provider(name: str, *, allow_network: bool = True) -> Optional[ProviderDef]: - """Look up a built-in provider by id or alias. +def _overlay_pdef(canonical, ov: HermesOverlay, name, env_vars, base_url, doc, source) -> ProviderDef: + return ProviderDef( + id=canonical, name=name, transport=ov.transport, api_key_env_vars=env_vars, base_url=base_url, + base_url_env_var=ov.base_url_env_var, is_aggregator=ov.is_aggregator, auth_type=ov.auth_type, doc=doc, + source=source, + ) - Order: models.dev catalog merged with the Hermes overlay; Hermes-only overlay (nous, - openai-codex, …); plugin provider profiles with a concrete endpoint. - """ + +def get_provider(name: str, *, allow_network: bool = True) -> Optional[ProviderDef]: + """Look up a built-in provider by id or alias: models.dev catalog merged with the Hermes overlay; + Hermes-only overlay (nous, openai-codex, …); plugin provider profiles with a concrete endpoint.""" canonical = normalize_provider(name) mdev_info = _models_dev_info(canonical, allow_network) overlay = HERMES_OVERLAYS.get(canonical) @@ -192,17 +197,14 @@ def get_provider(name: str, *, allow_network: bool = True) -> Optional[ProviderD for ev in ov.extra_env_vars: if ev not in env_vars: env_vars.append(ev) - return ProviderDef( - id=canonical, name=mdev_info.name, transport=ov.transport, api_key_env_vars=tuple(env_vars), - base_url=ov.base_url_override or mdev_info.api, base_url_env_var=ov.base_url_env_var, - is_aggregator=ov.is_aggregator, auth_type=ov.auth_type, doc=mdev_info.doc, source="models.dev", + return _overlay_pdef( + canonical, ov, mdev_info.name, tuple(env_vars), ov.base_url_override or mdev_info.api, mdev_info.doc, + "models.dev", ) if overlay is not None: - return ProviderDef( - id=canonical, name=_LABEL_OVERRIDES.get(canonical, canonical), transport=overlay.transport, - api_key_env_vars=overlay.extra_env_vars, base_url=overlay.base_url_override, - base_url_env_var=overlay.base_url_env_var, is_aggregator=overlay.is_aggregator, - auth_type=overlay.auth_type, source="hermes", + return _overlay_pdef( + canonical, overlay, _LABEL_OVERRIDES.get(canonical, canonical), overlay.extra_env_vars, + overlay.base_url_override, "", "hermes", ) # Plugin-registered profiles (plugins/model-providers//) absent from models.dev and # HERMES_OVERLAYS would otherwise be "Unknown provider" in /model, --provider and model-switch @@ -264,12 +266,10 @@ def is_routing_aggregator(provider: str) -> bool: def is_official_openai_host(base_url: str) -> bool: - """True when *base_url* points at OpenAI's official API host family. - - Hostname-parsed matching only — never substring — so lookalike hosts - (``api.openai.com.attacker.test``) and path-segment spoofs (``proxy.test/api.openai.com/v1``) - are rejected. A genuine ``*.api.openai.com`` subdomain requires control of openai.com DNS. - """ + """True when *base_url* points at OpenAI's official API host family. Hostname-parsed matching + only — never substring — so lookalike hosts (``api.openai.com.attacker.test``) and path-segment + spoofs (``proxy.test/api.openai.com/v1``) are rejected; a genuine ``*.api.openai.com`` + subdomain requires control of openai.com DNS.""" return base_url_host_matches(base_url, "api.openai.com") @@ -281,14 +281,12 @@ _RESPONSES_NATIVE_HOSTS: frozenset[str] = frozenset({"api.meta.ai", "api.router. def host_mandated_api_mode(base_url: str = "") -> Optional[str]: - """Return the wire protocol a specific endpoint *requires*, or None. - - Some hosts accept exactly one API mode (api.openai.com 400s chat/completions for reasoning - models with tools). These are *mandatory*: a session carrying a stale api_mode (a /model switch - that kept the previous provider's ``chat_completions``) must be overridden, not merely filled - in when empty. Exact-hostname matching only — never substring — so lookalike hosts and - path-segment spoofs are not treated as the real endpoint. - """ + """Return the wire protocol a specific endpoint *requires*, or None. Some hosts accept exactly + one API mode (api.openai.com 400s chat/completions for reasoning models with tools); these are + *mandatory*: a session carrying a stale api_mode (a /model switch that kept the previous + provider's ``chat_completions``) must be overridden, not merely filled in when empty. + Exact-hostname matching only — never substring — so lookalike hosts and path-segment spoofs are + not treated as the real endpoint.""" if not base_url: return None url_lower = base_url.rstrip("/").lower() @@ -383,11 +381,9 @@ def custom_provider_aliases(display_name: str, provider_key: str = "") -> frozen def resolve_custom_provider(name: str, custom_providers: Optional[List[Dict[str, Any]]]) -> Optional[ProviderDef]: - """Resolve a provider from the user's config.yaml ``custom_providers`` list. - - A stored bare ``"custom"`` (corrupt state from a prior model-switch bug) falls back to the first - valid entry so existing configs self-heal. - """ + """Resolve a provider from the user's config.yaml ``custom_providers`` list. A stored bare + ``"custom"`` (corrupt state from a prior model-switch bug) falls back to the first valid entry + so existing configs self-heal.""" if not custom_providers or not isinstance(custom_providers, list): return None requested = (name or "").strip().lower() @@ -459,12 +455,9 @@ def resolve_provider_full( ) -> Optional[ProviderDef]: """Full resolution chain: user ``providers.`` -> lossy-alias registry id -> built-in (models.dev + overlays) -> user providers (canonical, then raw) -> ``custom_providers`` -> - managed llamacpp -> models.dev directly. - - User-defined ``providers.`` is tried FIRST on the raw (pre-alias) name: a configured - ``providers.openai`` pointing at api.openai.com must not be hijacked by the legacy - "openai" -> "openrouter" alias. - """ + managed llamacpp -> models.dev directly. User-defined ``providers.`` is tried FIRST on + the raw (pre-alias) name: a configured ``providers.openai`` pointing at api.openai.com must not + be hijacked by the legacy "openai" -> "openrouter" alias.""" canonical = normalize_provider(name) raw = name.strip().lower() if user_providers: diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index 3784c69d68..a1e3fa90b5 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -1,10 +1,9 @@ -"""Shared runtime provider resolution for CLI, gateway, cron, and helpers. - -This module owns the resolution ORDER (:func:`resolve_runtime_provider`), the api_mode / base_url -helpers and the pool / OAuth / explicit paths. Custom-provider lookup lives in -:mod:`hermes_cli.runtime_provider_custom`; Azure Foundry, OpenRouter/bare-custom, Bedrock and -external-process builders in :mod:`hermes_cli.runtime_provider_backends`. Both are re-exported -here so ``hermes_cli.runtime_provider.`` imports and test patches keep working.""" +"""Shared runtime provider resolution for CLI, gateway, cron, and helpers: the resolution ORDER +(:func:`resolve_runtime_provider`), api_mode / base_url helpers and the pool / OAuth / explicit paths. +Custom-provider lookup lives in :mod:`hermes_cli.runtime_provider_custom`; Azure Foundry, +OpenRouter/bare-custom, Bedrock and external-process builders in +:mod:`hermes_cli.runtime_provider_backends` — both re-exported here so +``hermes_cli.runtime_provider.`` imports and test patches keep working.""" from __future__ import annotations @@ -72,12 +71,11 @@ def _resolves_to_custom(name: str) -> bool: def _config_base_url_trustworthy_for_bare_custom(cfg_base_url: str, cfg_provider: str) -> bool: - """Whether ``model.base_url`` may back bare ``custom`` runtime resolution. - - The model picker can select Custom while ``model.provider`` still reflects a previous provider. - Non-loopback URLs are rejected unless the YAML provider is already ``custom`` or a local-server - alias (ollama/vllm/llamacpp — else a legit LAN ollama endpoint silently falls through to - OpenRouter), so a stale OpenRouter/Z.ai base_url cannot hijack local sessions.""" + """Whether ``model.base_url`` may back bare ``custom`` runtime resolution. The picker can select + Custom while ``model.provider`` still names a previous provider, so non-loopback URLs are rejected + unless the YAML provider is already ``custom`` or a local-server alias (ollama/vllm/llamacpp — + else a legit LAN ollama endpoint falls through to OpenRouter): a stale OpenRouter/Z.ai base_url + cannot hijack local sessions.""" cfg_provider_norm = (cfg_provider or "").strip().lower() bu = (cfg_base_url or "").strip() if not bu: @@ -105,11 +103,10 @@ _VALID_API_MODES = {"chat_completions", "codex_responses", "anthropic_messages", def _detect_api_mode_for_url(base_url: str) -> Optional[str]: - """Auto-detect api_mode from the resolved base URL, or None. - - Exact-hostname matches reject lookalike subdomains (api.anthropic.com.attacker.test) and - path-segment spoofing (proxy.test/api.anthropic.com/v1). Official OpenAI hosts (incl. the - data-residency us./eu. regional hosts) need Responses for GPT-5.x tool calls with reasoning.""" + """Auto-detect api_mode from the resolved base URL, or None. Exact-hostname matches reject + lookalike subdomains (api.anthropic.com.attacker.test) and path-segment spoofing + (proxy.test/api.anthropic.com/v1). Official OpenAI hosts (incl. us./eu. data-residency hosts) + need Responses for GPT-5.x tool calls with reasoning.""" normalized = (base_url or "").strip().lower().rstrip("/") hostname = base_url_hostname(base_url) mandated = _HOST_MANDATED_API_MODES.get(hostname) @@ -214,7 +211,6 @@ def _configured_or_fallback_api_mode( provider: str, model_cfg: Dict[str, Any], base_url: str, effective_model: Any, *, opencode_by_model: bool ) -> str: """Persisted ``model.api_mode`` when it belongs to this provider, else URL/transport fallback. - OpenCode Zen/Go serve both anthropic_messages and chat_completions models, so (when ``opencode_by_model``) their mode is always re-derived from the effective model.""" if opencode_by_model and _models.opencode_provider_family(provider) is not None: @@ -320,10 +316,9 @@ def _host_derived_api_key(base_url: str) -> str: def _host_gated_env_key_candidates(base_url: str, *, ollama: bool) -> list: """Env API keys gated on their authoritative hosts, then the host-derived ``_API_KEY``. - Sending OPENAI/OPENROUTER/OLLAMA keys to an unrelated endpoint leaks credentials - (GHSA-76xc-57q6-vm5m); match on HOST, not substring. ``_host_derived_api_key`` skips OLLAMA, - so callers that want it opt in via ``ollama``.""" + (GHSA-76xc-57q6-vm5m); match on HOST, not substring. ``_host_derived_api_key`` skips OLLAMA, so + callers that want it opt in via ``ollama``.""" is_openai = base_url_host_matches(base_url, "openai.com") or base_url_host_matches(base_url, "openai.azure.com") candidates = [] if ollama: @@ -570,14 +565,12 @@ def _resolve_from_pool( if not (pool and pool.has_credentials()): return None entry = pool.select() - pool_api_key = _pool_entry_api_key(entry) if entry is not None else "" - if provider == "nous" and entry is not None: + if entry is None: + return None + pool_api_key = _pool_entry_api_key(entry) + if provider == "nous": entry, pool_api_key = _refresh_nous_pool_entry(pool, entry, pool_api_key) - if ( - entry is not None - and pool_api_key - and credential_pool_matches_provider(pool, provider, base_url=_pool_entry_base_url(entry)) - ): + if pool_api_key and credential_pool_matches_provider(pool, provider, base_url=_pool_entry_base_url(entry)): return _resolve_runtime_from_pool_entry( provider=provider, entry=entry, requested_provider=requested_provider, model_cfg=model_cfg, pool=pool, target_model=target_model, @@ -939,9 +932,8 @@ def resolve_runtime_provider( *, requested: Optional[str] = None, explicit_api_key: Optional[str] = None, explicit_base_url: Optional[str] = None, target_model: Optional[str] = None, ) -> Dict[str, Any]: - """Resolve runtime provider credentials for agent execution. - - Ladder (order is behavior — each rung returns or raises, else falls to the next): + """Resolve runtime provider credentials for agent execution. Ladder (order is behavior — each + rung returns or raises, else falls to the next): 1. disabled-provider guard (``providers..enabled: false``) 2. requested-name shortcuts: moa, anthropic@azure, azure-foundry, vertex 3. named custom provider / llamacpp alias / bare-custom direct alias @@ -951,9 +943,8 @@ def resolve_runtime_provider( 7. OAuth specs (nous/codex/xai/qwen; "auto" swallows AuthError and logs) → minimax-oauth → external-process → anthropic env → bedrock → registry api_key providers 8. OpenRouter / bare-custom fallback - - target_model: overrides model_cfg["default"] when computing provider-specific api_mode - (e.g. OpenCode Zen/Go where different models route through different API surfaces).""" + target_model overrides model_cfg["default"] when computing provider-specific api_mode (e.g. + OpenCode Zen/Go where different models route through different API surfaces).""" requested_provider = resolve_requested_provider(requested) _raise_if_provider_disabled(requested_provider) runtime = _resolve_requested_shortcuts(requested_provider, explicit_api_key, explicit_base_url, target_model) diff --git a/hermes_cli/runtime_provider_backends.py b/hermes_cli/runtime_provider_backends.py index aba924a26e..bbe8c30acd 100644 --- a/hermes_cli/runtime_provider_backends.py +++ b/hermes_cli/runtime_provider_backends.py @@ -1,10 +1,8 @@ """Provider-specific runtime builders for :mod:`hermes_cli.runtime_provider`: Azure Foundry, the -OpenRouter / bare-custom fallback resolver, Bedrock, and external-process providers. - -Origin-internal collaborators are resolved on the origin module at call time via :func:`_rp` so -test patches on ``hermes_cli.runtime_provider.*`` (``_get_model_config``, ``load_config``, -``has_usable_secret``, ``_try_resolve_from_custom_pool``, …) still apply. -""" +OpenRouter / bare-custom fallback resolver, Bedrock, and external-process providers. Origin-internal +collaborators are resolved on the origin module at call time via :func:`_rp` so test patches on +``hermes_cli.runtime_provider.*`` (``_get_model_config``, ``load_config``, ``has_usable_secret``, +``_try_resolve_from_custom_pool``, …) still apply.""" from __future__ import annotations @@ -120,13 +118,11 @@ def _resolve_azure_foundry_runtime( def _resolve_openrouter_runtime( *, requested_provider: str, explicit_api_key: Optional[str] = None, explicit_base_url: Optional[str] = None ) -> Dict[str, Any]: - """Terminal resolver: OpenRouter, or a bare/aliased ``custom`` endpoint. - - base_url precedence: explicit > CUSTOM_BASE_URL > trusted ``model.base_url`` > OPENROUTER_BASE_URL - > default. OPENAI_BASE_URL is deliberately NOT consulted — config.yaml is the single source of - truth for endpoint URLs. OpenRouter contexts prefer OPENROUTER_API_KEY; custom endpoints never - receive the OpenRouter key and only get env keys gated on their authoritative hosts. - """ + """Terminal resolver: OpenRouter, or a bare/aliased ``custom`` endpoint. base_url precedence: + explicit > CUSTOM_BASE_URL > trusted ``model.base_url`` > OPENROUTER_BASE_URL > default. + OPENAI_BASE_URL is deliberately NOT consulted — config.yaml is the single source of truth for + endpoint URLs. OpenRouter contexts prefer OPENROUTER_API_KEY; custom endpoints never receive the + OpenRouter key and only get env keys gated on their authoritative hosts.""" rp = _rp() model_cfg = rp._get_model_config() cfg_base_url = model_cfg.get("base_url") if isinstance(model_cfg.get("base_url"), str) else "" diff --git a/hermes_cli/runtime_provider_custom.py b/hermes_cli/runtime_provider_custom.py index 0dd10f34b8..f1e04c1267 100644 --- a/hermes_cli/runtime_provider_custom.py +++ b/hermes_cli/runtime_provider_custom.py @@ -1,12 +1,9 @@ -"""Custom-provider resolution: ``providers:`` / ``custom_providers:`` lookup, identity -recovery, custom credential pools, and the named-custom runtime builder. - -Extracted from :mod:`hermes_cli.runtime_provider`; every name here is re-exported there. -Origin-internal collaborators (``load_config``, ``_get_model_config``, ``load_pool``, -``has_usable_secret``, ``custom_provider_pool_key_candidates``, …) are looked up on the origin -module AT CALL TIME via :func:`_rp` so ``monkeypatch.setattr(runtime_provider, name, …)`` keeps -working for moved bodies. -""" +"""Custom-provider resolution: ``providers:`` / ``custom_providers:`` lookup, identity recovery, +custom credential pools, and the named-custom runtime builder. Extracted from +:mod:`hermes_cli.runtime_provider` (every name re-exported there); origin-internal collaborators +(``load_config``, ``_get_model_config``, ``load_pool``, ``has_usable_secret``, …) are looked up on +the origin module AT CALL TIME via :func:`_rp` so ``monkeypatch.setattr(runtime_provider, name, …)`` +keeps working for moved bodies.""" from __future__ import annotations @@ -98,23 +95,19 @@ def _lift_common_custom_fields( if api_mode: result["api_mode"] = api_mode _lift_max_output_tokens(entry, result) - capabilities = _filter_capabilities(entry.get("capabilities")) - if capabilities: - result["capabilities"] = capabilities + _lift_model_capabilities(entry, None, result) # ── config lookup ────────────────────────────────────────────────────────────────────────── def _shadowed_by_builtin(requested_norm: str) -> bool: - """Raw names map to custom providers only when they are not canonical built-ins. - - Explicit ``custom:`` keys always target the saved entry, and bare ``custom`` is exempt: a - user may literally name a ``providers:`` entry "custom" (returning None before the config scan - made such cron jobs fail with ``auth_unavailable``). Defer to the built-in only when the raw - name IS the canonical provider (``nous``); an entry matching merely an alias (``kimi`` → - ``kimi-coding``) is the user's target. - """ + """Raw names map to custom providers only when they are not canonical built-ins. Explicit + ``custom:`` keys always target the saved entry, and bare ``custom`` is exempt: a user may + literally name a ``providers:`` entry "custom" (returning None before the config scan made such + cron jobs fail with ``auth_unavailable``). Defer to the built-in only when the raw name IS the + canonical provider (``nous``); an entry matching merely an alias (``kimi`` → ``kimi-coding``) + is the user's target.""" if requested_norm == "custom" or requested_norm.startswith("custom:"): return False rp = _rp() @@ -288,15 +281,13 @@ def find_custom_provider_identity_by_model(model: str) -> Optional[str]: def canonical_custom_identity( *, base_url: Optional[str] = None, config_provider: Optional[str] = None, model: Optional[str] = None ) -> Optional[str]: - """Recover a routable ``custom:`` identity for a bare custom provider. - - Every path that persists or restores a session's provider override must run the resolved - provider through this so a bare ``"custom"`` is upgraded back to its durable ``custom:`` - menu key. Sources in priority order: (1) ``base_url`` reverse lookup — the one fact that always - survives the round-trip when a URL was recorded; (2) ``model`` reverse lookup - (``model``/``default_model``/``models`` catalog); (3) the configured provider (arg, then - ``model.provider``, then ``HERMES_INFERENCE_PROVIDER``) when it names a real entry. - """ + """Recover a routable ``custom:`` identity for a bare custom provider. Every path that + persists or restores a session's provider override must run the resolved provider through this + so a bare ``"custom"`` is upgraded back to its durable menu key. Sources in priority order: + (1) ``base_url`` reverse lookup — the one fact that always survives the round-trip when a URL + was recorded; (2) ``model`` reverse lookup (``model``/``default_model``/``models`` catalog); + (3) the configured provider (arg, ``model.provider``, ``HERMES_INFERENCE_PROVIDER``) when it + names a real entry.""" rp = _rp() if base_url: identity = find_custom_provider_identity(base_url) @@ -336,14 +327,11 @@ def canonical_custom_identity( def is_routable_provider(provider: Optional[str]) -> bool: - """Whether a provider name currently resolves to a routable route. - - Empty/None/``auto`` is vacuously routable (agent build falls back to the configured default). - Bare ``custom`` is the resolved billing class shared by every named entry — not a routable - identity; restore paths must heal it (:func:`canonical_custom_identity`) or fall back. Anything - else is routable iff the full chain (built-in -> ``providers:`` -> ``custom_providers:`` -> - models.dev) resolves it. - """ + """Whether a provider name currently resolves to a routable route. Empty/None/``auto`` is + vacuously routable (agent build falls back to the configured default). Bare ``custom`` is the + resolved billing class shared by every named entry — not a routable identity; restore paths + must heal it (:func:`canonical_custom_identity`) or fall back. Anything else is routable iff the + full chain (built-in -> ``providers:`` -> ``custom_providers:`` -> models.dev) resolves it.""" name = str(provider or "").strip() if not name or name.lower() == "auto": return True @@ -426,11 +414,9 @@ def _apply_custom_provider_extras( def _resolve_llamacpp_runtime(requested_provider: str, explicit_api_key: Optional[str]) -> Dict[str, Any]: """Managed llama.cpp runtime: the supervised (or detected external) server, or a typed error. - No server => say so and stop; falling through to the generic custom path would surface "local server is off" as OpenRouter's baffling "401 Invalid API key". The switch's state picks the - message (server off → point at the switch; else the setup pane). - """ + message (server off → point at the switch; else the setup pane).""" rp = _rp() try: from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint @@ -506,11 +492,9 @@ def _resolve_named_custom_runtime( target_model: Optional[str] = None, ) -> Optional[Dict[str, Any]]: """Runtime for a llamacpp alias, a bare-custom direct alias, or a configured custom entry. - - Aliases resolving to "custom" (ollama, vllm, llamacpp, …) are treated like bare ``custom``. - A llamacpp alias with no explicit base_url resolves to the managed server first; an explicit - base_url always wins. - """ + Aliases resolving to "custom" (ollama, vllm, llamacpp, …) are treated like bare ``custom``. A + llamacpp alias with no explicit base_url resolves to the managed server first; an explicit + base_url always wins.""" rp = _rp() requested_norm = (requested_provider or "").strip().lower() if requested_norm in _LLAMACPP_ALIASES and not explicit_base_url: From 951ee86ae647207713bb7ba35ecfeebac43442f6 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 22:28:12 -0700 Subject: [PATCH 10/19] =?UTF-8?q?refactor(hermes=5Fcli):=20backends/custom?= =?UTF-8?q?=20=E2=80=94=20flatten=20entra=20branch,=20registry=20via=20ori?= =?UTF-8?q?gin,=20canonical=5Fcustom=5Fidentity=20tail=20via=20custom=5Fpr?= =?UTF-8?q?ovider=5Fslug?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/runtime_provider_backends.py | 21 +++++++++---------- hermes_cli/runtime_provider_custom.py | 28 ++++++++++++------------- 2 files changed, 23 insertions(+), 26 deletions(-) diff --git a/hermes_cli/runtime_provider_backends.py b/hermes_cli/runtime_provider_backends.py index bbe8c30acd..7115961c71 100644 --- a/hermes_cli/runtime_provider_backends.py +++ b/hermes_cli/runtime_provider_backends.py @@ -91,19 +91,18 @@ def _resolve_azure_foundry_runtime( if cfg_api_mode == "anthropic_messages": base_url = re.sub(r"/v1/?$", "", base_url) if cfg_auth_mode == "entra_id": + # --api-key on the CLI while config says entra_id: honour the explicit string (escape hatch + # for one-off testing). if explicit_api_key: - # --api-key on the CLI while config says entra_id: honour the explicit string - # (escape hatch for one-off testing). - api_key, source, auth_mode = explicit_api_key, "explicit", "api_key" + api_key, source, auth_mode, entra = explicit_api_key, "explicit", "api_key", {} else: - api_key, source, auth_mode = _azure_entra_credentials(cfg_entra), "entra_id", "entra_id" - clean_entra = {} - configured_scope = str(cfg_entra.get("scope") or "").strip() - if auth_mode == "entra_id" and configured_scope: - clean_entra["scope"] = configured_scope + scope = str(cfg_entra.get("scope") or "").strip() + api_key, source, auth_mode, entra = _azure_entra_credentials(cfg_entra), "entra_id", "entra_id", ( + {"scope": scope} if scope else {} + ) return rp._runtime( "azure-foundry", cfg_api_mode, base_url, api_key, - auth_mode=auth_mode, entra=clean_entra, source=source, requested_provider=requested_provider, + auth_mode=auth_mode, entra=entra, source=source, requested_provider=requested_provider, ) return rp._runtime( "azure-foundry", cfg_api_mode, base_url, _azure_foundry_api_key(rp, explicit_api_key), @@ -258,14 +257,14 @@ def _is_external_process_provider(provider: str) -> bool: if not name: return False try: - from hermes_cli.auth import PROVIDER_REGISTRY - pconfig = PROVIDER_REGISTRY.get(name) + pconfig = _rp().PROVIDER_REGISTRY.get(name) if pconfig is not None: return pconfig.auth_type == "external_process" except Exception: pass try: from providers import get_provider_profile + profile = get_provider_profile(name) except Exception: return False diff --git a/hermes_cli/runtime_provider_custom.py b/hermes_cli/runtime_provider_custom.py index f1e04c1267..442ab2eaba 100644 --- a/hermes_cli/runtime_provider_custom.py +++ b/hermes_cli/runtime_provider_custom.py @@ -136,10 +136,8 @@ def _match_new_style_provider(requested_norm: str, providers: Dict[str, Any]) -> if not base_url: continue result: Dict[str, Any] = { - "name": entry.get("name", ep_name), - "base_url": base_url.strip(), - "api_key": api_key or _clean(entry.get("api_key", "")), - "model": entry.get("default_model", ""), + "name": entry.get("name", ep_name), "base_url": base_url.strip(), + "api_key": api_key or _clean(entry.get("api_key", "")), "model": entry.get("default_model", ""), } # Command that PRINTS a short-lived credential; wrapped in a per-request token provider. key_cmd = _clean(entry.get("key_cmd", "")) @@ -310,20 +308,20 @@ def canonical_custom_identity( if not candidate_norm or candidate_norm in {"custom", "auto", "openrouter"}: return None # Only when it resolves to a configured entry — never invent a ``custom:`` resolution - # can't honor. + # can't honor. ``candidate`` may be the entry's DISPLAY NAME, not the durable identity of a + # keyed ``providers:`` entry — re-resolve via its endpoint so every path returns the same + # config-key slug. try: entry = rp._get_named_custom_provider(candidate) - if entry is not None: - # ``candidate`` may be the entry's DISPLAY NAME, not the durable identity of a keyed - # ``providers:`` entry — re-resolve via its endpoint so every path returns the same - # config-key slug. - identity = find_custom_provider_identity(str(entry.get("base_url") or "")) - if identity: - return identity - return candidate_norm if candidate_norm.startswith("custom:") else f"custom:{candidate_norm}" except Exception: - pass - return None + return None + if entry is None: + return None + try: + identity = find_custom_provider_identity(str(entry.get("base_url") or "")) + except Exception: + return None + return identity or custom_provider_slug(candidate_norm) def is_routable_provider(provider: Optional[str]) -> bool: From 3462d82bc348722136e1a4f585b765298ba477a3 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 22:37:08 -0700 Subject: [PATCH 11/19] =?UTF-8?q?refactor(hermes=5Fcli):=20runtime=5Fprovi?= =?UTF-8?q?der=20=E2=80=94=20inline=20single-use=20registry/pool-state=20w?= =?UTF-8?q?rappers,=20compact=20call=20sites?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/runtime_provider.py | 140 ++++++++++++--------------------- 1 file changed, 50 insertions(+), 90 deletions(-) diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index a1e3fa90b5..e8862ff1b2 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -339,13 +339,8 @@ def _pool_entry_base_url(entry: Any) -> str: return getattr(entry, "runtime_base_url", None) or getattr(entry, "base_url", None) or "" -def _nous_pool_state(entry: Any) -> Dict[str, Any]: - return {k: getattr(entry, k, None) for k in ("agent_key", "agent_key_expires_at", "scope")} - - -def _registry_base_url(provider: str) -> str: - pconfig = PROVIDER_REGISTRY.get(provider) - return pconfig.inference_base_url if pconfig else "" +def _nous_entry_key_usable(entry: Any, min_ttl: int) -> bool: + return _agent_key_is_usable({k: getattr(entry, k, None) for k in ("agent_key", "agent_key_expires_at", "scope")}, min_ttl) def _nous_min_key_ttl() -> int: @@ -469,7 +464,8 @@ from hermes_cli.runtime_provider_backends import ( # noqa: E402,F401 _POOL_ENTRY_SIMPLE_MODES: Dict[str, tuple] = { "openai-codex": ("codex_responses", DEFAULT_CODEX_BASE_URL), "xai-oauth": ("codex_responses", DEFAULT_XAI_OAUTH_BASE_URL), "qwen-oauth": ("chat_completions", DEFAULT_QWEN_BASE_URL), "openrouter": ("chat_completions", OPENROUTER_BASE_URL), - "minimax-oauth": ("anthropic_messages", lambda: _registry_base_url("minimax-oauth")), "xai": ("codex_responses", ""), + "minimax-oauth": ("anthropic_messages", lambda: getattr(PROVIDER_REGISTRY.get("minimax-oauth"), "inference_base_url", "")), + "xai": ("codex_responses", ""), } @@ -507,15 +503,12 @@ def _resolve_runtime_from_pool_entry( pool: Optional[CredentialPool] = None, target_model: Optional[str] = None, ) -> Dict[str, Any]: model_cfg = model_cfg or _get_model_config() - api_mode, base_url = _pool_entry_mode_and_url( - provider, entry, model_cfg, _effective_model(model_cfg, target_model), _pool_entry_base_url(entry).rstrip("/") - ) + api_mode, base_url = _pool_entry_mode_and_url(provider, entry, model_cfg, _effective_model(model_cfg, target_model), + _pool_entry_base_url(entry).rstrip("/")) base_url = _finalize_base_url(provider, api_mode, base_url) api_mode = _maybe_apply_codex_app_server_runtime(provider=provider, api_mode=api_mode, model_cfg=model_cfg) - return _runtime( - provider, api_mode, base_url, _pool_entry_api_key(entry), - source=getattr(entry, "source", "pool"), credential_pool=pool, requested_provider=requested_provider, - ) + return _runtime(provider, api_mode, base_url, _pool_entry_api_key(entry), source=getattr(entry, "source", "pool"), + credential_pool=pool, requested_provider=requested_provider) def _openrouter_should_use_pool(requested_provider, model_cfg, explicit_api_key, explicit_base_url) -> bool: @@ -523,9 +516,7 @@ def _openrouter_should_use_pool(requested_provider, model_cfg, explicit_api_key, cfg_base_url = str(model_cfg.get("base_url") or "").strip() env_openai_base_url = _getenv("OPENAI_BASE_URL", "").strip() env_openrouter_base_url = _getenv("OPENROUTER_BASE_URL", "").strip() - has_custom_endpoint = bool(explicit_base_url or env_openai_base_url or env_openrouter_base_url) or bool( - cfg_base_url and _cfg_provider(model_cfg) in {"auto", "custom"} - ) + has_custom_endpoint = bool(explicit_base_url or env_openai_base_url or env_openrouter_base_url) or bool(cfg_base_url and _cfg_provider(model_cfg) in {"auto", "custom"}) return requested_provider in {"openrouter", "auto"} and not has_custom_endpoint and not bool(explicit_api_key or explicit_base_url) @@ -534,7 +525,7 @@ def _refresh_nous_pool_entry(pool: CredentialPool, entry: Any, pool_api_key: str selection (avoids network calls in `hermes auth list`); refresh here before falling back to singleton auth resolution. Returns (entry, pool_api_key) — key "" when still unusable.""" min_ttl = _nous_min_key_ttl() - if _agent_key_is_usable(_nous_pool_state(entry), min_ttl): + if _nous_entry_key_usable(entry, min_ttl): return entry, pool_api_key logger.debug("Nous pool entry agent_key expired/missing, refreshing selected pool entry") try: @@ -545,7 +536,7 @@ def _refresh_nous_pool_entry(pool: CredentialPool, entry: Any, pool_api_key: str if refreshed is not None: entry = refreshed pool_api_key = _pool_entry_api_key(entry) - if not pool_api_key or not _agent_key_is_usable(_nous_pool_state(entry), min_ttl): + if not pool_api_key or not _nous_entry_key_usable(entry, min_ttl): logger.debug("Nous pool entry agent_key still unavailable, falling through to runtime resolution") pool_api_key = "" return entry, pool_api_key @@ -555,9 +546,8 @@ def _resolve_from_pool( provider: str, requested_provider: str, model_cfg: Dict[str, Any], explicit_api_key, explicit_base_url, target_model ) -> Optional[Dict[str, Any]]: """Runtime from the provider's credential pool, or None to continue down the ladder.""" - should_use_pool = provider != "openrouter" or _openrouter_should_use_pool( - requested_provider, model_cfg, explicit_api_key, explicit_base_url - ) + should_use_pool = provider != "openrouter" or _openrouter_should_use_pool(requested_provider, model_cfg, explicit_api_key, + explicit_base_url) try: pool = load_pool(provider) if should_use_pool else None except Exception: @@ -571,10 +561,8 @@ def _resolve_from_pool( if provider == "nous": entry, pool_api_key = _refresh_nous_pool_entry(pool, entry, pool_api_key) if pool_api_key and credential_pool_matches_provider(pool, provider, base_url=_pool_entry_base_url(entry)): - return _resolve_runtime_from_pool_entry( - provider=provider, entry=entry, requested_provider=requested_provider, - model_cfg=model_cfg, pool=pool, target_model=target_model, - ) + return _resolve_runtime_from_pool_entry(provider=provider, entry=entry, requested_provider=requested_provider, + model_cfg=model_cfg, pool=pool, target_model=target_model) return None @@ -597,14 +585,10 @@ def _creds_fallback(api_key, explicit_base_url, base_url, expiry, expiry_key, re def _explicit_codex(requested_provider, model_cfg, api_key, explicit_base_url, target_model): - api_key, base_url, last_refresh = _creds_fallback( - api_key, explicit_base_url, explicit_base_url or DEFAULT_CODEX_BASE_URL, None, "last_refresh", - resolve_codex_runtime_credentials, - ) - return _runtime( - "openai-codex", "codex_responses", base_url, api_key, - source="explicit", last_refresh=last_refresh, requested_provider=requested_provider, - ) + api_key, base_url, last_refresh = _creds_fallback(api_key, explicit_base_url, explicit_base_url or DEFAULT_CODEX_BASE_URL, + None, "last_refresh", resolve_codex_runtime_credentials) + return _runtime("openai-codex", "codex_responses", base_url, api_key, source="explicit", last_refresh=last_refresh, + requested_provider=requested_provider) def _explicit_nous(requested_provider, model_cfg, api_key, explicit_base_url, target_model): @@ -616,23 +600,17 @@ def _explicit_nous(requested_provider, model_cfg, api_key, explicit_base_url, ta ) # The agent_key compatibility field is used for inference only when it holds a NAS invoke JWT; # raw OAuth access_token fallback is handled by resolve_nous_runtime_credentials(). - api_key = api_key or ( - str(state.get("agent_key") or "").strip() if _agent_key_is_usable(state, _nous_min_key_ttl()) else "" - ) - api_key, base_url, expires_at = _creds_fallback( - api_key, explicit_base_url, base_url, state.get("agent_key_expires_at") or state.get("expires_at"), "expires_at", - _resolve_nous_creds, - ) - return _runtime( - "nous", nous_api_mode(_effective_model(model_cfg, target_model)), base_url, api_key, - source="explicit", expires_at=expires_at, requested_provider=requested_provider, - ) + api_key = api_key or (str(state.get("agent_key") or "").strip() if _agent_key_is_usable(state, _nous_min_key_ttl()) else "") + api_key, base_url, expires_at = _creds_fallback(api_key, explicit_base_url, base_url, + state.get("agent_key_expires_at") or state.get("expires_at"), "expires_at", + _resolve_nous_creds) + return _runtime("nous", nous_api_mode(_effective_model(model_cfg, target_model)), base_url, api_key, source="explicit", + expires_at=expires_at, requested_provider=requested_provider) def _explicit_azure_foundry(requested_provider, model_cfg, api_key, base_url, target_model): - return _resolve_azure_foundry_runtime( - requested_provider=requested_provider, model_cfg=model_cfg, explicit_api_key=api_key, explicit_base_url=base_url - ) + return _resolve_azure_foundry_runtime(requested_provider=requested_provider, model_cfg=model_cfg, explicit_api_key=api_key, + explicit_base_url=base_url) def _actual_local_key(provider: str, api_key: str, base_url: str) -> str: @@ -658,9 +636,8 @@ def _explicit_api_key_provider(provider, pconfig, requested_provider, model_cfg, base_url = creds.get("base_url", "").rstrip("/") if provider == "actual": base_url = normalize_actual_base_url(base_url) - api_mode = _api_key_provider_api_mode( - provider, model_cfg, api_key, base_url, target_model or model_cfg.get("default", ""), opencode_by_model=False - ) + api_mode = _api_key_provider_api_mode(provider, model_cfg, api_key, base_url, target_model or model_cfg.get("default", ""), + opencode_by_model=False) api_key = _actual_local_key(provider, api_key, base_url) return _runtime(provider, api_mode, base_url.rstrip("/"), api_key, source="explicit", requested_provider=requested_provider) @@ -686,9 +663,8 @@ def _resolve_explicit_runtime( return resolver(requested_provider, model_cfg, explicit_api_key, explicit_base_url, target_model) pconfig = PROVIDER_REGISTRY.get(provider) if pconfig and pconfig.auth_type == "api_key": - return _explicit_api_key_provider( - provider, pconfig, requested_provider, model_cfg, explicit_api_key, explicit_base_url, target_model - ) + return _explicit_api_key_provider(provider, pconfig, requested_provider, model_cfg, explicit_api_key, + explicit_base_url, target_model) return None @@ -749,10 +725,8 @@ def _minimax_oauth_runtime(provider, requested_provider) -> Optional[Dict[str, A if not (pconfig and pconfig.auth_type == "oauth_minimax"): return None creds = auth_mod.resolve_minimax_oauth_runtime_credentials() - return _runtime( - provider, "anthropic_messages", creds["base_url"], creds["api_key"], - source=creds.get("source", "oauth"), requested_provider=requested_provider, - ) + return _runtime(provider, "anthropic_messages", creds["base_url"], creds["api_key"], source=creds.get("source", "oauth"), + requested_provider=requested_provider) # ── env/config paths for anthropic and registry api_key providers ────────────────────────── @@ -815,9 +789,8 @@ def _api_key_provider_runtime(provider, pconfig, requested_provider, model_cfg, base_url = _config_base_url_for_provider(model_cfg, provider) or creds.get("base_url", "").rstrip("/") if provider == "actual": base_url = normalize_actual_base_url(base_url) - api_mode = _api_key_provider_api_mode( - provider, model_cfg, creds.get("api_key", ""), base_url, target_model or model_cfg.get("default", ""), opencode_by_model=True - ) + api_mode = _api_key_provider_api_mode(provider, model_cfg, creds.get("api_key", ""), base_url, + target_model or model_cfg.get("default", ""), opencode_by_model=True) base_url = _finalize_base_url(provider, api_mode, base_url) api_key = _actual_local_key(provider, creds.get("api_key", ""), base_url) return _runtime(provider, api_mode, base_url, api_key, source=creds.get("source", "env"), requested_provider=requested_provider) @@ -864,30 +837,21 @@ def _resolve_vertex_runtime(requested_provider: str) -> Dict[str, Any]: def _resolve_requested_shortcuts(requested_provider, explicit_api_key, explicit_base_url, target_model) -> Optional[Dict[str, Any]]: """Providers decided on the REQUESTED name alone, before custom / pool / generic paths.""" if requested_provider == "moa": - return _runtime( - "moa", "chat_completions", "moa://local", "moa-virtual-provider", - source="moa-virtual-provider", requested_provider=requested_provider, - ) + return _runtime("moa", "chat_completions", "moa://local", "moa-virtual-provider", source="moa-virtual-provider", + requested_provider=requested_provider) # Azure Anthropic short-circuit: an explicit Azure endpoint with provider="anthropic" must # bypass _resolve_named_custom_runtime (which would yield custom/chat_completions/no key). eff_base = (explicit_base_url or "").strip() if requested_provider == "anthropic" and base_url_host_matches(eff_base, "azure.com"): - azure_key = ( - (explicit_api_key or "").strip() - or _getenv("AZURE_ANTHROPIC_KEY", "").strip() - or _getenv("ANTHROPIC_API_KEY", "").strip() - ) - return _runtime( - "anthropic", "anthropic_messages", eff_base.rstrip("/"), azure_key, - source="azure-explicit", requested_provider=requested_provider, - ) + azure_key = (explicit_api_key or "").strip() or _getenv("AZURE_ANTHROPIC_KEY", "").strip() or _getenv("ANTHROPIC_API_KEY", "").strip() + return _runtime("anthropic", "anthropic_messages", eff_base.rstrip("/"), azure_key, source="azure-explicit", + requested_provider=requested_provider) # Azure Foundry resolves before the custom-runtime / pool / generic paths so its config is # always picked up from model.base_url + model.api_mode, with or without explicit_* args. if requested_provider == "azure-foundry": - return _resolve_azure_foundry_runtime( - requested_provider=requested_provider, model_cfg=_get_model_config(), - explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, target_model=target_model, - ) + return _resolve_azure_foundry_runtime(requested_provider=requested_provider, model_cfg=_get_model_config(), + explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, + target_model=target_model) if requested_provider in _VERTEX_NAMES: return _resolve_vertex_runtime(requested_provider) return None @@ -908,9 +872,8 @@ def _local_endpoint_bypass(requested_provider: str, explicit_api_key, explicit_b def _openrouter_fallback(requested_provider, explicit_api_key, explicit_base_url) -> Dict[str, Any]: - runtime = _resolve_openrouter_runtime( - requested_provider=requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url - ) + runtime = _resolve_openrouter_runtime(requested_provider=requested_provider, explicit_api_key=explicit_api_key, + explicit_base_url=explicit_base_url) runtime["requested_provider"] = requested_provider return runtime @@ -950,10 +913,8 @@ def resolve_runtime_provider( runtime = _resolve_requested_shortcuts(requested_provider, explicit_api_key, explicit_base_url, target_model) if runtime: return runtime - runtime = _resolve_named_custom_runtime( - requested_provider=requested_provider, explicit_api_key=explicit_api_key, - explicit_base_url=explicit_base_url, target_model=target_model, - ) + runtime = _resolve_named_custom_runtime(requested_provider=requested_provider, explicit_api_key=explicit_api_key, + explicit_base_url=explicit_base_url, target_model=target_model) if runtime: runtime["requested_provider"] = requested_provider return runtime @@ -970,10 +931,9 @@ def _provider_rungs(provider, requested_provider, explicit_api_key, explicit_bas returned nothing; the last rung (OpenRouter / bare-custom fallback) always yields a runtime.""" model_cfg = _get_model_config() yield _opencode_free_runtime(provider, requested_provider, model_cfg, target_model) - yield _resolve_explicit_runtime( - provider=provider, requested_provider=requested_provider, model_cfg=model_cfg, - explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, target_model=target_model, - ) + yield _resolve_explicit_runtime(provider=provider, requested_provider=requested_provider, model_cfg=model_cfg, + explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, + target_model=target_model) yield _resolve_from_pool(provider, requested_provider, model_cfg, explicit_api_key, explicit_base_url, target_model) if provider in _OAUTH_RUNTIME_PROVIDERS: yield _resolve_oauth_runtime(provider, requested_provider, model_cfg, target_model) From 465c6dd4e98d4955e555114eb4c1b85d11a802ff Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 22:39:49 -0700 Subject: [PATCH 12/19] =?UTF-8?q?refactor(hermes=5Fcli):=20provider=20clus?= =?UTF-8?q?ter=20=E2=80=94=20hanging-indent=20repack=20of=20exploded=20cal?= =?UTF-8?q?l=20sites=20(AST-identical)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/providers.py | 77 +++++++++---------------- hermes_cli/route_identity.py | 6 +- hermes_cli/runtime_provider.py | 47 ++++++--------- hermes_cli/runtime_provider_backends.py | 54 +++++++---------- hermes_cli/runtime_provider_custom.py | 63 ++++++++------------ 5 files changed, 91 insertions(+), 156 deletions(-) diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py index eccefc684b..e394dc8f69 100644 --- a/hermes_cli/providers.py +++ b/hermes_cli/providers.py @@ -178,11 +178,9 @@ def _models_dev_info(canonical: str, allow_network: bool = True): def _overlay_pdef(canonical, ov: HermesOverlay, name, env_vars, base_url, doc, source) -> ProviderDef: - return ProviderDef( - id=canonical, name=name, transport=ov.transport, api_key_env_vars=env_vars, base_url=base_url, - base_url_env_var=ov.base_url_env_var, is_aggregator=ov.is_aggregator, auth_type=ov.auth_type, doc=doc, - source=source, - ) + return ProviderDef(id=canonical, name=name, transport=ov.transport, api_key_env_vars=env_vars, base_url=base_url, + base_url_env_var=ov.base_url_env_var, is_aggregator=ov.is_aggregator, auth_type=ov.auth_type, doc=doc, + source=source) def get_provider(name: str, *, allow_network: bool = True) -> Optional[ProviderDef]: @@ -197,15 +195,11 @@ def get_provider(name: str, *, allow_network: bool = True) -> Optional[ProviderD for ev in ov.extra_env_vars: if ev not in env_vars: env_vars.append(ev) - return _overlay_pdef( - canonical, ov, mdev_info.name, tuple(env_vars), ov.base_url_override or mdev_info.api, mdev_info.doc, - "models.dev", - ) + return _overlay_pdef(canonical, ov, mdev_info.name, tuple(env_vars), ov.base_url_override or mdev_info.api, + mdev_info.doc, "models.dev") if overlay is not None: - return _overlay_pdef( - canonical, overlay, _LABEL_OVERRIDES.get(canonical, canonical), overlay.extra_env_vars, - overlay.base_url_override, "", "hermes", - ) + return _overlay_pdef(canonical, overlay, _LABEL_OVERRIDES.get(canonical, canonical), overlay.extra_env_vars, + overlay.base_url_override, "", "hermes") # Plugin-registered profiles (plugins/model-providers//) absent from models.dev and # HERMES_OVERLAYS would otherwise be "Unknown provider" in /model, --provider and model-switch # even though the picker lists them. Only profiles with a concrete endpoint resolve here: @@ -217,12 +211,10 @@ def get_provider(name: str, *, allow_network: bool = True) -> Optional[ProviderD _prof = _profile(canonical) if _prof is not None and (_prof.base_url or "").strip(): _api_mode_to_transport = {v: k for k, v in TRANSPORT_TO_API_MODE.items()} - return ProviderDef( - id=canonical, name=_prof.display_name or _prof.name or canonical, - transport=_api_mode_to_transport.get(_prof.api_mode, "openai_chat"), - api_key_env_vars=tuple(_prof.env_vars or ()), base_url=_prof.base_url or "", - auth_type=_prof.auth_type or "api_key", source="plugin-profile", - ) + return ProviderDef(id=canonical, name=_prof.display_name or _prof.name or canonical, + transport=_api_mode_to_transport.get(_prof.api_mode, "openai_chat"), + api_key_env_vars=tuple(_prof.env_vars or ()), base_url=_prof.base_url or "", + auth_type=_prof.auth_type or "api_key", source="plugin-profile") except Exception: pass return None @@ -334,10 +326,8 @@ def determine_api_mode(provider: str, base_url: str = "", model: str = "") -> st def _user_pdef(pid: str, name: str, base_url: str, key_env: str, transport: str = "openai_chat") -> ProviderDef: """``source="user-config"`` ProviderDef shared by ``providers:`` and ``custom_providers:`` entries.""" - return ProviderDef( - id=pid, name=name, transport=transport, api_key_env_vars=(key_env,) if key_env else (), - base_url=base_url, is_aggregator=False, auth_type="api_key", source="user-config", - ) + return ProviderDef(id=pid, name=name, transport=transport, api_key_env_vars=(key_env,) if key_env else (), + base_url=base_url, is_aggregator=False, auth_type="api_key", source="user-config") def resolve_user_provider(name: str, user_config: Dict[str, Any]) -> Optional[ProviderDef]: @@ -347,12 +337,10 @@ def resolve_user_provider(name: str, user_config: Dict[str, Any]) -> Optional[Pr entry = user_config.get(name) if not isinstance(entry, dict): return None - return _user_pdef( - name, entry.get("name", "") or name, - entry.get("api", "") or entry.get("url", "") or entry.get("base_url", "") or "", - entry.get("key_env") or entry.get("api_key_env") or "", - entry.get("transport", "openai_chat") or "openai_chat", - ) + return _user_pdef(name, entry.get("name", "") or name, + entry.get("api", "") or entry.get("url", "") or entry.get("base_url", "") or "", + entry.get("key_env") or entry.get("api_key_env") or "", + entry.get("transport", "openai_chat") or "openai_chat") def custom_provider_slug(display_name: str, provider_key: str = "") -> str: @@ -398,9 +386,8 @@ def resolve_custom_provider(name: str, custom_providers: Optional[List[Dict[str, if not display_name or not api_url: continue provider_key = (entry.get("provider_key") or "").strip() - pdef = _user_pdef( - custom_provider_slug(display_name, provider_key), display_name, api_url, (entry.get("key_env") or "").strip() - ) + pdef = _user_pdef(custom_provider_slug(display_name, provider_key), display_name, api_url, + (entry.get("key_env") or "").strip()) if first_valid is None: first_valid = pdef if requested in custom_provider_aliases(display_name, provider_key): @@ -422,11 +409,9 @@ def _lossy_alias_registry_pdef(raw: str, canonical: str) -> Optional[ProviderDef if _pcfg is None: return None if sum(1 for _rid in _AUTH_PROVIDER_REGISTRY if normalize_provider(_rid) == canonical) > 1: - return ProviderDef( - id=_pcfg.id, name=_pcfg.name, transport="openai_chat", - api_key_env_vars=tuple(_pcfg.api_key_env_vars or ()), base_url=_pcfg.inference_base_url or "", - source="hermes-auth-registry", - ) + return ProviderDef(id=_pcfg.id, name=_pcfg.name, transport="openai_chat", + api_key_env_vars=tuple(_pcfg.api_key_env_vars or ()), base_url=_pcfg.inference_base_url or "", + source="hermes-auth-registry") except Exception: pass return None @@ -443,16 +428,12 @@ def _llamacpp_pdef() -> Optional[ProviderDef]: endpoint = None if not endpoint: return None - return ProviderDef( - id="llamacpp", name="Local", transport="openai_chat", api_key_env_vars=(), base_url=endpoint["base_url"], - source="local-runtime", - ) + return ProviderDef(id="llamacpp", name="Local", transport="openai_chat", api_key_env_vars=(), base_url=endpoint["base_url"], + source="local-runtime") -def resolve_provider_full( - name: str, user_providers: Optional[Dict[str, Any]] = None, - custom_providers: Optional[List[Dict[str, Any]]] = None, -) -> Optional[ProviderDef]: +def resolve_provider_full(name: str, user_providers: Optional[Dict[str, Any]] = None, + custom_providers: Optional[List[Dict[str, Any]]] = None) -> Optional[ProviderDef]: """Full resolution chain: user ``providers.`` -> lossy-alias registry id -> built-in (models.dev + overlays) -> user providers (canonical, then raw) -> ``custom_providers`` -> managed llamacpp -> models.dev directly. User-defined ``providers.`` is tried FIRST on @@ -486,10 +467,8 @@ def resolve_provider_full( try: mdev_info = _models_dev_info(canonical) if mdev_info is not None: - return ProviderDef( - id=canonical, name=mdev_info.name, transport="openai_chat", api_key_env_vars=mdev_info.env, - base_url=mdev_info.api, source="models.dev", - ) + return ProviderDef(id=canonical, name=mdev_info.name, transport="openai_chat", api_key_env_vars=mdev_info.env, + base_url=mdev_info.api, source="models.dev") except Exception: pass return None diff --git a/hermes_cli/route_identity.py b/hermes_cli/route_identity.py index 3732874d72..6ada71c1d0 100644 --- a/hermes_cli/route_identity.py +++ b/hermes_cli/route_identity.py @@ -44,10 +44,8 @@ def normalize_route_base_url(base_url: Any) -> str: return normalized -def should_clear_context_pin( - configured_model: Any, active_model: Any, configured_base_url: Any, active_base_url: Any, - configured_provider: Any, active_provider: Any, -) -> bool: +def should_clear_context_pin(configured_model: Any, active_model: Any, configured_base_url: Any, active_base_url: Any, + configured_provider: Any, active_provider: Any) -> bool: """True when a configured ``model.context_length`` pin no longer matches its runtime route. Fail-closed: any error during route comparison returns ``True`` (drop the pin) so a stale window never silently inflates the compression threshold.""" diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index e8862ff1b2..8bb2a7b0b3 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -147,9 +147,7 @@ def _resolve_plain_custom_api_mode(model_cfg: Dict[str, Any], base_url: str) -> configured_mode = _parse_api_mode(model_cfg.get("api_mode")) detected_mode = _detect_api_mode_for_url(base_url) if configured_mode == "codex_responses" and detected_mode != "codex_responses": - logger.info( - "Ignoring persisted custom api_mode=codex_responses for non-OpenAI endpoint %s", base_url or "(unknown)", - ) + logger.info("Ignoring persisted custom api_mode=codex_responses for non-OpenAI endpoint %s", base_url or "(unknown)") configured_mode = None return configured_mode or detected_mode or "chat_completions" @@ -207,9 +205,8 @@ def _azure_inferred_api_mode(effective_model: str, api_mode: str) -> str: return inferred or api_mode -def _configured_or_fallback_api_mode( - provider: str, model_cfg: Dict[str, Any], base_url: str, effective_model: Any, *, opencode_by_model: bool -) -> str: +def _configured_or_fallback_api_mode(provider: str, model_cfg: Dict[str, Any], base_url: str, effective_model: Any, *, + opencode_by_model: bool) -> str: """Persisted ``model.api_mode`` when it belongs to this provider, else URL/transport fallback. OpenCode Zen/Go serve both anthropic_messages and chat_completions models, so (when ``opencode_by_model``) their mode is always re-derived from the effective model.""" @@ -218,9 +215,8 @@ def _configured_or_fallback_api_mode( return _configured_api_mode(provider, model_cfg) or _fallback_api_mode(provider, base_url, effective_model) -def _api_key_provider_api_mode( - provider: str, model_cfg: Dict[str, Any], api_key: str, base_url: str, effective_model: Any, *, opencode_by_model: bool -) -> str: +def _api_key_provider_api_mode(provider: str, model_cfg: Dict[str, Any], api_key: str, base_url: str, effective_model: Any, *, + opencode_by_model: bool) -> str: """api_mode for a registry ``api_key`` provider (explicit and env/config paths).""" if provider == "copilot": return _copilot_runtime_api_mode(model_cfg, api_key, target_model=effective_model) @@ -498,10 +494,9 @@ def _pool_entry_mode_and_url(provider, entry, model_cfg, effective_model, base_u return _configured_or_fallback_api_mode(provider, model_cfg, base_url, effective_model, opencode_by_model=True), base_url -def _resolve_runtime_from_pool_entry( - *, provider: str, entry: PooledCredential, requested_provider: str, model_cfg: Optional[Dict[str, Any]] = None, - pool: Optional[CredentialPool] = None, target_model: Optional[str] = None, -) -> Dict[str, Any]: +def _resolve_runtime_from_pool_entry(*, provider: str, entry: PooledCredential, requested_provider: str, + model_cfg: Optional[Dict[str, Any]] = None, pool: Optional[CredentialPool] = None, + target_model: Optional[str] = None) -> Dict[str, Any]: model_cfg = model_cfg or _get_model_config() api_mode, base_url = _pool_entry_mode_and_url(provider, entry, model_cfg, _effective_model(model_cfg, target_model), _pool_entry_base_url(entry).rstrip("/")) @@ -542,9 +537,8 @@ def _refresh_nous_pool_entry(pool: CredentialPool, entry: Any, pool_api_key: str return entry, pool_api_key -def _resolve_from_pool( - provider: str, requested_provider: str, model_cfg: Dict[str, Any], explicit_api_key, explicit_base_url, target_model -) -> Optional[Dict[str, Any]]: +def _resolve_from_pool(provider: str, requested_provider: str, model_cfg: Dict[str, Any], explicit_api_key, explicit_base_url, + target_model) -> Optional[Dict[str, Any]]: """Runtime from the provider's credential pool, or None to continue down the ladder.""" should_use_pool = provider != "openrouter" or _openrouter_should_use_pool(requested_provider, model_cfg, explicit_api_key, explicit_base_url) @@ -650,10 +644,9 @@ _EXPLICIT_RESOLVERS: Dict[str, Callable[..., Dict[str, Any]]] = { } -def _resolve_explicit_runtime( - *, provider: str, requested_provider: str, model_cfg: Dict[str, Any], explicit_api_key: Optional[str] = None, - explicit_base_url: Optional[str] = None, target_model: Optional[str] = None, -) -> Optional[Dict[str, Any]]: +def _resolve_explicit_runtime(*, provider: str, requested_provider: str, model_cfg: Dict[str, Any], + explicit_api_key: Optional[str] = None, explicit_base_url: Optional[str] = None, + target_model: Optional[str] = None) -> Optional[Dict[str, Any]]: explicit_api_key = str(explicit_api_key or "").strip() explicit_base_url = str(explicit_base_url or "").strip().rstrip("/") if not explicit_api_key and not explicit_base_url: @@ -713,11 +706,9 @@ def _resolve_oauth_runtime(provider, requested_provider, model_cfg, target_model api_mode = spec.api_mode if callable(api_mode): api_mode = api_mode(_effective_model(model_cfg, target_model)) - return _runtime( - provider, api_mode, (creds.get("base_url") or "").rstrip("/") or spec.default_base_url, - creds.get("api_key", ""), source=creds.get("source", spec.default_source), - **{spec.expiry_key: creds.get(spec.expiry_key)}, requested_provider=requested_provider, - ) + return _runtime(provider, api_mode, (creds.get("base_url") or "").rstrip("/") or spec.default_base_url, + creds.get("api_key", ""), source=creds.get("source", spec.default_source), + **{spec.expiry_key: creds.get(spec.expiry_key)}, requested_provider=requested_provider) def _minimax_oauth_runtime(provider, requested_provider) -> Optional[Dict[str, Any]]: @@ -891,10 +882,8 @@ def _opencode_free_runtime(provider, requested_provider, model_cfg, target_model return free_runtime -def resolve_runtime_provider( - *, requested: Optional[str] = None, explicit_api_key: Optional[str] = None, - explicit_base_url: Optional[str] = None, target_model: Optional[str] = None, -) -> Dict[str, Any]: +def resolve_runtime_provider(*, requested: Optional[str] = None, explicit_api_key: Optional[str] = None, + explicit_base_url: Optional[str] = None, target_model: Optional[str] = None) -> Dict[str, Any]: """Resolve runtime provider credentials for agent execution. Ladder (order is behavior — each rung returns or raises, else falls to the next): 1. disabled-provider guard (``providers..enabled: false``) diff --git a/hermes_cli/runtime_provider_backends.py b/hermes_cli/runtime_provider_backends.py index 7115961c71..a69baa083e 100644 --- a/hermes_cli/runtime_provider_backends.py +++ b/hermes_cli/runtime_provider_backends.py @@ -61,10 +61,9 @@ def _azure_foundry_api_key(rp, explicit_api_key: str) -> str: return api_key -def _resolve_azure_foundry_runtime( - *, requested_provider: str, model_cfg: Dict[str, Any], explicit_api_key: Optional[str] = None, - explicit_base_url: Optional[str] = None, target_model: Optional[str] = None, -) -> Dict[str, Any]: +def _resolve_azure_foundry_runtime(*, requested_provider: str, model_cfg: Dict[str, Any], + explicit_api_key: Optional[str] = None, explicit_base_url: Optional[str] = None, + target_model: Optional[str] = None) -> Dict[str, Any]: """Azure Foundry: ``model.base_url`` + ``model.api_mode`` (or explicit overrides), API key from ``.env``/env or a per-request Entra ID token, trailing ``/v1`` stripped for Anthropic-style endpoints (the Anthropic SDK appends /v1/messages itself).""" @@ -100,15 +99,11 @@ def _resolve_azure_foundry_runtime( api_key, source, auth_mode, entra = _azure_entra_credentials(cfg_entra), "entra_id", "entra_id", ( {"scope": scope} if scope else {} ) - return rp._runtime( - "azure-foundry", cfg_api_mode, base_url, api_key, - auth_mode=auth_mode, entra=entra, source=source, requested_provider=requested_provider, - ) - return rp._runtime( - "azure-foundry", cfg_api_mode, base_url, _azure_foundry_api_key(rp, explicit_api_key), - auth_mode="api_key", source="explicit" if (explicit_api_key or explicit_base_url) else "config", - requested_provider=requested_provider, - ) + return rp._runtime("azure-foundry", cfg_api_mode, base_url, api_key, auth_mode=auth_mode, entra=entra, source=source, + requested_provider=requested_provider) + return rp._runtime("azure-foundry", cfg_api_mode, base_url, _azure_foundry_api_key(rp, explicit_api_key), + auth_mode="api_key", source="explicit" if (explicit_api_key or explicit_base_url) else "config", + requested_provider=requested_provider) # ── OpenRouter / bare custom fallback ────────────────────────────────────────────────────── @@ -168,10 +163,8 @@ def _resolve_openrouter_runtime( cfg_api_mode = rp._parse_api_mode(model_cfg.get("api_mode")) # Explicit "custom" stays "custom" rather than relabeling to "openrouter". if requested_norm != "custom": - return rp._runtime( - "openrouter", cfg_api_mode or rp._detect_api_mode_for_url(base_url) or "chat_completions", - base_url, api_key, source=source, - ) + return rp._runtime("openrouter", cfg_api_mode or rp._detect_api_mode_for_url(base_url) or "chat_completions", base_url, + api_key, source=source) if base_url: # provider_name makes pool lookup prefer name match over base_url (fixes credential # mix-ups when multiple custom providers share a base_url). @@ -205,10 +198,9 @@ def _resolve_bedrock_runtime(requested_provider: str, model_cfg: Dict[str, Any], Claude → AnthropicBedrock SDK (prompt caching, thinking budgets); others → Converse API. AWS_BEARER_TOKEN_BEDROCK auth is unsupported by AnthropicBedrock (SigV4 only), so bearer users go through Converse regardless of model.""" - from agent.bedrock_adapter import ( - bedrock_openai_base_url, has_aws_credentials, is_anthropic_bedrock_model, is_openai_bedrock_model, - resolve_aws_auth_env_var, resolve_bedrock_bearer_token, resolve_bedrock_runtime_region, - ) + from agent.bedrock_adapter import (bedrock_openai_base_url, has_aws_credentials, is_anthropic_bedrock_model, + is_openai_bedrock_model, resolve_aws_auth_env_var, resolve_bedrock_bearer_token, + resolve_bedrock_runtime_region) from hermes_cli.config import load_config # direct (not the origin delegate), as before rp = _rp() # Explicitly selected bedrock trusts boto3's credential chain (IMDS, ECS/Lambda roles, SSO) @@ -230,16 +222,12 @@ def _resolve_bedrock_runtime(requested_provider: str, model_cfg: Dict[str, Any], guardrail_config = _bedrock_guardrail_config(bedrock_cfg) current_model = str(target_model or model_cfg.get("default") or "").strip() has_bearer_token = bool(os.environ.get("AWS_BEARER_TOKEN_BEDROCK", "").strip()) - runtime = rp._runtime( - "bedrock", "bedrock_converse", f"https://bedrock-runtime.{region}.amazonaws.com", "aws-sdk", - source=auth_source, region=region, requested_provider=requested_provider, - ) + runtime = rp._runtime("bedrock", "bedrock_converse", f"https://bedrock-runtime.{region}.amazonaws.com", "aws-sdk", + source=auth_source, region=region, requested_provider=requested_provider) if is_openai_bedrock_model(current_model): bearer = resolve_bedrock_bearer_token() - runtime.update( - api_mode="codex_responses", base_url=bedrock_openai_base_url(region), api_key=bearer or "aws-sdk", - source="AWS_BEARER_TOKEN_BEDROCK" if bearer else auth_source, model=current_model, bedrock_openai=True, - ) + runtime.update(api_mode="codex_responses", base_url=bedrock_openai_base_url(region), api_key=bearer or "aws-sdk", + source="AWS_BEARER_TOKEN_BEDROCK" if bearer else auth_source, model=current_model, bedrock_openai=True) elif is_anthropic_bedrock_model(current_model) and not has_bearer_token: runtime.update(api_mode="anthropic_messages", bedrock_anthropic=True) if guardrail_config: @@ -274,8 +262,6 @@ def _is_external_process_provider(provider: str) -> bool: def _resolve_external_process_runtime(provider: str, requested_provider: str) -> Dict[str, Any]: rp = _rp() creds = rp.resolve_external_process_provider_credentials(provider) - return rp._runtime( - provider, "chat_completions", creds.get("base_url", "").rstrip("/"), creds.get("api_key", ""), - command=creds.get("command", ""), args=list(creds.get("args") or []), - source=creds.get("source", "process"), requested_provider=requested_provider, - ) + return rp._runtime(provider, "chat_completions", creds.get("base_url", "").rstrip("/"), creds.get("api_key", ""), + command=creds.get("command", ""), args=list(creds.get("args") or []), + source=creds.get("source", "process"), requested_provider=requested_provider) diff --git a/hermes_cli/runtime_provider_custom.py b/hermes_cli/runtime_provider_custom.py index 442ab2eaba..125ab9a639 100644 --- a/hermes_cli/runtime_provider_custom.py +++ b/hermes_cli/runtime_provider_custom.py @@ -80,9 +80,8 @@ def _lift_extra_headers(entry: Dict[str, Any], result: Dict[str, Any]) -> None: result["extra_headers"] = extra_headers -def _lift_common_custom_fields( - entry: Dict[str, Any], result: Dict[str, Any], *, provider_key: str, key_env: str, api_mode: Optional[str] -) -> None: +def _lift_common_custom_fields(entry: Dict[str, Any], result: Dict[str, Any], *, provider_key: str, key_env: str, + api_mode: Optional[str]) -> None: """Copy the optional fields shared by ``providers:`` and legacy ``custom_providers:`` entries.""" if key_env: result["key_env"] = key_env @@ -168,10 +167,8 @@ def _match_legacy_custom_provider(requested_norm: str, custom_providers) -> Opti model_name = _clean(entry.get("model", "")) if model_name: result["model"] = model_name - _lift_common_custom_fields( - entry, result, provider_key=provider_key, key_env=_clean(entry.get("key_env", "")), - api_mode=_rp()._parse_api_mode(entry.get("api_mode")), - ) + _lift_common_custom_fields(entry, result, provider_key=provider_key, key_env=_clean(entry.get("key_env", "")), + api_mode=_rp()._parse_api_mode(entry.get("api_mode"))) return result return None @@ -276,9 +273,8 @@ def find_custom_provider_identity_by_model(model: str) -> Optional[str]: return _find_custom_identity(_entry_serves_model) -def canonical_custom_identity( - *, base_url: Optional[str] = None, config_provider: Optional[str] = None, model: Optional[str] = None -) -> Optional[str]: +def canonical_custom_identity(*, base_url: Optional[str] = None, config_provider: Optional[str] = None, + model: Optional[str] = None) -> Optional[str]: """Recover a routable ``custom:`` identity for a bare custom provider. Every path that persists or restores a session's provider override must run the resolved provider through this so a bare ``"custom"`` is upgraded back to its durable menu key. Sources in priority order: @@ -374,10 +370,8 @@ def _try_resolve_from_custom_pool( # services; has_usable_secret's 4-char floor rejects them. Every other path # substitutes "no-key-required" for a loopback endpoint — this was the one gap. pool_api_key = "no-key-required" - return rp._runtime( - provider_label, api_mode_override or rp._detect_api_mode_for_url(base_url) or "chat_completions", - base_url, pool_api_key, source=f"pool:{pool_key}", credential_pool=pool, - ) + return rp._runtime(provider_label, api_mode_override or rp._detect_api_mode_for_url(base_url) or "chat_completions", + base_url, pool_api_key, source=f"pool:{pool_key}", credential_pool=pool) except Exception: continue return None @@ -390,9 +384,7 @@ def _custom_provider_request_overrides(custom_provider: Dict[str, Any]) -> Dict[ return {"extra_body": dict(extra_body)} -def _apply_custom_provider_extras( - custom_provider: Dict[str, Any], target_model: Optional[str], result: Dict[str, Any] -) -> None: +def _apply_custom_provider_extras(custom_provider: Dict[str, Any], target_model: Optional[str], result: Dict[str, Any]) -> None: """Copy model / capabilities / max_output_tokens / extra_headers / request_overrides onto a resolved custom runtime. An explicit ``target_model`` wins over the provider's configured default (auxiliary slots / background-review resolve a concrete model and must not fall back to @@ -422,11 +414,9 @@ def _resolve_llamacpp_runtime(requested_provider: str, explicit_api_key: Optiona except Exception: # noqa: BLE001 — resolution is best-effort endpoint = None if endpoint: - return rp._runtime( - "custom", "chat_completions", endpoint["base_url"], - (explicit_api_key or "").strip() or endpoint["api_key"] or "no-key-required", - source="local-runtime", requested_provider=requested_provider, - ) + return rp._runtime("custom", "chat_completions", endpoint["base_url"], + (explicit_api_key or "").strip() or endpoint["api_key"] or "no-key-required", source="local-runtime", + requested_provider=requested_provider) try: enabled = bool((rp.load_config().get("local_runtime") or {}).get("enabled")) except Exception: # noqa: BLE001 @@ -446,15 +436,12 @@ def _resolve_llamacpp_runtime(requested_provider: str, explicit_api_key: Optiona def _custom_runtime(rp, base_url: str, api_key: Any, api_mode: Optional[str], **extra: Any) -> Dict[str, Any]: """``custom`` runtime dict with URL-detected api_mode fallback and the no-auth placeholder.""" - return rp._runtime( - "custom", api_mode or rp._detect_api_mode_for_url(base_url) or "chat_completions", base_url, - api_key or "no-key-required", **extra, - ) + return rp._runtime("custom", api_mode or rp._detect_api_mode_for_url(base_url) or "chat_completions", base_url, + api_key or "no-key-required", **extra) -def _resolve_direct_alias_runtime( - requested_provider: str, explicit_api_key: Optional[str], explicit_base_url: str -) -> Dict[str, Any]: +def _resolve_direct_alias_runtime(requested_provider: str, explicit_api_key: Optional[str], + explicit_base_url: str) -> Dict[str, Any]: """Bare ``custom`` + explicit base_url (e.g. a ``model_aliases:`` direct alias).""" rp = _rp() base_url = explicit_base_url.strip().rstrip("/") @@ -485,10 +472,9 @@ def _opencode_family_for_custom(requested_provider: str, base_url: str) -> Optio return None -def _resolve_named_custom_runtime( - *, requested_provider: str, explicit_api_key: Optional[str] = None, explicit_base_url: Optional[str] = None, - target_model: Optional[str] = None, -) -> Optional[Dict[str, Any]]: +def _resolve_named_custom_runtime(*, requested_provider: str, explicit_api_key: Optional[str] = None, + explicit_base_url: Optional[str] = None, + target_model: Optional[str] = None) -> Optional[Dict[str, Any]]: """Runtime for a llamacpp alias, a bare-custom direct alias, or a configured custom entry. Aliases resolving to "custom" (ollama, vllm, llamacpp, …) are treated like bare ``custom``. A llamacpp alias with no explicit base_url resolves to the managed server first; an explicit @@ -529,15 +515,12 @@ def _resolve_named_custom_runtime( key_cmd = _clean(custom_provider.get("key_cmd", "")) if key_cmd and not rp.has_usable_secret(explicit_key): from agent.command_token_source import build_command_token_provider - token_provider = build_command_token_provider( - key_cmd, str(custom_provider.get("name", requested_provider) or "custom") - ) + token_provider = build_command_token_provider(key_cmd, str(custom_provider.get("name", requested_provider) or "custom")) if token_provider is not None: api_key = token_provider - result = _custom_runtime( - rp, base_url, api_key, custom_provider.get("api_mode"), - source=f"custom_provider:{custom_provider.get('name', requested_provider)}", requested_provider=requested_provider, - ) + result = _custom_runtime(rp, base_url, api_key, custom_provider.get("api_mode"), + source=f"custom_provider:{custom_provider.get('name', requested_provider)}", + requested_provider=requested_provider) _apply_custom_provider_extras(custom_provider, target_model, result) # OpenCode-family custom providers (opencode-go/zen names, or opencode.ai hosts) serve models # on different API surfaces — a static api_mode 503s for /v1/responses-only models. Re-derive From 7ab1754f6ffdaaee427041578cc7aa2822adb045 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 22:43:10 -0700 Subject: [PATCH 13/19] =?UTF-8?q?refactor(hermes=5Fcli):=20runtime=5Fprovi?= =?UTF-8?q?der/custom=20=E2=80=94=20=5Factual=5Furl=20helper,=20pool-selec?= =?UTF-8?q?t=20fold,=20minor=20collapses?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/runtime_provider.py | 30 ++++++++++----------------- hermes_cli/runtime_provider_custom.py | 10 +++------ 2 files changed, 14 insertions(+), 26 deletions(-) diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index 8bb2a7b0b3..031823b1f9 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -302,9 +302,7 @@ def _host_derived_api_key(base_url: str) -> str: labels = [lbl for lbl in hostname.split(".") if lbl] while labels and labels[0] in ("api", "www"): labels.pop(0) - if len(labels) < 2: - return "" - sanitized = "".join(ch if ch.isalnum() else "_" for ch in labels[-2]).upper() + sanitized = "".join(ch if ch.isalnum() else "_" for ch in labels[-2]).upper() if len(labels) >= 2 else "" if not sanitized or not sanitized[0].isalpha() or sanitized in ("OPENAI", "OPENROUTER", "OLLAMA"): return "" return (_getenv(f"{sanitized}_API_KEY", "") or "").strip() @@ -368,9 +366,7 @@ def _auto_detect_local_model(base_url: str) -> str: try: import requests url = base_url.rstrip("/") - if not url.endswith("/v1"): - url += "/v1" - resp = requests.get(url + "/models", timeout=(2, 3)) + resp = requests.get((url if url.endswith("/v1") else url + "/v1") + "/models", timeout=(2, 3)) if resp.ok: models = resp.json().get("data", []) if len(models) == 1 and models[0].get("id", ""): @@ -529,8 +525,7 @@ def _refresh_nous_pool_entry(pool: CredentialPool, entry: Any, pool_api_key: str logger.debug("Nous pool entry refresh failed: %s", exc) refreshed = None if refreshed is not None: - entry = refreshed - pool_api_key = _pool_entry_api_key(entry) + entry, pool_api_key = refreshed, _pool_entry_api_key(refreshed) if not pool_api_key or not _nous_entry_key_usable(entry, min_ttl): logger.debug("Nous pool entry agent_key still unavailable, falling through to runtime resolution") pool_api_key = "" @@ -614,6 +609,10 @@ def _actual_local_key(provider: str, api_key: str, base_url: str) -> str: return api_key +def _actual_url(provider: str, base_url: str) -> str: + return normalize_actual_base_url(base_url) if provider == "actual" else base_url + + def _explicit_api_key_provider(provider, pconfig, requested_provider, model_cfg, api_key, base_url, target_model): if not base_url: if provider in {"kimi-coding", "kimi-coding-cn"}: @@ -621,15 +620,12 @@ def _explicit_api_key_provider(provider, pconfig, requested_provider, model_cfg, else: env_url = _getenv(pconfig.base_url_env_var, "").strip().rstrip("/") if pconfig.base_url_env_var else "" base_url = env_url or pconfig.inference_base_url - if provider == "actual": - base_url = normalize_actual_base_url(base_url) + base_url = _actual_url(provider, base_url) if not api_key: creds = resolve_api_key_provider_credentials(provider) api_key = creds.get("api_key", "") if not base_url: - base_url = creds.get("base_url", "").rstrip("/") - if provider == "actual": - base_url = normalize_actual_base_url(base_url) + base_url = _actual_url(provider, creds.get("base_url", "").rstrip("/")) api_mode = _api_key_provider_api_mode(provider, model_cfg, api_key, base_url, target_model or model_cfg.get("default", ""), opencode_by_model=False) api_key = _actual_local_key(provider, api_key, base_url) @@ -767,9 +763,7 @@ def _api_key_provider_runtime(provider, pconfig, requested_provider, model_cfg, if provider == "actual" and not has_usable_secret(creds.get("api_key")): cfg_url = _config_base_url_for_provider(model_cfg, provider) if is_actual_local_base_url(normalize_actual_base_url(cfg_url or creds.get("base_url", "").rstrip("/"))): - creds = dict(creds) - creds["api_key"] = ACTUAL_LOCAL_NOAUTH_PLACEHOLDER - creds["source"] = creds.get("source") or "local-offline" + creds = {**creds, "api_key": ACTUAL_LOCAL_NOAUTH_PLACEHOLDER, "source": creds.get("source") or "local-offline"} # An explicitly selected API-key provider is authoritative: an empty key would defer failure # to the first request and make a later fallback look like a silent provider switch. if not has_usable_secret(creds.get("api_key")): @@ -777,9 +771,7 @@ def _api_key_provider_runtime(provider, pconfig, requested_provider, model_cfg, hint = f" Set {env_names}." if env_names else "" raise AuthError(f"No usable credentials found for provider '{provider}'.{hint}", provider=provider, code="missing_api_key") # Honour model.base_url when the configured provider matches (e.g. api.minimaxi.com China endpoint). - base_url = _config_base_url_for_provider(model_cfg, provider) or creds.get("base_url", "").rstrip("/") - if provider == "actual": - base_url = normalize_actual_base_url(base_url) + base_url = _actual_url(provider, _config_base_url_for_provider(model_cfg, provider) or creds.get("base_url", "").rstrip("/")) api_mode = _api_key_provider_api_mode(provider, model_cfg, creds.get("api_key", ""), base_url, target_model or model_cfg.get("default", ""), opencode_by_model=True) base_url = _finalize_base_url(provider, api_mode, base_url) diff --git a/hermes_cli/runtime_provider_custom.py b/hermes_cli/runtime_provider_custom.py index 125ab9a639..adc43c3a80 100644 --- a/hermes_cli/runtime_provider_custom.py +++ b/hermes_cli/runtime_provider_custom.py @@ -357,12 +357,8 @@ def _try_resolve_from_custom_pool( for pool_key in candidates: try: pool = rp.load_pool(pool_key) - if not pool.has_credentials(): - continue - entry = pool.select() - if entry is None: - continue - pool_api_key = rp._pool_entry_api_key(entry) + entry = pool.select() if pool.has_credentials() else None + pool_api_key = rp._pool_entry_api_key(entry) if entry is not None else "" if not pool_api_key: continue if not rp.has_usable_secret(pool_api_key) and rp._loopback_hostname(base_url_hostname(base_url)): @@ -399,7 +395,7 @@ def _apply_custom_provider_extras(custom_provider: Dict[str, Any], target_model: result["extra_headers"] = dict(custom_provider["extra_headers"]) request_overrides = _custom_provider_request_overrides(custom_provider) if request_overrides: - result["request_overrides"] = {**dict(result.get("request_overrides") or {}), **request_overrides} + result["request_overrides"] = {**(result.get("request_overrides") or {}), **request_overrides} def _resolve_llamacpp_runtime(requested_provider: str, explicit_api_key: Optional[str]) -> Dict[str, Any]: From c9268741521f4ff6a0b5e1853c5efa2522002baa Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 22:50:07 -0700 Subject: [PATCH 14/19] =?UTF-8?q?refactor(hermes=5Fcli):=20backends/custom?= =?UTF-8?q?=20=E2=80=94=20fold=20candidate=20lists,=20message=20literals;?= =?UTF-8?q?=20drop=20always-None=20provider=5Fname=20conditional?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/runtime_provider.py | 36 +++++++++-------------- hermes_cli/runtime_provider_backends.py | 32 ++++++--------------- hermes_cli/runtime_provider_custom.py | 38 +++++++++---------------- 3 files changed, 35 insertions(+), 71 deletions(-) diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index 031823b1f9..63317c01d7 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -240,10 +240,8 @@ def _maybe_apply_codex_app_server_runtime(*, provider: str, api_mode: str, model # ── base_url / credential helpers ────────────────────────────────────────────────────────── _ANTHROPIC_DEFAULT_BASE_URL = "https://api.anthropic.com" -_NO_ANTHROPIC_CREDENTIALS_MSG = ( - "No Anthropic credentials found. Set ANTHROPIC_TOKEN or ANTHROPIC_API_KEY, " - "run 'claude setup-token', or authenticate with 'claude /login'." -) +_NO_ANTHROPIC_CREDENTIALS_MSG = ("No Anthropic credentials found. Set ANTHROPIC_TOKEN or ANTHROPIC_API_KEY, " + "run 'claude setup-token', or authenticate with 'claude /login'.") def _runtime(provider: str, api_mode: str, base_url: Any, api_key: Any, **extra: Any) -> Dict[str, Any]: @@ -745,11 +743,9 @@ def _anthropic_env_runtime(requested_provider: str, model_cfg: Dict[str, Any]) - if base_url_host_matches(base_url, "azure.com") or (cfg_base_url and base_url_host_matches(cfg_base_url, "azure.com")): token = _azure_anthropic_env_key(model_cfg) if not token: - raise AuthError( - "No Azure Anthropic API key found. Set AZURE_ANTHROPIC_KEY or " - "ANTHROPIC_API_KEY, or point key_env/api_key_env in your " - "config.yaml model section at a custom env var." - ) + raise AuthError("No Azure Anthropic API key found. Set AZURE_ANTHROPIC_KEY or " + "ANTHROPIC_API_KEY, or point key_env/api_key_env in your " + "config.yaml model section at a custom env var.") else: token = _anthropic_token_or_raise() return _runtime("anthropic", "anthropic_messages", base_url, token, source="env", requested_provider=requested_provider) @@ -792,10 +788,8 @@ def _raise_if_provider_disabled(requested_provider: str) -> None: provs_cfg = full_cfg.get("providers") if isinstance(full_cfg, dict) else None block = provs_cfg.get(requested_provider) if isinstance(provs_cfg, dict) else None if isinstance(block, dict) and not _config_mod.is_provider_enabled(block): - raise ValueError( - f"provider {requested_provider!r} is disabled in config " - f"(providers.{requested_provider}.enabled: false)" - ) + raise ValueError(f"provider {requested_provider!r} is disabled in config " + f"(providers.{requested_provider}.enabled: false)") def _resolve_vertex_runtime(requested_provider: str) -> Dict[str, Any]: @@ -805,15 +799,13 @@ def _resolve_vertex_runtime(requested_provider: str) -> Dict[str, Any]: from agent.vertex_adapter import get_vertex_config token, base_url = get_vertex_config() if not token or not base_url: - raise AuthError( - "Vertex AI credentials could not be resolved. Vertex uses " - "OAuth2 (not a static API key): provide a service-account JSON " - "via GOOGLE_APPLICATION_CREDENTIALS (or VERTEX_CREDENTIALS_PATH) " - "in ~/.hermes/.env, or run 'gcloud auth application-default " - "login' for ADC. Set the GCP project/region under vertex: in " - "config.yaml if they aren't embedded in the credentials. " - "Run `hermes setup` to install Vertex support." - ) + raise AuthError("Vertex AI credentials could not be resolved. Vertex uses " + "OAuth2 (not a static API key): provide a service-account JSON " + "via GOOGLE_APPLICATION_CREDENTIALS (or VERTEX_CREDENTIALS_PATH) " + "in ~/.hermes/.env, or run 'gcloud auth application-default " + "login' for ADC. Set the GCP project/region under vertex: in " + "config.yaml if they aren't embedded in the credentials. " + "Run `hermes setup` to install Vertex support.") return _runtime("vertex", "chat_completions", base_url.rstrip("/"), token, source="vertex-oauth", requested_provider=requested_provider) diff --git a/hermes_cli/runtime_provider_backends.py b/hermes_cli/runtime_provider_backends.py index a69baa083e..400d4bc204 100644 --- a/hermes_cli/runtime_provider_backends.py +++ b/hermes_cli/runtime_provider_backends.py @@ -120,12 +120,9 @@ def _resolve_openrouter_runtime( rp = _rp() model_cfg = rp._get_model_config() cfg_base_url = model_cfg.get("base_url") if isinstance(model_cfg.get("base_url"), str) else "" - cfg_provider = model_cfg.get("provider") if isinstance(model_cfg.get("provider"), str) else "" - cfg_api_key = next( - (v.strip() for v in (model_cfg.get("api_key"), model_cfg.get("api")) if isinstance(v, str) and v.strip()), "" - ) + cfg_provider = (model_cfg.get("provider") if isinstance(model_cfg.get("provider"), str) else "").strip().lower() + cfg_api_key = next((v.strip() for v in (model_cfg.get("api_key"), model_cfg.get("api")) if isinstance(v, str) and v.strip()), "") requested_norm = (requested_provider or "").strip().lower() - cfg_provider = cfg_provider.strip().lower() # Aliases resolving to "custom" (ollama, vllm, …) follow bare-custom trust + routing rules. if requested_norm and requested_norm != "custom" and rp._resolves_to_custom(requested_norm): requested_norm = "custom" @@ -135,29 +132,20 @@ def _resolve_openrouter_runtime( (requested_norm == "auto" and cfg_provider in ("", "auto")) or (requested_norm == "custom" and rp._config_base_url_trustworthy_for_bare_custom(cfg_base_url, cfg_provider)) ) - base_url = ( - (explicit_base_url or "").strip() - or env_custom_base_url - or (cfg_base_url.strip() if use_config_base_url else "") - or env_openrouter_base_url - or OPENROUTER_BASE_URL - ).rstrip("/") + base_url = ((explicit_base_url or "").strip() or env_custom_base_url or (cfg_base_url.strip() if use_config_base_url else "") + or env_openrouter_base_url or OPENROUTER_BASE_URL).rstrip("/") is_openrouter_url = base_url_host_matches(base_url, "openrouter.ai") # Explicitly-configured OpenRouter mirrors (OPENROUTER_BASE_URL + provider=openrouter) still # count as OpenRouter for key selection. is_openrouter_context = is_openrouter_url or ( - requested_norm == "openrouter" - and (env_openrouter_base_url or base_url == env_openrouter_base_url) + requested_norm == "openrouter" and (env_openrouter_base_url or base_url == env_openrouter_base_url) and base_url == (env_openrouter_base_url or "").rstrip("/") ) if is_openrouter_context: candidates = [explicit_api_key, rp._getenv("OPENROUTER_API_KEY"), rp._getenv("OPENAI_API_KEY")] else: - candidates = [ - explicit_api_key, - (cfg_api_key if use_config_base_url else ""), - *rp._host_gated_env_key_candidates(base_url, ollama=True), - ] + candidates = [explicit_api_key, (cfg_api_key if use_config_base_url else ""), + *rp._host_gated_env_key_candidates(base_url, ollama=True)] api_key = next((str(c or "").strip() for c in candidates if rp.has_usable_secret(c)), "") source = "explicit" if (explicit_api_key or explicit_base_url) else "env/config" cfg_api_mode = rp._parse_api_mode(model_cfg.get("api_mode")) @@ -166,11 +154,7 @@ def _resolve_openrouter_runtime( return rp._runtime("openrouter", cfg_api_mode or rp._detect_api_mode_for_url(base_url) or "chat_completions", base_url, api_key, source=source) if base_url: - # provider_name makes pool lookup prefer name match over base_url (fixes credential - # mix-ups when multiple custom providers share a base_url). - pool_result = rp._try_resolve_from_custom_pool( - base_url, "custom", cfg_api_mode, provider_name=requested_provider if requested_norm != "custom" else None - ) + pool_result = rp._try_resolve_from_custom_pool(base_url, "custom", cfg_api_mode, provider_name=None) if pool_result: return pool_result # Local no-auth servers get a placeholder key — the OpenAI SDK requires a non-empty string. diff --git a/hermes_cli/runtime_provider_custom.py b/hermes_cli/runtime_provider_custom.py index adc43c3a80..0e6d31e4f3 100644 --- a/hermes_cli/runtime_provider_custom.py +++ b/hermes_cli/runtime_provider_custom.py @@ -134,10 +134,8 @@ def _match_new_style_provider(requested_norm: str, providers: Dict[str, Any]) -> base_url = _entry_url(entry) if not base_url: continue - result: Dict[str, Any] = { - "name": entry.get("name", ep_name), "base_url": base_url.strip(), - "api_key": api_key or _clean(entry.get("api_key", "")), "model": entry.get("default_model", ""), - } + result: Dict[str, Any] = {"name": entry.get("name", ep_name), "base_url": base_url.strip(), + "api_key": api_key or _clean(entry.get("api_key", "")), "model": entry.get("default_model", "")} # Command that PRINTS a short-lived credential; wrapped in a per-request token provider. key_cmd = _clean(entry.get("key_cmd", "")) if key_cmd: @@ -185,16 +183,12 @@ def _get_named_custom_provider(requested_provider: str) -> Optional[Dict[str, An if found: return found if isinstance(config.get("custom_providers"), dict): - logger.warning( - "custom_providers in config.yaml is a dict, not a list. " - "Each entry must be prefixed with '-' in YAML. " - "Run 'hermes doctor' for details." - ) + logger.warning("custom_providers in config.yaml is a dict, not a list. " + "Each entry must be prefixed with '-' in YAML. " + "Run 'hermes doctor' for details.") return None custom_providers = rp.get_compatible_custom_providers(config) - if not custom_providers: - return None - return _match_legacy_custom_provider(requested_norm, custom_providers) + return _match_legacy_custom_provider(requested_norm, custom_providers) if custom_providers else None def has_named_custom_provider(requested_provider: str) -> bool: @@ -418,16 +412,12 @@ def _resolve_llamacpp_runtime(requested_provider: str, explicit_api_key: Optiona except Exception: # noqa: BLE001 enabled = False if enabled: - raise ValueError( - "The local model server isn't running. It may still be " - "starting — try again in a moment, or check Settings → " - "Providers → Local models." - ) - raise ValueError( - "The local model server is turned off. Turn it back on in " - "Settings → Providers → Local models, or switch to another " - "model." - ) + raise ValueError("The local model server isn't running. It may still be " + "starting — try again in a moment, or check Settings → " + "Providers → Local models.") + raise ValueError("The local model server is turned off. Turn it back on in " + "Settings → Providers → Local models, or switch to another " + "model.") def _custom_runtime(rp, base_url: str, api_key: Any, api_mode: Optional[str], **extra: Any) -> Dict[str, Any]: @@ -524,9 +514,7 @@ def _resolve_named_custom_runtime(*, requested_provider: str, explicit_api_key: family = _opencode_family_for_custom(requested_provider, base_url) if family is not None and not custom_provider.get("api_mode"): from hermes_cli.models import normalize_opencode_base_url, opencode_model_api_mode - effective_model = str( - target_model or custom_provider.get("model") or rp._get_model_config().get("default") or "" - ).strip() + effective_model = str(target_model or custom_provider.get("model") or rp._get_model_config().get("default") or "").strip() if effective_model: result["api_mode"] = opencode_model_api_mode(family, effective_model) result["base_url"] = normalize_opencode_base_url(family, result["api_mode"], result["base_url"]) From 2c094dc7808c9c1d4e7c1625c3f19a31cc01bb6c Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 22:55:15 -0700 Subject: [PATCH 15/19] =?UTF-8?q?refactor(hermes=5Fcli):=20provider=20clus?= =?UTF-8?q?ter=20=E2=80=94=20fold=20multi-line=20boolean=20chains?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/provider_catalog.py | 8 ++------ hermes_cli/runtime_provider.py | 31 +++++++++---------------------- 2 files changed, 11 insertions(+), 28 deletions(-) diff --git a/hermes_cli/provider_catalog.py b/hermes_cli/provider_catalog.py index 3b6bdd6c81..8f381e6a05 100644 --- a/hermes_cli/provider_catalog.py +++ b/hermes_cli/provider_catalog.py @@ -86,12 +86,8 @@ def provider_catalog() -> list[ProviderDescriptor]: prof = profiles.get(slug) overlay = HERMES_OVERLAYS.get(slug) # auth_type: registry is authoritative; then profile, then overlay (moa → "virtual"), then api_key. - auth_type = ( - (cfg.auth_type if cfg else "") - or (prof.auth_type if prof else "") - or (overlay.auth_type if overlay else "") - or "api_key" - ) + auth_type = ((cfg.auth_type if cfg else "") or (prof.auth_type if prof else "") + or (overlay.auth_type if overlay else "") or "api_key") # Credential env vars: registry first (already normalized), else derived from the profile. if cfg and cfg.api_key_env_vars: api_key_vars, base_url_var = tuple(cfg.api_key_env_vars), cfg.base_url_env_var or "" diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index 63317c01d7..ee7d1e51a1 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -228,11 +228,7 @@ def _api_key_provider_api_mode(provider: str, model_cfg: Dict[str, Any], api_key def _maybe_apply_codex_app_server_runtime(*, provider: str, api_mode: str, model_cfg: Optional[Dict[str, Any]]) -> str: """Opt-in rewrite to "codex_app_server" via ``model.openai_runtime``; only ``openai`` / ``openai-codex`` are eligible. No-op when unset, "auto", or empty.""" - if ( - model_cfg - and provider in {"openai", "openai-codex"} - and str(model_cfg.get("openai_runtime") or "").strip().lower() == "codex_app_server" - ): + if model_cfg and provider in {"openai", "openai-codex"} and str(model_cfg.get("openai_runtime") or "").strip().lower() == "codex_app_server": return "codex_app_server" return api_mode @@ -477,9 +473,7 @@ def _pool_entry_mode_and_url(provider, entry, model_cfg, effective_model, base_u base_url = _config_base_url_for_provider(model_cfg, "azure-foundry") or base_url api_mode = _parse_api_mode(model_cfg.get("api_mode")) or api_mode api_mode = _azure_inferred_api_mode(effective_model, api_mode) - if api_mode == "anthropic_messages": - base_url = re.sub(r"/v1/?$", "", base_url) - return api_mode, base_url + return api_mode, (re.sub(r"/v1/?$", "", base_url) if api_mode == "anthropic_messages" else base_url) # Honour model.base_url only when the pool entry carries no explicit base_url (i.e. it fell # back to the registry default). Env var overrides win. pconfig = PROVIDER_REGISTRY.get(provider) @@ -505,7 +499,8 @@ def _openrouter_should_use_pool(requested_provider, model_cfg, explicit_api_key, cfg_base_url = str(model_cfg.get("base_url") or "").strip() env_openai_base_url = _getenv("OPENAI_BASE_URL", "").strip() env_openrouter_base_url = _getenv("OPENROUTER_BASE_URL", "").strip() - has_custom_endpoint = bool(explicit_base_url or env_openai_base_url or env_openrouter_base_url) or bool(cfg_base_url and _cfg_provider(model_cfg) in {"auto", "custom"}) + has_custom_endpoint = bool(explicit_base_url or env_openai_base_url or env_openrouter_base_url) or bool( + cfg_base_url and _cfg_provider(model_cfg) in {"auto", "custom"}) return requested_provider in {"openrouter", "auto"} and not has_custom_endpoint and not bool(explicit_api_key or explicit_base_url) @@ -580,11 +575,8 @@ def _explicit_codex(requested_provider, model_cfg, api_key, explicit_base_url, t def _explicit_nous(requested_provider, model_cfg, api_key, explicit_base_url, target_model): state = auth_mod.get_provider_auth_state("nous") or {} - base_url = ( - explicit_base_url - or (_nous_inference_env_override() or "") - or str(state.get("inference_base_url") or auth_mod.DEFAULT_NOUS_INFERENCE_URL).strip().rstrip("/") - ) + base_url = (explicit_base_url or _nous_inference_env_override() + or str(state.get("inference_base_url") or auth_mod.DEFAULT_NOUS_INFERENCE_URL).strip().rstrip("/")) # The agent_key compatibility field is used for inference only when it holds a NAS invoke JWT; # raw OAuth access_token fallback is handled by resolve_nous_runtime_credentials(). api_key = api_key or (str(state.get("agent_key") or "").strip() if _agent_key_is_usable(state, _nous_min_key_ttl()) else "") @@ -697,9 +689,7 @@ def _resolve_oauth_runtime(provider, requested_provider, model_cfg, target_model raise logger.info("%s; falling through to next provider.", spec.failure_msg) return None - api_mode = spec.api_mode - if callable(api_mode): - api_mode = api_mode(_effective_model(model_cfg, target_model)) + api_mode = spec.api_mode(_effective_model(model_cfg, target_model)) if callable(spec.api_mode) else spec.api_mode return _runtime(provider, api_mode, (creds.get("base_url") or "").rstrip("/") or spec.default_base_url, creds.get("api_key", ""), source=creds.get("source", spec.default_source), **{spec.expiry_key: creds.get(spec.expiry_key)}, requested_provider=requested_provider) @@ -726,11 +716,8 @@ def _azure_anthropic_env_key(model_cfg: Dict[str, Any]) -> str: token = _getenv(env_var, "").strip() if token: return token - return ( - str(model_cfg.get("api_key") or "").strip() - or _getenv("AZURE_ANTHROPIC_KEY", "").strip() - or _getenv("ANTHROPIC_API_KEY", "").strip() - ) + return (str(model_cfg.get("api_key") or "").strip() or _getenv("AZURE_ANTHROPIC_KEY", "").strip() + or _getenv("ANTHROPIC_API_KEY", "").strip()) def _anthropic_env_runtime(requested_provider: str, model_cfg: Dict[str, Any]) -> Dict[str, Any]: From dfa6bc0103ed675e0917e61cf88ab135d8a2c032 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 23:03:08 -0700 Subject: [PATCH 16/19] =?UTF-8?q?refactor(hermes=5Fcli):=20custom/provider?= =?UTF-8?q?s=20=E2=80=94=20fold=20isinstance=20guards=20and=20identity=20l?= =?UTF-8?q?ookup=20chain?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/providers.py | 8 ++----- hermes_cli/runtime_provider_custom.py | 33 +++++++++------------------ 2 files changed, 13 insertions(+), 28 deletions(-) diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py index e394dc8f69..3e91e69f14 100644 --- a/hermes_cli/providers.py +++ b/hermes_cli/providers.py @@ -332,9 +332,7 @@ def _user_pdef(pid: str, name: str, base_url: str, key_env: str, transport: str def resolve_user_provider(name: str, user_config: Dict[str, Any]) -> Optional[ProviderDef]: """Resolve a provider from the user's config.yaml ``providers:`` section.""" - if not user_config or not isinstance(user_config, dict): - return None - entry = user_config.get(name) + entry = user_config.get(name) if isinstance(user_config, dict) and user_config else None if not isinstance(entry, dict): return None return _user_pdef(name, entry.get("name", "") or name, @@ -372,10 +370,8 @@ def resolve_custom_provider(name: str, custom_providers: Optional[List[Dict[str, """Resolve a provider from the user's config.yaml ``custom_providers`` list. A stored bare ``"custom"`` (corrupt state from a prior model-switch bug) falls back to the first valid entry so existing configs self-heal.""" - if not custom_providers or not isinstance(custom_providers, list): - return None requested = (name or "").strip().lower() - if not requested: + if not requested or not custom_providers or not isinstance(custom_providers, list): return None first_valid: Optional[ProviderDef] = None for entry in custom_providers: diff --git a/hermes_cli/runtime_provider_custom.py b/hermes_cli/runtime_provider_custom.py index 0e6d31e4f3..48f9f64428 100644 --- a/hermes_cli/runtime_provider_custom.py +++ b/hermes_cli/runtime_provider_custom.py @@ -153,9 +153,7 @@ def _match_new_style_provider(requested_norm: str, providers: Dict[str, Any]) -> def _match_legacy_custom_provider(requested_norm: str, custom_providers) -> Optional[Dict[str, Any]]: """Scan the legacy ``custom_providers:`` list for ``requested_norm``.""" for entry in custom_providers: - if not isinstance(entry, dict): - continue - name, base_url = entry.get("name"), entry.get("base_url") + name, base_url = (entry.get("name"), entry.get("base_url")) if isinstance(entry, dict) else (None, None) if not isinstance(name, str) or not isinstance(base_url, str): continue provider_key = _clean(entry.get("provider_key", "")) @@ -178,10 +176,9 @@ def _get_named_custom_provider(requested_provider: str) -> Optional[Dict[str, An rp = _rp() config = rp.load_config() providers = config.get("providers") - if isinstance(providers, dict): - found = _match_new_style_provider(requested_norm, providers) - if found: - return found + found = _match_new_style_provider(requested_norm, providers) if isinstance(providers, dict) else None + if found: + return found if isinstance(config.get("custom_providers"), dict): logger.warning("custom_providers in config.yaml is a dict, not a list. " "Each entry must be prefixed with '-' in YAML. " @@ -221,9 +218,7 @@ def _find_custom_identity(matches: Callable[[Dict[str, Any]], bool]) -> Optional except Exception: custom_providers = None for entry in custom_providers or []: - if not isinstance(entry, dict): - continue - name = entry.get("name") + name = entry.get("name") if isinstance(entry, dict) else None if isinstance(name, str) and name.strip() and matches(entry): return custom_provider_slug(name, str(entry.get("provider_key", "") or "")) return None @@ -258,10 +253,8 @@ def find_custom_provider_identity_by_model(model: str) -> Optional[str]: if isinstance(models, dict): return any(str(mid).strip().lower() == target for mid in models) if isinstance(models, list): - return any( - _model_id_matches(item.get("id") or item.get("name") if isinstance(item, dict) else item, target) - for item in models - ) + return any(_model_id_matches(item.get("id") or item.get("name") if isinstance(item, dict) else item, target) + for item in models) return False return _find_custom_identity(_entry_serves_model) @@ -277,14 +270,10 @@ def canonical_custom_identity(*, base_url: Optional[str] = None, config_provider (3) the configured provider (arg, ``model.provider``, ``HERMES_INFERENCE_PROVIDER``) when it names a real entry.""" rp = _rp() - if base_url: - identity = find_custom_provider_identity(base_url) - if identity: - return identity - if model: - identity = find_custom_provider_identity_by_model(model) - if identity: - return identity + identity = (find_custom_provider_identity(base_url) if base_url else None) or ( + find_custom_provider_identity_by_model(model) if model else None) + if identity: + return identity candidate = str(config_provider or "").strip() if not candidate: try: From b58a88d6f452d264ab0ef14b4dd08cc4bd7c79d7 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 23:06:19 -0700 Subject: [PATCH 17/19] =?UTF-8?q?refactor(hermes=5Fcli):=20runtime=5Fprovi?= =?UTF-8?q?der=20=E2=80=94=20whole=20ladder=20as=20one=20lazy=20rung=20gen?= =?UTF-8?q?erator?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/runtime_provider.py | 37 +++++++++++++++++----------------- 1 file changed, 18 insertions(+), 19 deletions(-) diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index ee7d1e51a1..d631cc0548 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -641,10 +641,10 @@ def _resolve_explicit_runtime(*, provider: str, requested_provider: str, model_c if resolver is not None: return resolver(requested_provider, model_cfg, explicit_api_key, explicit_base_url, target_model) pconfig = PROVIDER_REGISTRY.get(provider) - if pconfig and pconfig.auth_type == "api_key": - return _explicit_api_key_provider(provider, pconfig, requested_provider, model_cfg, explicit_api_key, - explicit_base_url, target_model) - return None + if not (pconfig and pconfig.auth_type == "api_key"): + return None + return _explicit_api_key_provider(provider, pconfig, requested_provider, model_cfg, explicit_api_key, explicit_base_url, + target_model) # ── OAuth / auth-store providers ─────────────────────────────────────────────────────────── @@ -670,8 +670,7 @@ _OAUTH_RUNTIME_PROVIDERS: Dict[str, _OAuthRuntimeSpec] = { "openai-codex": _OAuthRuntimeSpec(lambda: resolve_codex_runtime_credentials(), "codex_responses", "hermes-auth-store", "last_refresh", "Auto-detected Codex provider but credentials failed"), "xai-oauth": _OAuthRuntimeSpec(lambda: resolve_xai_oauth_runtime_credentials(), "codex_responses", "hermes-auth-store", - "last_refresh", "Auto-detected xAI OAuth provider but credentials failed", - default_base_url=DEFAULT_XAI_OAUTH_BASE_URL), + "last_refresh", "Auto-detected xAI OAuth provider but credentials failed", DEFAULT_XAI_OAUTH_BASE_URL), "qwen-oauth": _OAuthRuntimeSpec(lambda: resolve_qwen_runtime_credentials(), "chat_completions", "qwen-cli", "expires_at_ms", "Qwen OAuth credentials failed"), } @@ -870,25 +869,25 @@ def resolve_runtime_provider(*, requested: Optional[str] = None, explicit_api_ke OpenCode Zen/Go where different models route through different API surfaces).""" requested_provider = resolve_requested_provider(requested) _raise_if_provider_disabled(requested_provider) - runtime = _resolve_requested_shortcuts(requested_provider, explicit_api_key, explicit_base_url, target_model) - if runtime: - return runtime + return next(r for r in _ladder_rungs(requested_provider, explicit_api_key, explicit_base_url, target_model) if r) + + +def _named_custom_rung(requested_provider, explicit_api_key, explicit_base_url, target_model): runtime = _resolve_named_custom_runtime(requested_provider=requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, target_model=target_model) if runtime: runtime["requested_provider"] = requested_provider - return runtime + return runtime + + +def _ladder_rungs(requested_provider, explicit_api_key, explicit_base_url, target_model): + """Ladder rungs 2-8, yielded lazily so each is evaluated only when the previous one returned + nothing; the last rung (OpenRouter / bare-custom fallback) always yields a runtime.""" + yield _resolve_requested_shortcuts(requested_provider, explicit_api_key, explicit_base_url, target_model) + yield _named_custom_rung(requested_provider, explicit_api_key, explicit_base_url, target_model) if not explicit_base_url and not explicit_api_key: - runtime = _local_endpoint_bypass(requested_provider, explicit_api_key, explicit_base_url) - if runtime: - return runtime + yield _local_endpoint_bypass(requested_provider, explicit_api_key, explicit_base_url) provider = resolve_provider(requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url) - return next(r for r in _provider_rungs(provider, requested_provider, explicit_api_key, explicit_base_url, target_model) if r) - - -def _provider_rungs(provider, requested_provider, explicit_api_key, explicit_base_url, target_model): - """Rungs 5-8 of the ladder, yielded lazily so each is evaluated only when the previous one - returned nothing; the last rung (OpenRouter / bare-custom fallback) always yields a runtime.""" model_cfg = _get_model_config() yield _opencode_free_runtime(provider, requested_provider, model_cfg, target_model) yield _resolve_explicit_runtime(provider=provider, requested_provider=requested_provider, model_cfg=model_cfg, From c2fae8796e993390e4b6d039d8eb67e7fc2323af Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 23:09:20 -0700 Subject: [PATCH 18/19] =?UTF-8?q?refactor(hermes=5Fcli):=20runtime=5Fprovi?= =?UTF-8?q?der=20=E2=80=94=20collapse=20redundant=20azure=20host=20check,?= =?UTF-8?q?=20try/except=20returns?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/runtime_provider.py | 14 +++++--------- 1 file changed, 5 insertions(+), 9 deletions(-) diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index d631cc0548..d59ea443d8 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -185,10 +185,8 @@ def _copilot_runtime_api_mode(model_cfg: Dict[str, Any], api_key: str, *, target # Use the model being resolved, not the persisted default: a Claude MoA slot inheriting # codex_responses from a GPT-5 default fails with "model ... does not support Responses API". model_name = str(_effective_model(model_cfg, target_model)).strip() - if not model_name: - return "chat_completions" try: - return _models.copilot_model_api_mode(model_name, api_key=api_key) + return _models.copilot_model_api_mode(model_name, api_key=api_key) if model_name else "chat_completions" except Exception: return "chat_completions" @@ -199,10 +197,9 @@ def _azure_inferred_api_mode(effective_model: str, api_mode: str) -> str: if not effective_model or api_mode == "anthropic_messages": return api_mode try: - inferred = _models.azure_foundry_model_api_mode(effective_model) + return _models.azure_foundry_model_api_mode(effective_model) or api_mode except Exception: - inferred = None - return inferred or api_mode + return api_mode def _configured_or_fallback_api_mode(provider: str, model_cfg: Dict[str, Any], base_url: str, effective_model: Any, *, @@ -722,11 +719,10 @@ def _azure_anthropic_env_key(model_cfg: Dict[str, Any]) -> str: def _anthropic_env_runtime(requested_provider: str, model_cfg: Dict[str, Any]) -> Dict[str, Any]: """Native Anthropic (Messages API) from env/auth store; ``model.base_url`` honoured only when the configured provider is anthropic (else a Codex endpoint would leak into Anthropic requests).""" - cfg_base_url = _anthropic_cfg_base_url(model_cfg) - base_url = cfg_base_url or _ANTHROPIC_DEFAULT_BASE_URL + base_url = _anthropic_cfg_base_url(model_cfg) or _ANTHROPIC_DEFAULT_BASE_URL # Microsoft Foundry endpoints reject Claude Code OAuth tokens, which resolve_anthropic_token() # would return first — use the env key directly. - if base_url_host_matches(base_url, "azure.com") or (cfg_base_url and base_url_host_matches(cfg_base_url, "azure.com")): + if base_url_host_matches(base_url, "azure.com"): token = _azure_anthropic_env_key(model_cfg) if not token: raise AuthError("No Azure Anthropic API key found. Set AZURE_ANTHROPIC_KEY or " From 8f800b3a9225680f86758ee7a662bed3b26b68e0 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 23:10:14 -0700 Subject: [PATCH 19/19] =?UTF-8?q?refactor(hermes=5Fcli):=20runtime=5Fprovi?= =?UTF-8?q?der=20=E2=80=94=20pack=20re-export=20import=20lists?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/runtime_provider.py | 30 +++++++----------------------- 1 file changed, 7 insertions(+), 23 deletions(-) diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index d59ea443d8..7d70ceb8cc 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -410,31 +410,15 @@ def resolve_requested_provider(requested: Optional[str] = None) -> str: # ── extracted collaborators (re-exported; see module docstring) ──────────────────────────── from hermes_cli.runtime_provider_custom import ( # noqa: E402,F401 - _apply_custom_provider_extras, - _custom_provider_request_overrides, - _filter_capabilities, - _find_custom_identity, - _get_named_custom_provider, - _lift_common_custom_fields, - _lift_extra_headers, - _lift_max_output_tokens, - _lift_model_capabilities, - _normalize_base_url_for_match, - _normalize_custom_provider_name, - _resolve_named_custom_runtime, - _try_resolve_from_custom_pool, - canonical_custom_identity, - find_custom_provider_identity, - find_custom_provider_identity_by_model, - has_named_custom_provider, - is_routable_provider, + _apply_custom_provider_extras, _custom_provider_request_overrides, _filter_capabilities, _find_custom_identity, + _get_named_custom_provider, _lift_common_custom_fields, _lift_extra_headers, _lift_max_output_tokens, + _lift_model_capabilities, _normalize_base_url_for_match, _normalize_custom_provider_name, _resolve_named_custom_runtime, + _try_resolve_from_custom_pool, canonical_custom_identity, find_custom_provider_identity, + find_custom_provider_identity_by_model, has_named_custom_provider, is_routable_provider, ) from hermes_cli.runtime_provider_backends import ( # noqa: E402,F401 - _is_external_process_provider, - _resolve_azure_foundry_runtime, - _resolve_bedrock_runtime, - _resolve_external_process_runtime, - _resolve_openrouter_runtime, + _is_external_process_provider, _resolve_azure_foundry_runtime, _resolve_bedrock_runtime, + _resolve_external_process_runtime, _resolve_openrouter_runtime, )