Files
hermes-agent/hermes_cli/auth_constants.py
Robin Fernandes 51e39af967 feat(free-tier): ruled behaviour for every welcome-api failure, with friendly copy and a fault-injecting rehearsal server
The free tier depends on the account service (NAS) and the welcome inference
host, and Hermes had no honest answer for most of the ways either can refuse
or fail: the NAS codes it matched were never sent, the tier-dark 403 carried
no message to match, a single boot-time blip disabled minting for the whole
process, and a structured rate-limit refusal never reached the cross-session
guard, so the "sign in for a bigger allowance" prompt was dead code.

Backend
- anon_auth: classify what NAS actually sends (404 not_found, 503
  temporarily_disabled, 429 + Retry-After, 428 pow_*, 403 account_locked)
  into one ANON_* code each, carrying retry_after / retryable on AuthError.
- Replace the process-lifetime mint memo with a per-profile cooldown that
  honours the server's wait, climbs a short ladder when the service is
  unreachable, never retries terminal codes, and yields to the user's own
  retry (force=True).
- Bootstrap record carries error_code / retryable / retry_after; a bounded
  background loop retries transient failures and re-announces setup.ready.
  setup.status and free_tier.status expose the block; free_tier.provision is
  the forced retry.
- Inference: a generic 403 from a welcome host is the tier refusing (keyed on
  the route); model_not_free moves onto the gateway's alternate once;
  anon_on_paid_host re-reads the route once; a long rate_limited refusal
  trips the cross-session guard; a locked account is retired but never
  replaced; terminal copy on the free route is one plain sentence.
- Sign-in: Failed keeps the service's code and wait; account_busy is
  retryable; the OAuth poll reports retryable / retry_after.
- All user-facing copy rewritten for first-time users: never "the free
  service is off" (what is unavailable is using Hermes without signing in,
  and signing in is free), no jargon, spoken waits.

Desktop
- A setup-failure notice above the provider picker: one sentence per code,
  a retry when the backend says one can work, the sign-in pointer only when
  the account service answered at all. The overlay re-checks readiness on
  setup.ready so a background success dismisses it.
- Sign-in dialog gains busy / unreachable / unavailable screens.

Rehearsal
- scripts/free_tier_fault_server.py stands in for both services with the
  real wire contract and a CORS-open scenario switch; HERMES_EXTRA_WELCOME_HOSTS
  (dev-only, env-only) lets the route rules treat it as the welcome host.
  Walkthrough in website/docs/developer-guide/free-tier-fault-rehearsal.md.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-09-15 20:44:42 +05:30

184 lines
8.9 KiB
Python

"""Shared constants, the lazy ``httpx`` proxy and :class:`AuthError` for the auth package.
Pure leaf: imports nothing from ``hermes_cli.auth`` so the per-provider modules
(``auth_nous``, ``auth_codex``, ...) can import it at module scope without cycles."""
from __future__ import annotations
import base64
import json
from typing import Any, Callable, Dict, Optional
# httpx is imported lazily (~30ms) because hermes_cli.auth is on the interactive-CLI startup path
# (credential_pool -> auxiliary_client -> cli_commands_mixin). The proxy resolves on first attribute
# access; ``from __future__ import annotations`` keeps ``httpx.Client`` annotations unevaluated.
import importlib as _importlib
from typing import TYPE_CHECKING
if TYPE_CHECKING:
import httpx
else:
class _LazyHttpx:
__slots__ = ("_mod",)
def __init__(self) -> None:
object.__setattr__(self, "_mod", None)
def _resolve(self):
mod = object.__getattribute__(self, "_mod")
if mod is None:
mod = _importlib.import_module("httpx")
object.__setattr__(self, "_mod", mod)
return mod
def __getattr__(self, name):
return getattr(self._resolve(), name)
# set/del forward to the real module so monkeypatch.setattr("hermes_cli.auth.httpx.Client")
# keeps working in tests.
def __setattr__(self, name, value):
setattr(self._resolve(), name, value)
def __delattr__(self, name):
delattr(self._resolve(), name)
httpx = _LazyHttpx()
# ── Constants ───────────────────────────────────────────────────────────────────────────────────────
AUTH_STORE_VERSION = 1
AUTH_LOCK_TIMEOUT_SECONDS = 15.0
# Nous Portal defaults
DEFAULT_NOUS_PORTAL_URL = "https://portal.nousresearch.com"
DEFAULT_NOUS_INFERENCE_URL = "https://inference-api.nousresearch.com/v1"
# The free tier's (anonymous account) inference host. NAS hands it to the client on every token
# exchange (``inference_base_url``); this literal is the fallback when that field is absent or fails
# the host allowlist, because the paid host cross-refuses an anonymous JWT with a 400.
DEFAULT_NOUS_WELCOME_URL = "https://welcome-api.nousresearch.com/v1"
DEFAULT_NOUS_CLIENT_ID = "hermes-cli"
NOUS_INFERENCE_INVOKE_SCOPE = "inference:invoke"
NOUS_BILLING_MANAGE_SCOPE = "billing:manage"
DEFAULT_NOUS_SCOPE = NOUS_INFERENCE_INVOKE_SCOPE
NOUS_DEVICE_CODE_SOURCE = "device_code"
NOUS_AUTH_PATH_INVOKE_JWT = "invoke_jwt"
ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 120 # refresh 2 min before expiry
NOUS_INVOKE_JWT_MIN_TTL_SECONDS = ACCESS_TOKEN_REFRESH_SKEW_SECONDS
DEVICE_AUTH_POLL_INTERVAL_CAP_SECONDS = 1 # poll at most every 1s
DEVICE_CODE_GRANT_TYPE = "urn:ietf:params:oauth:grant-type:device_code"
_FORM_JSON_HEADERS = {"Content-Type": "application/x-www-form-urlencoded", "Accept": "application/json"}
DEFAULT_CODEX_BASE_URL = "https://chatgpt.com/backend-api/codex"
DEFAULT_XAI_OAUTH_BASE_URL = "https://api.x.ai/v1"
MINIMAX_OAUTH_CLIENT_ID = "78257093-7e40-4613-99e0-527b14b39113"
MINIMAX_OAUTH_SCOPE = "group_id profile model.completion"
MINIMAX_OAUTH_GRANT_TYPE = "urn:ietf:params:oauth:grant-type:user_code"
MINIMAX_OAUTH_GLOBAL_BASE = "https://api.minimax.io"
MINIMAX_OAUTH_CN_BASE = "https://api.minimaxi.com"
MINIMAX_OAUTH_GLOBAL_INFERENCE = "https://api.minimax.io/anthropic"
MINIMAX_OAUTH_CN_INFERENCE = "https://api.minimaxi.com/anthropic"
MINIMAX_OAUTH_REFRESH_SKEW_SECONDS = 60
DEFAULT_QWEN_BASE_URL = "https://portal.qwen.ai/v1"
DEFAULT_GITHUB_MODELS_BASE_URL = "https://api.githubcopilot.com"
DEFAULT_COPILOT_ACP_BASE_URL = "acp://copilot"
DEFAULT_OLLAMA_CLOUD_BASE_URL = "https://ollama.com/v1"
DEFAULT_ACTUAL_BASE_URL = "https://api.actual.inc/v1"
DEFAULT_ACTUAL_LOCAL_BASE_URL = "http://127.0.0.1:8080/v1"
STEPFUN_STEP_PLAN_INTL_BASE_URL = "https://api.stepfun.ai/step_plan/v1"
STEPFUN_STEP_PLAN_CN_BASE_URL = "https://api.stepfun.com/step_plan/v1"
CODEX_OAUTH_CLIENT_ID = "app_EMoamEEZ73f0CkXaXp7hrann"
CODEX_OAUTH_TOKEN_URL = "https://auth.openai.com/oauth/token"
try: # Version tag for the Codex token-endpoint User-Agent; fall back if unavailable.
from hermes_cli import __version__ as _HERMES_CLI_VERSION
except Exception: # pragma: no cover - version import should always succeed
_HERMES_CLI_VERSION = "unknown"
CODEX_OAUTH_USER_AGENT = f"hermes-cli/{_HERMES_CLI_VERSION}"
CODEX_ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 120
XAI_OAUTH_ISSUER = "https://auth.x.ai"
XAI_OAUTH_DISCOVERY_URL = f"{XAI_OAUTH_ISSUER}/.well-known/openid-configuration"
XAI_OAUTH_CLIENT_ID = "b1a00492-073a-47ea-816f-4c329264a828"
XAI_OAUTH_SCOPE = "openid profile email offline_access grok-cli:access api:access"
XAI_OAUTH_DEVICE_CODE_URL = f"{XAI_OAUTH_ISSUER}/oauth2/device/code"
# xAI/Grok OAuth access tokens are short-lived (~6h). A two-minute refresh window leaves noisy
# credential-expiry gaps for gateway/cron workloads that touch the provider every ~30 min, so refresh
# up to an hour early.
XAI_ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 3600
QWEN_OAUTH_CLIENT_ID = "f0304373b74a44d2b584a3fb70ca9e56"
QWEN_OAUTH_TOKEN_URL = "https://chat.qwen.ai/api/v1/oauth2/token"
QWEN_ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 120
DEFAULT_SPOTIFY_ACCOUNTS_BASE_URL = "https://accounts.spotify.com"
DEFAULT_SPOTIFY_API_BASE_URL = "https://api.spotify.com/v1"
DEFAULT_SPOTIFY_REDIRECT_URI = "http://127.0.0.1:43827/spotify/callback"
SPOTIFY_DOCS_URL = "https://hermes-agent.nousresearch.com/docs/user-guide/features/spotify"
SPOTIFY_DASHBOARD_URL = "https://developer.spotify.com/dashboard"
SPOTIFY_ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 120
# OpenRouter PKCE (https://openrouter.ai/docs/guides/overview/auth/oauth): the "token" endpoint
# mints a plain user-controlled API key; there is no refresh token.
OPENROUTER_AUTH_URL = "https://openrouter.ai/auth"
OPENROUTER_AUTH_KEYS_URL = "https://openrouter.ai/api/v1/auth/keys"
OPENROUTER_OAUTH_DOCS_URL = "https://openrouter.ai/docs/guides/overview/auth/oauth"
OAUTH_OVER_SSH_DOCS_URL = "https://hermes-agent.nousresearch.com/docs/guides/oauth-over-ssh"
DEFAULT_SPOTIFY_SCOPE = " ".join((
"user-modify-playback-state", "user-read-playback-state", "user-read-currently-playing",
"user-read-recently-played", "playlist-read-private", "playlist-read-collaborative",
"playlist-modify-public", "playlist-modify-private", "user-library-read", "user-library-modify",
))
SERVICE_PROVIDER_NAMES: Dict[str, str] = {"spotify": "Spotify"}
# LM Studio's default no-auth mode still needs *some* non-empty bearer for the API-key code paths to
# treat the provider as configured. Sent only to LM Studio, never to a remote service.
LMSTUDIO_NOAUTH_PLACEHOLDER = "dummy-lm-api-key"
ACTUAL_LOCAL_NOAUTH_PLACEHOLDER = "dummy-actual-local-api-key"
# Upstream rate-limit / usage-quota exhaustion (HTTP 429): transient, re-authenticating cannot resolve
# it, so it must stay distinct from missing/expired-credential errors.
CODEX_RATE_LIMITED_CODE = "codex_rate_limited"
class AuthError(RuntimeError):
"""Structured auth error with UX mapping hints."""
def __init__(
self, message: str, *, provider: str = "", code: Optional[str] = None, relogin_required: bool = False,
retry_after: Optional[float] = None, retryable: Optional[bool] = None,
) -> None:
super().__init__(message)
self.provider = provider
self.code = code
self.relogin_required = relogin_required
# Optional wait hint in seconds (a server ``Retry-After`` or a client cooldown) and whether a
# later attempt can succeed at all. None = the raiser did not say; callers treat None as
# "retryable, no hint" for transport-shaped errors and as terminal for auth refusals.
self.retry_after = retry_after
self.retryable = retryable
def _provider_error_factory(provider: str) -> Callable[..., AuthError]:
def factory(message: str, code: Optional[str] = None, *, relogin: bool = False) -> AuthError:
return AuthError(message, provider=provider, code=code, relogin_required=relogin)
return factory
# Per-provider AuthError constructors: ``_xai_err(message, code, relogin=True)``.
_nous_err = _provider_error_factory("nous")
_xai_err = _provider_error_factory("xai-oauth")
_codex_err = _provider_error_factory("openai-codex")
_spotify_err = _provider_error_factory("spotify")
_qwen_err = _provider_error_factory("qwen-oauth")
_minimax_err = _provider_error_factory("minimax-oauth")
_openrouter_err = _provider_error_factory("openrouter")
def _decode_jwt_claims(token: Any) -> Dict[str, Any]:
if not isinstance(token, str) or token.count(".") != 2:
return {}
payload = token.split(".")[1]
payload += "=" * ((4 - len(payload) % 4) % 4)
try:
raw = base64.urlsafe_b64decode(payload.encode("utf-8"))
claims = json.loads(raw.decode("utf-8"))
except Exception:
return {}
return claims if isinstance(claims, dict) else {}