The free tier depends on the account service (NAS) and the welcome inference host, and Hermes had no honest answer for most of the ways either can refuse or fail: the NAS codes it matched were never sent, the tier-dark 403 carried no message to match, a single boot-time blip disabled minting for the whole process, and a structured rate-limit refusal never reached the cross-session guard, so the "sign in for a bigger allowance" prompt was dead code. Backend - anon_auth: classify what NAS actually sends (404 not_found, 503 temporarily_disabled, 429 + Retry-After, 428 pow_*, 403 account_locked) into one ANON_* code each, carrying retry_after / retryable on AuthError. - Replace the process-lifetime mint memo with a per-profile cooldown that honours the server's wait, climbs a short ladder when the service is unreachable, never retries terminal codes, and yields to the user's own retry (force=True). - Bootstrap record carries error_code / retryable / retry_after; a bounded background loop retries transient failures and re-announces setup.ready. setup.status and free_tier.status expose the block; free_tier.provision is the forced retry. - Inference: a generic 403 from a welcome host is the tier refusing (keyed on the route); model_not_free moves onto the gateway's alternate once; anon_on_paid_host re-reads the route once; a long rate_limited refusal trips the cross-session guard; a locked account is retired but never replaced; terminal copy on the free route is one plain sentence. - Sign-in: Failed keeps the service's code and wait; account_busy is retryable; the OAuth poll reports retryable / retry_after. - All user-facing copy rewritten for first-time users: never "the free service is off" (what is unavailable is using Hermes without signing in, and signing in is free), no jargon, spoken waits. Desktop - A setup-failure notice above the provider picker: one sentence per code, a retry when the backend says one can work, the sign-in pointer only when the account service answered at all. The overlay re-checks readiness on setup.ready so a background success dismisses it. - Sign-in dialog gains busy / unreachable / unavailable screens. Rehearsal - scripts/free_tier_fault_server.py stands in for both services with the real wire contract and a CORS-open scenario switch; HERMES_EXTRA_WELCOME_HOSTS (dev-only, env-only) lets the route rules treat it as the welcome host. Walkthrough in website/docs/developer-guide/free-tier-fault-rehearsal.md. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
184 lines
8.9 KiB
Python
184 lines
8.9 KiB
Python
"""Shared constants, the lazy ``httpx`` proxy and :class:`AuthError` for the auth package.
|
|
|
|
Pure leaf: imports nothing from ``hermes_cli.auth`` so the per-provider modules
|
|
(``auth_nous``, ``auth_codex``, ...) can import it at module scope without cycles."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import base64
|
|
import json
|
|
from typing import Any, Callable, Dict, Optional
|
|
|
|
# httpx is imported lazily (~30ms) because hermes_cli.auth is on the interactive-CLI startup path
|
|
# (credential_pool -> auxiliary_client -> cli_commands_mixin). The proxy resolves on first attribute
|
|
# access; ``from __future__ import annotations`` keeps ``httpx.Client`` annotations unevaluated.
|
|
import importlib as _importlib
|
|
from typing import TYPE_CHECKING
|
|
|
|
if TYPE_CHECKING:
|
|
import httpx
|
|
else:
|
|
class _LazyHttpx:
|
|
__slots__ = ("_mod",)
|
|
|
|
def __init__(self) -> None:
|
|
object.__setattr__(self, "_mod", None)
|
|
|
|
def _resolve(self):
|
|
mod = object.__getattribute__(self, "_mod")
|
|
if mod is None:
|
|
mod = _importlib.import_module("httpx")
|
|
object.__setattr__(self, "_mod", mod)
|
|
return mod
|
|
|
|
def __getattr__(self, name):
|
|
return getattr(self._resolve(), name)
|
|
|
|
# set/del forward to the real module so monkeypatch.setattr("hermes_cli.auth.httpx.Client")
|
|
# keeps working in tests.
|
|
def __setattr__(self, name, value):
|
|
setattr(self._resolve(), name, value)
|
|
|
|
def __delattr__(self, name):
|
|
delattr(self._resolve(), name)
|
|
|
|
httpx = _LazyHttpx()
|
|
|
|
# ── Constants ───────────────────────────────────────────────────────────────────────────────────────
|
|
|
|
AUTH_STORE_VERSION = 1
|
|
AUTH_LOCK_TIMEOUT_SECONDS = 15.0
|
|
|
|
# Nous Portal defaults
|
|
DEFAULT_NOUS_PORTAL_URL = "https://portal.nousresearch.com"
|
|
DEFAULT_NOUS_INFERENCE_URL = "https://inference-api.nousresearch.com/v1"
|
|
# The free tier's (anonymous account) inference host. NAS hands it to the client on every token
|
|
# exchange (``inference_base_url``); this literal is the fallback when that field is absent or fails
|
|
# the host allowlist, because the paid host cross-refuses an anonymous JWT with a 400.
|
|
DEFAULT_NOUS_WELCOME_URL = "https://welcome-api.nousresearch.com/v1"
|
|
DEFAULT_NOUS_CLIENT_ID = "hermes-cli"
|
|
NOUS_INFERENCE_INVOKE_SCOPE = "inference:invoke"
|
|
NOUS_BILLING_MANAGE_SCOPE = "billing:manage"
|
|
DEFAULT_NOUS_SCOPE = NOUS_INFERENCE_INVOKE_SCOPE
|
|
NOUS_DEVICE_CODE_SOURCE = "device_code"
|
|
NOUS_AUTH_PATH_INVOKE_JWT = "invoke_jwt"
|
|
ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 120 # refresh 2 min before expiry
|
|
NOUS_INVOKE_JWT_MIN_TTL_SECONDS = ACCESS_TOKEN_REFRESH_SKEW_SECONDS
|
|
DEVICE_AUTH_POLL_INTERVAL_CAP_SECONDS = 1 # poll at most every 1s
|
|
DEVICE_CODE_GRANT_TYPE = "urn:ietf:params:oauth:grant-type:device_code"
|
|
_FORM_JSON_HEADERS = {"Content-Type": "application/x-www-form-urlencoded", "Accept": "application/json"}
|
|
DEFAULT_CODEX_BASE_URL = "https://chatgpt.com/backend-api/codex"
|
|
DEFAULT_XAI_OAUTH_BASE_URL = "https://api.x.ai/v1"
|
|
MINIMAX_OAUTH_CLIENT_ID = "78257093-7e40-4613-99e0-527b14b39113"
|
|
MINIMAX_OAUTH_SCOPE = "group_id profile model.completion"
|
|
MINIMAX_OAUTH_GRANT_TYPE = "urn:ietf:params:oauth:grant-type:user_code"
|
|
MINIMAX_OAUTH_GLOBAL_BASE = "https://api.minimax.io"
|
|
MINIMAX_OAUTH_CN_BASE = "https://api.minimaxi.com"
|
|
MINIMAX_OAUTH_GLOBAL_INFERENCE = "https://api.minimax.io/anthropic"
|
|
MINIMAX_OAUTH_CN_INFERENCE = "https://api.minimaxi.com/anthropic"
|
|
MINIMAX_OAUTH_REFRESH_SKEW_SECONDS = 60
|
|
DEFAULT_QWEN_BASE_URL = "https://portal.qwen.ai/v1"
|
|
DEFAULT_GITHUB_MODELS_BASE_URL = "https://api.githubcopilot.com"
|
|
DEFAULT_COPILOT_ACP_BASE_URL = "acp://copilot"
|
|
DEFAULT_OLLAMA_CLOUD_BASE_URL = "https://ollama.com/v1"
|
|
DEFAULT_ACTUAL_BASE_URL = "https://api.actual.inc/v1"
|
|
DEFAULT_ACTUAL_LOCAL_BASE_URL = "http://127.0.0.1:8080/v1"
|
|
STEPFUN_STEP_PLAN_INTL_BASE_URL = "https://api.stepfun.ai/step_plan/v1"
|
|
STEPFUN_STEP_PLAN_CN_BASE_URL = "https://api.stepfun.com/step_plan/v1"
|
|
CODEX_OAUTH_CLIENT_ID = "app_EMoamEEZ73f0CkXaXp7hrann"
|
|
CODEX_OAUTH_TOKEN_URL = "https://auth.openai.com/oauth/token"
|
|
try: # Version tag for the Codex token-endpoint User-Agent; fall back if unavailable.
|
|
from hermes_cli import __version__ as _HERMES_CLI_VERSION
|
|
except Exception: # pragma: no cover - version import should always succeed
|
|
_HERMES_CLI_VERSION = "unknown"
|
|
CODEX_OAUTH_USER_AGENT = f"hermes-cli/{_HERMES_CLI_VERSION}"
|
|
CODEX_ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 120
|
|
XAI_OAUTH_ISSUER = "https://auth.x.ai"
|
|
XAI_OAUTH_DISCOVERY_URL = f"{XAI_OAUTH_ISSUER}/.well-known/openid-configuration"
|
|
XAI_OAUTH_CLIENT_ID = "b1a00492-073a-47ea-816f-4c329264a828"
|
|
XAI_OAUTH_SCOPE = "openid profile email offline_access grok-cli:access api:access"
|
|
XAI_OAUTH_DEVICE_CODE_URL = f"{XAI_OAUTH_ISSUER}/oauth2/device/code"
|
|
# xAI/Grok OAuth access tokens are short-lived (~6h). A two-minute refresh window leaves noisy
|
|
# credential-expiry gaps for gateway/cron workloads that touch the provider every ~30 min, so refresh
|
|
# up to an hour early.
|
|
XAI_ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 3600
|
|
QWEN_OAUTH_CLIENT_ID = "f0304373b74a44d2b584a3fb70ca9e56"
|
|
QWEN_OAUTH_TOKEN_URL = "https://chat.qwen.ai/api/v1/oauth2/token"
|
|
QWEN_ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 120
|
|
DEFAULT_SPOTIFY_ACCOUNTS_BASE_URL = "https://accounts.spotify.com"
|
|
DEFAULT_SPOTIFY_API_BASE_URL = "https://api.spotify.com/v1"
|
|
DEFAULT_SPOTIFY_REDIRECT_URI = "http://127.0.0.1:43827/spotify/callback"
|
|
SPOTIFY_DOCS_URL = "https://hermes-agent.nousresearch.com/docs/user-guide/features/spotify"
|
|
SPOTIFY_DASHBOARD_URL = "https://developer.spotify.com/dashboard"
|
|
SPOTIFY_ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 120
|
|
# OpenRouter PKCE (https://openrouter.ai/docs/guides/overview/auth/oauth): the "token" endpoint
|
|
# mints a plain user-controlled API key; there is no refresh token.
|
|
OPENROUTER_AUTH_URL = "https://openrouter.ai/auth"
|
|
OPENROUTER_AUTH_KEYS_URL = "https://openrouter.ai/api/v1/auth/keys"
|
|
OPENROUTER_OAUTH_DOCS_URL = "https://openrouter.ai/docs/guides/overview/auth/oauth"
|
|
|
|
OAUTH_OVER_SSH_DOCS_URL = "https://hermes-agent.nousresearch.com/docs/guides/oauth-over-ssh"
|
|
DEFAULT_SPOTIFY_SCOPE = " ".join((
|
|
"user-modify-playback-state", "user-read-playback-state", "user-read-currently-playing",
|
|
"user-read-recently-played", "playlist-read-private", "playlist-read-collaborative",
|
|
"playlist-modify-public", "playlist-modify-private", "user-library-read", "user-library-modify",
|
|
))
|
|
SERVICE_PROVIDER_NAMES: Dict[str, str] = {"spotify": "Spotify"}
|
|
|
|
# LM Studio's default no-auth mode still needs *some* non-empty bearer for the API-key code paths to
|
|
# treat the provider as configured. Sent only to LM Studio, never to a remote service.
|
|
LMSTUDIO_NOAUTH_PLACEHOLDER = "dummy-lm-api-key"
|
|
ACTUAL_LOCAL_NOAUTH_PLACEHOLDER = "dummy-actual-local-api-key"
|
|
|
|
# Upstream rate-limit / usage-quota exhaustion (HTTP 429): transient, re-authenticating cannot resolve
|
|
# it, so it must stay distinct from missing/expired-credential errors.
|
|
CODEX_RATE_LIMITED_CODE = "codex_rate_limited"
|
|
|
|
|
|
class AuthError(RuntimeError):
|
|
"""Structured auth error with UX mapping hints."""
|
|
|
|
def __init__(
|
|
self, message: str, *, provider: str = "", code: Optional[str] = None, relogin_required: bool = False,
|
|
retry_after: Optional[float] = None, retryable: Optional[bool] = None,
|
|
) -> None:
|
|
super().__init__(message)
|
|
self.provider = provider
|
|
self.code = code
|
|
self.relogin_required = relogin_required
|
|
# Optional wait hint in seconds (a server ``Retry-After`` or a client cooldown) and whether a
|
|
# later attempt can succeed at all. None = the raiser did not say; callers treat None as
|
|
# "retryable, no hint" for transport-shaped errors and as terminal for auth refusals.
|
|
self.retry_after = retry_after
|
|
self.retryable = retryable
|
|
|
|
|
|
def _provider_error_factory(provider: str) -> Callable[..., AuthError]:
|
|
def factory(message: str, code: Optional[str] = None, *, relogin: bool = False) -> AuthError:
|
|
return AuthError(message, provider=provider, code=code, relogin_required=relogin)
|
|
|
|
return factory
|
|
|
|
|
|
# Per-provider AuthError constructors: ``_xai_err(message, code, relogin=True)``.
|
|
_nous_err = _provider_error_factory("nous")
|
|
_xai_err = _provider_error_factory("xai-oauth")
|
|
_codex_err = _provider_error_factory("openai-codex")
|
|
_spotify_err = _provider_error_factory("spotify")
|
|
_qwen_err = _provider_error_factory("qwen-oauth")
|
|
_minimax_err = _provider_error_factory("minimax-oauth")
|
|
_openrouter_err = _provider_error_factory("openrouter")
|
|
|
|
|
|
def _decode_jwt_claims(token: Any) -> Dict[str, Any]:
|
|
if not isinstance(token, str) or token.count(".") != 2:
|
|
return {}
|
|
payload = token.split(".")[1]
|
|
payload += "=" * ((4 - len(payload) % 4) % 4)
|
|
try:
|
|
raw = base64.urlsafe_b64decode(payload.encode("utf-8"))
|
|
claims = json.loads(raw.decode("utf-8"))
|
|
except Exception:
|
|
return {}
|
|
return claims if isinstance(claims, dict) else {}
|