Files
hermes-agent/agent/error_classifier.py

1083 lines
51 KiB
Python

"""API error classification for smart failover and recovery.
A priority-ordered pipeline maps an API exception to a ``ClassifiedError``
whose recovery hints (retry, rotate credential, fallback, compress, abort) the
retry loop in run_agent.py consults instead of re-matching strings itself.
"""
from __future__ import annotations
import enum
import json
import logging
from dataclasses import dataclass, field
from typing import Any, Callable, Dict, Iterator, Optional, Sequence
logger = logging.getLogger(__name__)
# Synthetic code for the OpenAI SDK rejecting a provider's SSE ``data:`` field
# before any completion chunk arrives; distinct from generic JSON parse errors.
PROVIDER_STREAM_NON_JSON_ERROR_CODE = "provider_stream_non_json_data"
# ── Error taxonomy ──────────────────────────────────────────────────────
class FailoverReason(enum.Enum):
"""Why an API call failed — determines recovery strategy."""
auth = "auth" # Transient auth (401/403) — refresh/rotate
auth_permanent = "auth_permanent" # Auth failed after refresh — abort
billing = "billing" # 402 or confirmed credit exhaustion — rotate immediately
rate_limit = "rate_limit" # 429 or quota-based throttling — backoff then rotate
upstream_rate_limit = "upstream_rate_limit" # Aggregator's upstream model 429 — fallback model, key is healthy
overloaded = "overloaded" # 503/529 — provider overloaded, backoff
server_error = "server_error" # 500/502 — internal server error, retry
timeout = "timeout" # Connection/read timeout — rebuild client + retry
ssl_cert_verification = "ssl_cert_verification" # Deterministic TLS chain failure — fail fast with guidance
context_overflow = "context_overflow" # Context too large — compress, not failover
payload_too_large = "payload_too_large" # 413 — compress payload
image_too_large = "image_too_large" # Native image part exceeds provider's per-image limit — shrink and retry
image_corrupt = "image_corrupt" # Provider can't decode image bytes — strip and retry (shrinking won't help)
model_not_found = "model_not_found" # 404 or invalid model — fallback to different model
provider_policy_blocked = "provider_policy_blocked" # Aggregator account data/privacy policy excluded the only endpoint
content_policy_blocked = "content_policy_blocked" # Provider safety filter rejected this prompt — don't retry unchanged
format_error = "format_error" # 400 bad request — abort or strip + retry
invalid_encrypted_content = "invalid_encrypted_content" # Responses replay blob rejected — strip replay state and retry
multimodal_tool_content_unsupported = "multimodal_tool_content_unsupported" # Provider rejected list content in tool messages — downgrade to text
thinking_signature = "thinking_signature" # Anthropic thinking block sig invalid
long_context_tier = "long_context_tier" # Anthropic "extra usage" tier gate
oauth_long_context_beta_forbidden = "oauth_long_context_beta_forbidden" # Anthropic OAuth rejects 1M beta — disable beta and retry
llama_cpp_grammar_pattern = "llama_cpp_grammar_pattern" # llama.cpp grammar rejects regex `pattern`/`format` — strip from tools and retry
unknown = "unknown" # Unclassifiable — retry with backoff
@dataclass
class ClassifiedError:
"""Structured classification of an API error with recovery hints."""
reason: FailoverReason
status_code: Optional[int] = None
provider: Optional[str] = None
model: Optional[str] = None
message: str = ""
error_context: Dict[str, Any] = field(default_factory=dict)
# Recovery hints — the retry loop checks these instead of re-classifying.
retryable: bool = True
should_compress: bool = False
should_rotate_credential: bool = False
should_fallback: bool = False
@property
def is_auth(self) -> bool:
return self.reason in {FailoverReason.auth, FailoverReason.auth_permanent}
@property
def billing_unverified(self) -> bool:
"""True when a ``billing`` verdict rests on an ambiguous body (#82154)."""
return bool(self.error_context.get("billing_unverified"))
# ── Pattern tables (lowercased substrings) ──────────────────────────────
# Billing exhaustion (not transient rate limit).
_BILLING_PATTERNS = (
"insufficient credits", "insufficient_quota", "insufficient balance", "credit balance",
"credits exhausted", "credits have been exhausted", "requires available credits",
"account balance is too low", "no usable credits", "top up your credits", "payment required",
"billing hard limit", "exceeded your current quota", "account is deactivated",
"plan does not include",
"out of extra usage", # Anthropic OAuth Pro/Max overage bucket depleted (HTTP 400)
"out of funds", "run out of funds", "balance_depleted",
"model_not_supported_on_free_tier", "not available on the free tier",
)
# Billing matches that are NOT proof of exhaustion: Anthropic returns the same
# "out of extra usage" body for a content-filter rejection (#82154). Verdict
# stays ``billing`` but error_context marks it unverified so surfaces hedge
# and the credential pool uses a short cooldown instead of the 1h bench.
_UNVERIFIED_BILLING_PATTERNS = ("out of extra usage",)
# xAI's explicit Grok credit-exhaustion code, returned as HTTP 403 rather than
# 402. The 403 special case stays provider-scoped: other providers' billing
# codes on a 403 remain auth failures.
_XAI_SPENDING_LIMIT_ERROR_CODE = "personal-team-blocked:spending-limit"
# Structured codes meaning the account cannot serve paid traffic.
_BILLING_ERROR_CODES = frozenset({
"insufficient_quota", "billing_not_active", "payment_required", "insufficient_credits",
"no_usable_credits", "balance_depleted", "model_not_supported_on_free_tier",
"member_spend_cap_exceeded", _XAI_SPENDING_LIMIT_ERROR_CODE,
})
# Rate limiting (transient, will resolve). Bedrock "Throttling error: Too many
# tokens..." also contains the overflow phrase "too many tokens"; rate limit is
# matched first so throttle wins.
_RATE_LIMIT_PATTERNS = (
"rate limit", "rate_limit", "too many requests", "throttled", "requests per minute",
"tokens per minute", "requests per day", "try again in", "please retry after",
"resource_exhausted",
"rate increased too quickly", # Alibaba/DashScope throttling
"throttlingexception", "too many concurrent requests", "servicequotaexceededexception", # AWS Bedrock
"throttling",
)
# Provider-side overload: the credential is valid, the server is busy, so back
# off and retry the same key — never rotate. Some providers (Z.AI/Zhipu) reuse
# HTTP 429 for this, so the 429 path checks these first. Kept narrow so a
# normal "you have been rate-limited" doesn't land here. (#14038, #15297)
_OVERLOADED_PATTERNS = (
"overloaded", "temporarily overloaded", "service is temporarily overloaded",
"service may be temporarily overloaded", "server is overloaded", "server overloaded",
"service overloaded", "service is overloaded", "upstream overloaded", "currently overloaded",
"at capacity", "over capacity",
)
# Usage-limit patterns that need disambiguation (billing OR rate_limit), and
# the signals that mark such a limit as transient (periodic quota, not billing).
_USAGE_LIMIT_PATTERNS = ("usage limit", "quota", "limit exceeded", "key limit exceeded")
_USAGE_LIMIT_TRANSIENT_SIGNALS = (
"try again", "retry", "resets at", "reset in", "resets in", "reset after", "available in",
"wait", "requests remaining", "periodic", "window", "per minute", "per second",
)
# Payload-too-large detected from message text (proxies embed the status).
_PAYLOAD_TOO_LARGE_PATTERNS = (
"request entity too large", "payload too large", "error code: 413",
"request_too_large", # Anthropic's 413 type, re-wrapped by proxies without a status
"request exceeds the maximum size",
)
# Image-size rejections. Matched on 400 bodies (not 413): providers return a
# specific 400 before the whole request hits the size limit (Anthropic: hard
# 5 MB per image, "image exceeds 5 MB maximum"; dimension cap 8000 px).
# MiniMax Anthropic-compat says "media exceeds size limit" (#76039); a
# non-image media rejection landing here is harmless: the shrink pass finds
# no image parts and the original error surfaces.
_IMAGE_TOO_LARGE_PATTERNS = (
"image exceeds", "image too large", "image_too_large", "image size exceeds",
"image dimensions exceed", "dimensions exceed max allowed size", "max allowed size: 8000",
"media exceeds", "media too large",
)
# Image bytes undecodable (e.g. re-serialized history lost data). Shrinking
# can't fix corruption, so these route to strip-and-retry, never shrink.
# xAI wordings (#69078); the last is the full sentence on purpose — shorter
# fragments also match non-image download failures.
_IMAGE_CORRUPT_PATTERNS = (
"invalid png image", "invalid jpeg image", "base64 string of provided image cannot be decoded",
"downloaded response does not contain a valid jpg, png, webp, or ico image",
)
# Providers that reject list-type ``content`` in tool messages with a 400
# (Xiaomi MiMo "Param Incorrect ... text is not set", some Alibaba endpoints,
# OpenAI-compat long tail). Recovery: strip image parts from tool messages,
# remember (provider, model), retry. (#27344)
_MULTIMODAL_TOOL_CONTENT_PATTERNS = (
"text is not set", "tool message content must be a string", "tool content must be a string",
"tool message must be a string", "expected string, got list", "expected string, got array",
"tool_call.content must be string",
)
# Bare "max_tokens" is load-bearing: the output-cap-retry path keys off it.
# Empty-response advisories mentioning it are intercepted earlier by
# _EMPTY_PROVIDER_RESPONSE_PATTERNS, so they never route into compression.
_CONTEXT_OVERFLOW_PATTERNS = (
"context length", "context size", "maximum context", "token limit", "too many tokens",
"reduce the length", "exceeds the limit", "context window", "prompt is too long",
"prompt exceeds max length", "max_tokens", "maximum number of tokens",
# vLLM / local inference servers
"exceeds the max_model_len", "max_model_len", "prompt length", "input is too long",
"maximum model length",
# Ollama
"context length exceeded", "truncating input",
# llama.cpp / llama-server ("slot context: N tokens, prompt N tokens")
"slot context", "n_ctx_slot",
# Chinese error messages (some providers return these)
"超过最大长度", "上下文长度",
# Z.AI / Zhipu GLM (English form; error code 1210)
"tokens in request more than max tokens allowed",
# AWS Bedrock Converse API
"input is too long", "max input token", "input token", "exceeds the maximum number of input tokens",
# Together/Fireworks: "Input length N exceeds the maximum allowed input length of M tokens."
"maximum allowed input length",
)
_MODEL_NOT_FOUND_PATTERNS = (
"is not a valid model", "invalid model", "model not found", "model_not_found", "does not exist",
"no such model", "unknown model", "unsupported model",
# OpenRouter 404 when no endpoint for the model supports tool calling;
# model_not_found triggers fallback instead of burning retries (#58446).
"no endpoints found that support tool use",
)
# Qwen/vLLM chat-template raise_exception("No user query found in messages").
# Shared by _INVALID_MESSAGE_BODY_PATTERNS (→ format_error) and the llama.cpp
# grammar exclusion guard so the two sites cannot drift.
_NO_USER_QUERY_SIGNAL = "no user query found"
# Malformed-message-array 400s: deterministic rejections of the *transcript*
# (e.g. a content-less assistant stub after a dead stream). NOT context
# overflow — the input may be tiny — so they must fail fast as format_error
# instead of thrashing the compression loop. Compression cannot invent a
# missing user turn, and local engines may wrap that as a grammar error.
_INVALID_MESSAGE_BODY_PATTERNS = (
"must have non-empty content", "messages must have non-empty", "invalid_request_body",
"text content blocks must be non-empty", "content field is required",
"messages: at least one message is required", _NO_USER_QUERY_SIGNAL,
)
# Request-validation signals: malformed request, identical on every retry.
# Some gateways (codex.nekos.me) return these as 5xx, so the 5xx path also
# checks them to avoid a retry flood on a deterministic rejection.
_REQUEST_VALIDATION_PATTERNS = (
"unknown parameter", "unsupported parameter", "unrecognized request argument",
"invalid_request_error", "unknown_parameter", "unsupported_parameter",
)
# Parameters Hermes sends on SOME routes only → hosts where sending them is
# deliberate. A rejection from any other host means the provider's own gateway
# injected the field, so the 400 is a server-side flake, not our request shape.
# ``prompt_cache_retention``: only sent for api.meta.ai / bedrock-mantle
# (agent/transports/codex.py); the Codex OAuth backend rejects it spontaneously.
_SERVER_INJECTED_PARAM_SENDERS: Dict[str, tuple] = {
"prompt_cache_retention": ("meta", "muse", "msl", "model-api", "bedrock", "mantle"),
}
_PARAM_REJECTION_WORDS = ("not supported", "unsupported", "unknown", "unrecognized")
# OpenRouter 404 when the account privacy setting (or per-request
# ``provider.data_collection: deny``) excludes the only endpoint for a model.
# Not model_not_found: the model exists, fallback can't help (account-level),
# and the body already carries the fix URL.
_PROVIDER_POLICY_BLOCKED_PATTERNS = (
"no endpoints available matching your guardrail", "no endpoints available matching your data policy",
"no endpoints found matching your data policy",
)
# Per-prompt provider safety-filter blocks (distinct from the account-level
# provider_policy_blocked). Deterministic for the unchanged request, so
# fallback immediately. Each phrase is verbatim from a specific provider —
# never a generic word like "policy" that could collide with billing/auth.
_CONTENT_POLICY_BLOCKED_PATTERNS = (
# OpenAI Codex (#18028) — message may arrive without an HTTP status
"flagged for possible cybersecurity risk", "trusted access for cyber",
# OpenAI moderation — chat completions / responses
"violates our usage policies", "violates openai's usage policies", "your request was flagged by",
# Anthropic safety system
"prompt was flagged by our safety", "responses cannot be generated due to safety",
# OpenAI-standard token / Azure error code. Deliberately NOT the space
# variant "content filter", which appears in benign echoed config text.
"content_filter", "responsibleaipolicyviolation",
"new_sensitive", # MiniMax output-layer safety filter, "output new_sensitive (1027)" (#32421)
)
# Auth patterns (non-status-code signals).
_AUTH_PATTERNS = (
"invalid api key", "invalid_api_key", "gateway_auth_failed", "authentication", "unauthorized",
"forbidden", "invalid token", "token expired", "token revoked", "access denied",
)
# Provider empty-response advisories (OpenRouter / nano-gpt / similar). Checked
# before context-overflow matching because the text often mentions
# "max_tokens", which used to send healthy sessions into a compression spiral.
_EMPTY_PROVIDER_RESPONSE_PATTERNS = (
"returned an empty response", "empty response despite retries", "provider returned an empty response",
"model returning empty responses", "empty response stream",
)
# Timeout wording from generic exception types (RuntimeError from a shim
# wrapping a subprocess timeout) that the type-based heuristics would miss.
_TIMEOUT_MESSAGE_PATTERNS = (
"timed out", "turn timed out", "request timed out", "deadline exceeded", "operation timed out",
"upstream timed out",
)
# Connect/DNS failures surfaced by generic exception types with no status, so
# _TRANSPORT_ERROR_TYPES never fires. Deliberately EXCLUDES mid-stream
# disconnect strings — those belong to _SERVER_DISCONNECT_PATTERNS, which may
# route large sessions to compression; a never-established connection cannot
# be an overflow rejection.
_CONNECTION_MESSAGE_PATTERNS = (
# TCP connect failures
"connection refused", "econnrefused", "no route to host", "network is unreachable", "network unreachable",
# DNS resolution failures (Python, glibc, macOS, Node bridge phrasings)
"name or service not known", "temporary failure in name resolution", "nodename nor servname provided",
"getaddrinfo failed", "getaddrinfo enotfound", "eai_again",
# Node/undici bridge generic network failure (MCP servers, local shims)
"fetch failed", "failed to fetch",
# Envoy/proxy upstream connect failure (cloud gateways)
"upstream connect error",
)
# SSL type names are listed so provider-wrapped SSL errors (chain lost) still
# classify as transport instead of unknown; OpenAI SDK errors are not
# subclasses of Python builtins.
_TRANSPORT_ERROR_TYPES = frozenset({
"ReadTimeout", "ConnectTimeout", "PoolTimeout", "ConnectError", "RemoteProtocolError",
"ConnectionError", "ConnectionResetError", "ConnectionAbortedError", "BrokenPipeError",
"TimeoutError", "ReadError", "ServerDisconnectedError",
"SSLError", "SSLZeroReturnError", "SSLWantReadError", "SSLWantWriteError", "SSLEOFError", "SSLSyscallError",
"APIConnectionError", "APITimeoutError",
})
# Ambiguous disconnects (no status): transient hiccup OR a gateway dropping an
# oversized request. A large session + one of these → context-overflow path.
_SERVER_DISCONNECT_PATTERNS = (
"server disconnected", "peer closed connection", "connection reset by peer", "connection was closed",
"network connection lost", "unexpected eof", "incomplete chunked read",
)
# SSL certificate verification failures are deterministic (proxy, missing CA,
# expired/self-signed cert) — fail fast. Checked BEFORE _SSL_TRANSIENT_PATTERNS
# because these messages usually also contain "[SSL:".
_SSL_CERT_VERIFY_PATTERNS = (
"certificate verify failed", "certificate_verify_failed", "unable to get local issuer certificate",
"self-signed certificate", "self signed certificate", "certificate has expired",
"hostname mismatch, certificate is not valid",
"unable to verify the first certificate", # Node/undici phrasing (MCP bridges)
)
# Transient SSL alerts: retry but NOT compression (kept apart from
# _SERVER_DISCONNECT_PATTERNS). Matched on stable substrings because OpenSSL 3
# changed token separators (SSLV3_ALERT_... → SSL/TLS_ALERT_...); both the
# space-separated human form and the underscore OpenSSL tokens are listed,
# plus the Python ssl module prefix "[SSL: BAD_RECORD_MAC]".
_SSL_TRANSIENT_PATTERNS = (
"bad record mac", "ssl alert", "tls alert", "ssl handshake failure", "tlsv1 alert", "sslv3 alert",
"bad_record_mac", "ssl_alert", "tls_alert", "tls_alert_internal_error", "[ssl:",
)
# ── Verdicts and rule tables ────────────────────────────────────────────
#
# A verdict is the ClassifiedError kwargs a stage decided on: ``reason`` plus
# the hint overrides (unlisted hints keep the dataclass defaults: retryable
# True, everything else False/empty). Rule tables are ordered
# ``(patterns, verdict)`` pairs matched first-hit; ``verdict`` may be a
# callable of the error message.
Verdict = Dict[str, Any]
def _v(reason: FailoverReason, **hints: Any) -> Verdict:
return {"reason": reason, **hints}
_ROTATE_FALLBACK = {"should_rotate_credential": True, "should_fallback": True}
_V_BILLING = _v(FailoverReason.billing, retryable=False, **_ROTATE_FALLBACK)
_V_RATE_LIMIT = _v(FailoverReason.rate_limit, **_ROTATE_FALLBACK)
_V_OVERLOADED = _v(FailoverReason.overloaded)
_V_SERVER_ERROR = _v(FailoverReason.server_error)
_V_CONTEXT_OVERFLOW = _v(FailoverReason.context_overflow, should_compress=True)
_V_PAYLOAD_TOO_LARGE = _v(FailoverReason.payload_too_large, should_compress=True)
_V_MODEL_NOT_FOUND = _v(FailoverReason.model_not_found, retryable=False, should_fallback=True)
_V_POLICY_BLOCKED = _v(FailoverReason.provider_policy_blocked, retryable=False)
_V_CONTENT_BLOCKED = _v(FailoverReason.content_policy_blocked, retryable=False, should_fallback=True)
_V_FORMAT_ERROR = _v(FailoverReason.format_error, retryable=False, should_fallback=True)
_V_AUTH_ROTATE = _v(FailoverReason.auth, retryable=False, **_ROTATE_FALLBACK)
_V_AUTH_FALLBACK = _v(FailoverReason.auth, retryable=False, should_fallback=True)
_V_TIMEOUT = _v(FailoverReason.timeout)
_V_SSL_CERT = _v(FailoverReason.ssl_cert_verification, retryable=False)
_V_IMAGE_TOO_LARGE = _v(FailoverReason.image_too_large)
_V_IMAGE_CORRUPT = _v(FailoverReason.image_corrupt)
_V_MULTIMODAL = _v(FailoverReason.multimodal_tool_content_unsupported)
_V_INVALID_ENCRYPTED = _v(FailoverReason.invalid_encrypted_content)
_V_UNKNOWN = _v(FailoverReason.unknown)
def _billing_hints(error_msg: str) -> Verdict:
"""Billing verdict carrying the #82154 ambiguity marker when applicable."""
ctx: Dict[str, Any] = {}
if any(p in error_msg for p in _UNVERIFIED_BILLING_PATTERNS):
ctx = {"billing_unverified": True, "possible_content_filter": True}
return {**_V_BILLING, "error_context": ctx}
def _first_match(error_msg: str, rules: Sequence[tuple[Sequence[str], Any]]) -> Optional[Verdict]:
"""Verdict of the first rule whose pattern list hits ``error_msg``."""
for patterns, verdict in rules:
if any(p in error_msg for p in patterns):
return verdict(error_msg) if callable(verdict) else verdict
return None
# Image/tool-content 400s, ordered: multimodal recovery differs from image
# shrink; corrupt bytes need strip-and-retry, not shrink; image-shrink is a
# cheaper recovery than context compression for "exceeds" + "image" bodies.
_IMAGE_TOOL_RULES = (
(_MULTIMODAL_TOOL_CONTENT_PATTERNS, _V_MULTIMODAL),
(_IMAGE_CORRUPT_PATTERNS, _V_IMAGE_CORRUPT),
(_IMAGE_TOO_LARGE_PATTERNS, _V_IMAGE_TOO_LARGE),
)
# Overflow signals arriving as 5xx (llama.cpp reports overflow as 500; busy /
# model-load OOM as 503). Empty-response advisories must not enter compression.
_OVERFLOW_AS_5XX_RULES = (
(_EMPTY_PROVIDER_RESPONSE_PATTERNS, _V_SERVER_ERROR),
(_CONTEXT_OVERFLOW_PATTERNS, _V_CONTEXT_OVERFLOW),
)
# 404: Nous API surfaces credit depletion as a paid model vanishing from the
# Free Tier (billing, not missing model); OpenRouter policy block before
# model_not_found.
_404_RULES = (
(_BILLING_PATTERNS, _V_BILLING),
(_PROVIDER_POLICY_BLOCKED_PATTERNS, _V_POLICY_BLOCKED),
(_MODEL_NOT_FOUND_PATTERNS, _V_MODEL_NOT_FOUND),
)
# 400 tail after the deterministic request-shape checks. Some providers return
# model-not-found / rate-limit / billing as 400 instead of 404/429/402.
_400_TAIL_RULES = _OVERFLOW_AS_5XX_RULES + (
(_PROVIDER_POLICY_BLOCKED_PATTERNS, _V_POLICY_BLOCKED),
(_MODEL_NOT_FOUND_PATTERNS, _V_MODEL_NOT_FOUND),
(_RATE_LIMIT_PATTERNS, _V_RATE_LIMIT),
(_BILLING_PATTERNS, _billing_hints),
)
# Status-less message path, head (before usage-limit disambiguation).
_MESSAGE_HEAD_RULES = ((_PAYLOAD_TOO_LARGE_PATTERNS, _V_PAYLOAD_TOO_LARGE),) + _IMAGE_TOOL_RULES
# Status-less message path, tail. Overload before rate_limit/billing so a
# message-only "overloaded" backs off instead of rotating; auth is
# non-retryable (same key always fails); policy block before model_not_found;
# timeout/connection wording last, classified as transport (never compression).
_MESSAGE_TAIL_RULES = (
(_OVERLOADED_PATTERNS, _V_OVERLOADED),
(_BILLING_PATTERNS, _billing_hints),
(_RATE_LIMIT_PATTERNS, _V_RATE_LIMIT),
(_EMPTY_PROVIDER_RESPONSE_PATTERNS, _V_SERVER_ERROR),
(_CONTEXT_OVERFLOW_PATTERNS, _V_CONTEXT_OVERFLOW),
(_AUTH_PATTERNS, _V_AUTH_ROTATE),
(_PROVIDER_POLICY_BLOCKED_PATTERNS, _V_POLICY_BLOCKED),
(_MODEL_NOT_FOUND_PATTERNS, _V_MODEL_NOT_FOUND),
(_TIMEOUT_MESSAGE_PATTERNS, _V_TIMEOUT),
(_CONNECTION_MESSAGE_PATTERNS, _V_TIMEOUT),
)
# Structured error code → verdict. The error-code rate_limit verdict rotates
# but does not set should_fallback (unlike the message/status paths).
_ERROR_CODE_VERDICTS: Dict[str, Verdict] = {
**dict.fromkeys(("resource_exhausted", "throttled", "rate_limit_exceeded"),
_v(FailoverReason.rate_limit, should_rotate_credential=True)),
**dict.fromkeys(_BILLING_ERROR_CODES, _V_BILLING),
**dict.fromkeys(("model_not_found", "model_not_available", "invalid_model"), _V_MODEL_NOT_FOUND),
**dict.fromkeys(("context_length_exceeded", "max_tokens_exceeded"), _V_CONTEXT_OVERFLOW),
"invalid_encrypted_content": _V_INVALID_ENCRYPTED,
}
_5XX_VALIDATION_CODES = {"invalid_request_error", "unknown_parameter", "unsupported_parameter"}
_400_VALIDATION_CODES = {"unknown_parameter", "unsupported_parameter"}
# Generic ``invalid_request_error`` is deliberately NOT a 400 validation
# signal — OpenAI stamps it on genuine overflow 400s too.
_400_VALIDATION_PATTERNS = tuple(p for p in _REQUEST_VALIDATION_PATTERNS if p != "invalid_request_error")
# ── Classification pipeline ─────────────────────────────────────────────
@dataclass
class _Ctx:
"""Everything the classifier stages need about one failed call."""
error: Exception
status_code: Optional[int]
error_type: str
body: dict
error_code: str
headers: Any
msg: str # lowercased str(error) + body message(s)
provider: str # as passed by the caller
model: str
provider_slug: str # stripped + lowercased
model_slug: str
approx_tokens: int
context_length: int
num_messages: int
def large_session(self, frac: float, tokens: int, messages: int) -> bool:
"""Absolute thresholds only proxy for smaller context windows."""
return self.approx_tokens > self.context_length * frac or (
self.context_length <= 256000 and (self.approx_tokens > tokens or self.num_messages > messages)
)
def _plugin_verdict(c: _Ctx) -> Optional[Verdict]:
"""First valid plugin classification, so a provider plugin can add or
correct classifications before the built-in pipeline. invoke_hook
isolates callback failures; this guard only covers import/dispatch failure."""
try:
from hermes_cli.plugins import get_plugin_error_classification
verdict = get_plugin_error_classification(
provider=c.provider, model=c.model, status_code=c.status_code, error_type=c.error_type,
error_code=c.error_code, error_message=c.msg, error_body=c.body, error=c.error,
approx_tokens=c.approx_tokens, context_length=c.context_length, num_messages=c.num_messages,
)
except Exception as exc:
logger.debug("Plugin error classification unavailable: %s", exc)
return None
if verdict is not None:
logger.info(
"API error classified by plugin hook: %s (provider=%s, status=%s)",
verdict["reason"].value, c.provider, c.status_code,
)
return verdict
def _provider_special_cases(c: _Ctx) -> Optional[Verdict]:
"""Highest-priority provider-specific shapes that a status code would misroute."""
msg, status = c.msg, c.status_code
# Deterministic per-prompt safety refusal. Before status classification so
# a 400 block isn't downgraded to format_error and a status-less block
# isn't left retryable (#18028).
if any(p in msg for p in _CONTENT_POLICY_BLOCKED_PATTERNS):
return _V_CONTENT_BLOCKED
# Anthropic thinking-block 400s: signature mismatch after any transcript
# mutation, or "blocks in the latest assistant message cannot be modified".
# Not gated on provider — OpenRouter proxies Anthropic errors.
if status == 400 and "thinking" in msg and any(
p in msg for p in ("signature", "cannot be modified", "must remain as they were")
):
return _v(FailoverReason.thinking_signature)
# Anthropic long-context tier gate (429 "extra usage" + "long context").
if status == 429 and "extra usage" in msg and "long context" in msg:
return _v(FailoverReason.long_context_tier, should_compress=True)
# Anthropic OAuth subscription rejects the 1M-context beta header;
# run_agent rebuilds the client without the beta and retries once.
if status == 400 and "long context beta" in msg and "not yet available" in msg:
return _v(FailoverReason.oauth_long_context_beta_forbidden)
# llama.cpp json-schema-to-grammar rejects regex escapes / ``format`` in
# tool schemas (400); the retry loop strips pattern/format and retries.
# Exclude the Qwen/vLLM "No user query found" template error that local
# engines wrap as "Unable to generate parser for this template" — that is
# a poisoned transcript (→ format_error), not a grammar problem.
grammar_hit = status == 400 and (
"error parsing grammar" in msg
or "json-schema-to-grammar" in msg
or ("unable to generate parser" in msg and "template" in msg)
)
if grammar_hit and _NO_USER_QUERY_SIGNAL not in msg:
return _v(FailoverReason.llama_cpp_grammar_pattern)
# xAI Grok subscription entitlement. As HTTP 403 the status path handles
# it; as an SSE ``type=error`` frame there is no status and the message
# matches no pattern list, so it would burn max_retries as ``unknown``.
if "do not have an active grok subscription" in msg or (
"out of available resources" in msg and "grok" in msg
):
return _V_AUTH_FALLBACK
return None
def _moa_special_cases(c: _Ctx) -> Optional[Verdict]:
# Local MoA streaming adapter-shape bugs are not a provider outage; falling
# back would silently replace the MoA route with a single model (#55933).
if c.provider_slug == "moa" and (
"'types.SimpleNamespace' object is not iterable" in str(c.error)
or "'types.SimpleNamespace' object has no attribute 'index'" in str(c.error)
):
return _v(FailoverReason.format_error, retryable=False)
# Persisted MoA preset name that was renamed/deleted — deterministic config error.
from agent.errors import MoAPresetNotFoundError
if isinstance(c.error, MoAPresetNotFoundError):
return _v(FailoverReason.model_not_found, retryable=False)
return None
def _by_error_code(c: _Ctx) -> Optional[Verdict]:
"""Structured error codes from the response body."""
code = c.error_code.lower()
# Deterministic request-validation failures encoded as plain-text
# ``event: error`` SSE data behind HTTP 200: retrying cannot succeed, a
# configured fallback still may.
if code == PROVIDER_STREAM_NON_JSON_ERROR_CODE and "request validation failed:" in c.msg:
return _V_FORMAT_ERROR
return _ERROR_CODE_VERDICTS.get(code)
def _by_message(c: _Ctx) -> Optional[Verdict]:
"""Message patterns when no status code settled it."""
verdict = _first_match(c.msg, _MESSAGE_HEAD_RULES)
if verdict is not None:
return verdict
# Status-less usage limits need the same disambiguation as 402.
if any(p in c.msg for p in _USAGE_LIMIT_PATTERNS):
return _classify_402(c.msg, dict)
return _first_match(c.msg, _MESSAGE_TAIL_RULES)
def _by_transport(c: _Ctx) -> Optional[Verdict]:
"""SSL, disconnect, circuit-breaker and transport-type heuristics, in that order."""
msg = c.msg
# Deterministic cert failure → fail fast; transient alert → retry.
# Cert-verify first: those messages also contain "[ssl:". Transient alerts
# are classified before the disconnect check so a large session doesn't
# compress on a flaky TLS handshake.
if any(p in msg for p in _SSL_CERT_VERIFY_PATTERNS):
return _V_SSL_CERT
if any(p in msg for p in _SSL_TRANSIENT_PATTERNS):
return _V_TIMEOUT
# Server disconnect + large session → context overflow. Before the generic
# transport catch: a disconnect on a large session is more likely an
# overflow rejection than a transport hiccup.
if any(p in msg for p in _SERVER_DISCONNECT_PATTERNS) and not c.status_code:
# Reasoning models: a disconnect is far more likely the gateway
# idle-killing a long thinking stream than overflow — never compress
# (and silently drop history) on a phantom overflow (#52310).
from agent.reasoning_timeouts import get_reasoning_stale_timeout_floor
if get_reasoning_stale_timeout_floor(c.model) is not None:
return _V_TIMEOUT
return _V_CONTEXT_OVERFLOW if c.large_session(0.6, 120000, 200) else _V_TIMEOUT
# Stale-call circuit breaker → failover immediately. _check_stale_giveup()
# raises RuntimeError before any network call; as ``unknown`` it would burn
# every retry instantly against the dead provider.
if c.error_type == "RuntimeError" and "consecutive stale attempts" in msg and "aborting this call" in msg:
return _v(FailoverReason.timeout, retryable=False, should_fallback=True)
if c.error_type in _TRANSPORT_ERROR_TYPES or isinstance(c.error, (TimeoutError, ConnectionError, OSError)):
return _V_TIMEOUT
return None
def _by_status(c: _Ctx) -> Optional[Verdict]:
"""HTTP status code with message-aware refinement."""
if c.status_code is None:
return None
handler = _STATUS_HANDLERS.get(c.status_code)
if handler is not None:
return handler(c)
if 400 <= c.status_code < 500:
return _V_FORMAT_ERROR
if 500 <= c.status_code < 600:
return _V_SERVER_ERROR
return None
# Stage order: plugin hooks → provider-specific special cases → HTTP status →
# MoA shapes → structured error code → message patterns → SSL → disconnect +
# large session → transport types → unknown (retryable with backoff).
_STAGES: Sequence[Callable[[_Ctx], Optional[Verdict]]] = (
_plugin_verdict, _provider_special_cases, _by_status, _moa_special_cases,
_by_error_code, _by_message, _by_transport,
)
def classify_api_error(
error: Exception,
*,
provider: str = "",
model: str = "",
approx_tokens: int = 0,
context_length: int = 200000,
num_messages: int = 0,
) -> ClassifiedError:
"""Classify an API error into a structured recovery recommendation (see ``_STAGES``)."""
status_code = _extract_status_code(error)
error_type = type(error).__name__
# Copilot/GitHub Models RateLimitError may not set .status_code; force 429.
if status_code is None and error_type == "RateLimitError":
status_code = 429
body = _extract_error_body(error)
c = _Ctx(
error=error, status_code=status_code, error_type=error_type, body=body,
error_code=_extract_error_code(body), headers=_extract_response_headers(error),
msg=_build_error_msg(error, body), provider=provider, model=model,
provider_slug=(provider or "").strip().lower(), model_slug=(model or "").strip().lower(),
approx_tokens=approx_tokens, context_length=context_length, num_messages=num_messages,
)
verdict = next((v for v in (stage(c) for stage in _STAGES) if v is not None), _V_UNKNOWN)
return ClassifiedError(**{
"status_code": status_code, "provider": provider, "model": model,
"message": _extract_message(error, body), **verdict,
})
# ── Status code handlers ────────────────────────────────────────────────
def _status_403(c: _Ctx) -> Verdict:
# OpenRouter 403 "key limit exceeded" and similar plan/credit exhaustion are billing.
if (
(c.provider_slug == "xai-oauth" and c.error_code.lower() == _XAI_SPENDING_LIMIT_ERROR_CODE)
or "key limit exceeded" in c.msg
or "spending limit" in c.msg
or any(p in c.msg for p in _BILLING_PATTERNS)
):
return _V_BILLING
return _V_AUTH_FALLBACK
def _status_404(c: _Ctx) -> Verdict:
verdict = _first_match(c.msg, _404_RULES)
if verdict is not None:
return verdict
# Bare id the catalogue only knows prefixed → malformed id (NVIDIA NIM
# "404 page not found"), deterministic (#78796). A generic 404 (wrong
# endpoint path, proxy glitch) stays unknown so the real error surfaces
# instead of model_not_found silently falling back and misreporting.
return _V_MODEL_NOT_FOUND if _model_id_missing_known_prefix(c.model_slug, c.provider_slug) else _V_UNKNOWN
def _status_429(c: _Ctx) -> Verdict:
# Z.AI/Zhipu reuse 429 for server-wide overload: back off on the same
# key instead of burning the pool (#14038).
if any(p in c.msg for p in _OVERLOADED_PATTERNS):
return _V_OVERLOADED
# OpenRouter-wrapped upstream 429: the user's key is healthy, so fall back
# to another model rather than rotating/benching the key.
if _is_openrouter_upstream_error(c.body, c.provider_slug):
upstream = _extract_upstream_provider_name(c.body)
return _v(
FailoverReason.upstream_rate_limit, should_fallback=True,
error_context={"upstream_provider": upstream} if upstream else {},
)
# Quota walls returned as 429 (Anthropic ``usage_limit_reached``, other
# providers' "quota"/"limit exceeded", explicit billing phrases) are
# billing — but ONLY when the body is not itself an explicit rate-limit
# phrase ("Rate limit exceeded" contains "limit exceeded") and carries
# no reset/retry signal (#93419, #39441).
has_usage_limit = (
c.error_code.lower() == "usage_limit_reached"
or "usage_limit_reached" in c.msg
or any(p in c.msg for p in _USAGE_LIMIT_PATTERNS)
)
if (
(has_usage_limit or any(p in c.msg for p in _BILLING_PATTERNS))
and not any(p in c.msg for p in _RATE_LIMIT_PATTERNS)
and not _has_usage_limit_transient_signal(c.msg, c.body, c.headers)
):
return _V_BILLING
return _V_RATE_LIMIT
def _status_5xx(c: _Ctx) -> Verdict:
# Deterministic request-validation errors returned as 5xx (codex.nekos.me)
# must fail fast, not retry-flood — unless the rejected parameter was
# injected server-side (see _classify_400).
if (
any(p in c.msg for p in _REQUEST_VALIDATION_PATTERNS)
or c.error_code.lower() in _5XX_VALIDATION_CODES
) and not _is_server_injected_param_rejection(c.msg, c.provider_slug):
return _V_FORMAT_ERROR
return _first_match(c.msg, _OVERFLOW_AS_5XX_RULES) or _V_SERVER_ERROR
def _status_overloaded(c: _Ctx) -> Verdict:
return _first_match(c.msg, _OVERFLOW_AS_5XX_RULES) or _V_OVERLOADED
def _classify_402(error_msg: str, result_fn: Callable[..., Any]) -> Any:
"""Disambiguate 402: "usage limit, try again in 5 minutes" is a periodic quota, not billing."""
if (
any(p in error_msg for p in _USAGE_LIMIT_PATTERNS)
and any(p in error_msg for p in _USAGE_LIMIT_TRANSIENT_SIGNALS)
):
return result_fn(**_V_RATE_LIMIT)
return result_fn(**_V_BILLING)
def _classify_400(c: _Ctx) -> Verdict:
"""400 Bad Request — image/tool shapes, request-shape rejections, overflow, or generic."""
msg, code = c.msg, c.error_code.lower()
verdict = _first_match(msg, _IMAGE_TOOL_RULES)
if verdict is not None:
return verdict
# Invalid encrypted reasoning replay blob (OpenAI Responses). Before
# context_overflow: "encrypted content … could not be verified" can trip
# the overflow heuristics.
if (
code == "invalid_encrypted_content"
or "invalid_encrypted_content" in msg
or ("encrypted content for item" in msg and "could not be verified" in msg)
or "could not decrypt the provided encrypted_content" in msg
):
return _V_INVALID_ENCRYPTED
# A 400 blaming a field this route never sent (Codex OAuth backend injects
# and then rejects prompt_cache_retention ~20% of the time): transient,
# retry the identical request; never compress. Before the validation branch.
if _is_server_injected_param_rejection(msg, c.provider_slug):
return _V_SERVER_ERROR
# Unsupported/unknown parameter before context_overflow: GPT-5's
# "Unsupported parameter: 'max_tokens'…" contains the overflow pattern.
if any(p in msg for p in _400_VALIDATION_PATTERNS) or code in _400_VALIDATION_CODES:
return _V_FORMAT_ERROR
# Malformed message array (empty-content assistant stub, etc.) before
# context_overflow: the input can be tiny and compression cannot fix it.
# Proxies (litellm/Bedrock) surface it as errorCode=INVALID_REQUEST_BODY.
if any(p in msg for p in _INVALID_MESSAGE_BODY_PATTERNS) or code == "invalid_request_body":
logger.warning(
"Malformed message array 400 (invalid request body) classified as "
"format_error, NOT context overflow — failing fast + falling back "
"instead of entering the compression loop. This usually means an "
"empty-content assistant stub is in the transcript; num_messages=%s "
"approx_tokens=%s. error=%.200s",
c.num_messages, c.approx_tokens, msg,
)
return _V_FORMAT_ERROR
verdict = _first_match(msg, _400_TAIL_RULES)
if verdict is not None:
return verdict
# Generic 400 + large session → probable context overflow (Anthropic can
# return a bare "Error"). Proxy shapes are recognised so a long, descriptive
# rejection is not mistaken for a bare error.
body_msg = next((m for m in (str(x or "").strip().lower() for x in _body_message_candidates(c.body)) if m), "")
is_generic = len(body_msg) < 30 or body_msg in {"error", ""}
if is_generic and c.large_session(0.4, 80000, 80):
return _V_CONTEXT_OVERFLOW
return _V_FORMAT_ERROR
# 401 is not retryable on its own: credential rotation / provider refresh run
# before the retryability check; if they fail, the client-error abort path
# (fallback first) is correct. 408 Request Timeout is retry-safe (RFC 9110
# §15.5.9) — proxies in front of self-hosted backends emit it when generation
# outruns the read window. Unlisted 4xx → format_error, 5xx → server_error.
_STATUS_HANDLERS: Dict[int, Callable[[_Ctx], Verdict]] = {
400: _classify_400,
401: lambda c: _V_AUTH_ROTATE,
402: lambda c: _classify_402(c.msg, dict),
403: _status_403,
404: _status_404,
408: lambda c: _V_TIMEOUT,
413: lambda c: _V_PAYLOAD_TOO_LARGE,
429: _status_429,
500: _status_5xx,
502: _status_5xx,
503: _status_overloaded,
529: _status_overloaded,
}
# ── Helpers ─────────────────────────────────────────────────────────────
_RESET_FIELDS = ("resets_in_seconds", "resets_at", "reset_at", "retry_after")
_RESET_HEADERS = ("retry-after", "Retry-After", "x-ratelimit-reset", "X-RateLimit-Reset")
def _has_usage_limit_transient_signal(error_msg: str, body: dict, response_headers) -> bool:
"""Whether a usage-limit response identifies a reset window (message, body fields, or headers)."""
if any(pattern in error_msg for pattern in _USAGE_LIMIT_TRANSIENT_SIGNALS):
return True
payloads = [body]
if isinstance(body, dict) and isinstance(body.get("error"), dict):
payloads.append(body["error"])
for payload in payloads:
if isinstance(payload, dict) and any(payload.get(f) not in (None, "") for f in _RESET_FIELDS):
return True
if response_headers and hasattr(response_headers, "get"):
return any(response_headers.get(h) not in (None, "") for h in _RESET_HEADERS)
return False
def _model_id_missing_known_prefix(model: str, provider: str) -> bool:
"""True when a bare model id is only known to the provider as ``vendor/id``.
NVIDIA NIM answers a bare id with a naked ``404 page not found``; the
curated catalogue tells that apart from a bad endpoint. Never guesses: an
id absent from the catalogue returns False so real endpoint problems keep
their retryable ``unknown`` classification.
"""
name = (model or "").strip()
if not name or "/" in name:
return False
try:
from hermes_cli.model_normalize import suggest_prefixed_model_id
return bool(suggest_prefixed_model_id((provider or "").strip(), name))
except Exception:
return False
def _is_server_injected_param_rejection(error_msg: str, provider: str) -> bool:
"""True when a 400 blames a one-route-only parameter this route never sends.
Conservative: fires only for known parameters AND only when ``provider``
is not a route that sends them, so a genuine client-side bad parameter
(``max_tokens`` on GPT-5) still fails fast as ``format_error``.
"""
if not error_msg:
return False
provider_slug = (provider or "").strip().lower()
for param, senders in _SERVER_INJECTED_PARAM_SENDERS.items():
if param not in error_msg or not any(w in error_msg for w in _PARAM_REJECTION_WORDS):
continue
return not any(sender in provider_slug for sender in senders)
return False
def _openrouter_wrapped_message(err_obj: dict) -> str:
"""Lowercased inner message from OpenRouter's ``error.metadata.raw`` JSON wrapper."""
metadata = err_obj.get("metadata", {})
raw = metadata.get("raw") or "" if isinstance(metadata, dict) else ""
if not (isinstance(raw, str) and raw.strip()):
return ""
try:
inner = json.loads(raw)
except (json.JSONDecodeError, TypeError):
return ""
inner_err = inner.get("error", {}) if isinstance(inner, dict) else None
if isinstance(inner_err, dict):
return str(inner_err.get("message") or "").lower()
return ""
def _build_error_msg(error: Exception, body: Any) -> str:
"""Lowercased str(error) + body message + OpenRouter-wrapped upstream message.
str(error) alone may omit the body (OpenAI SDK's APIStatusError.__str__
returns only the first arg), so body text is appended for pattern matching.
"""
raw_msg = str(error).lower()
body_msg = metadata_msg = ""
if isinstance(body, dict):
err_obj = body.get("error", {})
if isinstance(err_obj, dict):
body_msg = str(err_obj.get("message") or "").lower()
metadata_msg = _openrouter_wrapped_message(err_obj)
if not body_msg:
body_msg = str(body.get("message") or "").lower()
parts = [raw_msg]
if body_msg and body_msg not in raw_msg:
parts.append(body_msg)
if metadata_msg and metadata_msg not in raw_msg and metadata_msg not in body_msg:
parts.append(metadata_msg)
return " ".join(parts)
def _body_message_candidates(body: dict) -> Iterator[Any]:
"""Body message fields in priority order (OpenAI, flat, litellm/Bedrock proxy shapes)."""
err_obj = body.get("error", {})
yield err_obj.get("message") if isinstance(err_obj, dict) else None
yield body.get("message")
yield body.get("errorMessage")
args = body.get("errorArgs")
yield args.get("reason") if isinstance(args, dict) else None
def _cause_chain(error: Exception) -> Iterator[Any]:
"""Yield the error and its __cause__/__context__ chain, at most 5 deep."""
current = error
for _ in range(5):
yield current
cause = getattr(current, "__cause__", None) or getattr(current, "__context__", None)
if cause is None or cause is current:
return
current = cause
def _extract_status_code(error: Exception) -> Optional[int]:
"""Walk the error and its cause chain to find an HTTP status code."""
for current in _cause_chain(error):
code = getattr(current, "status_code", None)
if isinstance(code, int):
return code
code = getattr(current, "status", None) # some SDKs use .status
if isinstance(code, int) and 100 <= code < 600:
return code
return None
def _extract_error_body(error: Exception) -> dict:
"""Extract the structured error body from an SDK exception or its cause chain."""
for current in _cause_chain(error):
body = getattr(current, "body", None)
if isinstance(body, dict):
return body
response = getattr(current, "response", None)
if response is not None:
try:
json_body = response.json()
if isinstance(json_body, dict):
return json_body
except Exception:
pass
return {}
def _extract_response_headers(error: Exception):
"""Walk the error and its cause chain to find response headers."""
for current in _cause_chain(error):
headers = getattr(getattr(current, "response", None), "headers", None)
if headers and hasattr(headers, "get"):
return headers
return {}
def _code_from_payload(payload: Any, top_keys: Sequence[str], peek_message: bool) -> str:
"""Code/type from ``payload.error`` or a top-level key; ``"400"`` is not a code.
With ``peek_message``, a JSON string in ``error.message`` is parsed for a
nested code (Responses API surfaces ``invalid_encrypted_content`` this way).
"""
if not isinstance(payload, dict):
return ""
error_obj = payload.get("error", {})
if isinstance(error_obj, dict):
code = error_obj.get("code") or error_obj.get("type") or ""
if isinstance(code, str) and code.strip() and code.strip() != "400":
return code.strip()
message = error_obj.get("message")
if peek_message and isinstance(message, str) and message.strip().startswith("{"):
try:
inner = json.loads(message)
except (json.JSONDecodeError, TypeError):
inner = None
nested_code = _code_from_payload(inner, ("code", "error_code"), False)
if nested_code:
return nested_code
code = next((payload.get(k) for k in top_keys if payload.get(k)), "")
if isinstance(code, (str, int)):
text = str(code).strip()
if text and text != "400":
return text
return ""
def _extract_error_code(body: dict) -> str:
"""Extract an error code string from the response body."""
return _code_from_payload(body, ("code", "error_code", "errorCode"), True) if body else ""
def _extract_message(error: Exception, body: dict) -> str:
"""Extract the most informative error message (structured body first)."""
for msg in _body_message_candidates(body or {}):
if isinstance(msg, str) and msg.strip():
return msg.strip()[:500]
return str(error)[:500]
def _is_openrouter_upstream_error(body: Any, provider: str) -> bool:
"""Detect OpenRouter's "Provider returned error" wrapper around an upstream failure.
The user's OpenRouter key is healthy — the upstream provider failed — so
credential rotation is the wrong recovery.
"""
if not isinstance(body, dict):
return False
err = body.get("error")
if not isinstance(err, dict):
return False
if str(err.get("message") or "").strip().lower() != "provider returned error":
return False
if (provider or "").strip().lower() == "openrouter":
return True
# Otherwise require the metadata shape only OpenRouter produces.
metadata = err.get("metadata")
return isinstance(metadata, dict) and ("raw" in metadata or "provider_name" in metadata)
def _extract_upstream_provider_name(body: Any) -> Optional[str]:
"""Pull the upstream provider name out of OpenRouter's error metadata."""
err = body.get("error") if isinstance(body, dict) else None
metadata = err.get("metadata") if isinstance(err, dict) else None
name = metadata.get("provider_name") if isinstance(metadata, dict) else None
if isinstance(name, str) and name.strip():
return name.strip()
return None