Files
hermes-agent/agent/fallback_cooldown.py
teknium1 6c89cb032e refactor(agent): move the entitlement rejection marker into agent/fallback_cooldown
The two facades post-#102117 (chat_completion_helpers, agent_runtime_helpers) must not grow
behaviour; the marker/predicate belong with the sibling that already owns primary-cooldown
state shared by the fallback walk and restore_primary_runtime. Also drops the two
try/except-return-False wrappers around pool.entries() and normalize_model_for_provider
(neither raises for a live pool / the never-raising normalizer).

Tests trimmed to the two invariants (#106475): the marker fires only on the single-credential
Codex entitlement 400 and the walk skips the rejected slug in either form; restore is gated on
a rejected primary and still restores an unrelated one.
2026-09-09 11:29:56 -07:00

85 lines
4.3 KiB
Python

"""Primary rate-limit cooldown arming and per-session model rejection markers, shared by the
fallback walk (chat_completion_helpers) and restore_primary_runtime (agent_runtime_helpers)."""
import logging
import time
from agent.error_classifier import FailoverReason
logger = logging.getLogger(__name__)
_RATE_LIMIT_FAILOVER_REASONS = frozenset({FailoverReason.rate_limit, FailoverReason.billing, FailoverReason.upstream_rate_limit})
def _arm_rate_limit_cooldown(agent, reason: "FailoverReason | None") -> int | None:
"""Arm the primary's exponential cooldown (60s → 2m → ... → 4h cap) on CONSECUTIVE rate-limits;
restore_primary_runtime resets the counter. Only when leaving the primary: chain-switching from
an active fallback means the primary was not the 429 source, so its cooldown is left alone.
Return the armed cooldown in seconds, or None when no cooldown was armed."""
if reason not in _RATE_LIMIT_FAILOVER_REASONS:
return None
current_provider = (getattr(agent, "provider", "") or "").strip().lower()
primary_provider = ((agent._primary_runtime or {}).get("provider") or "").strip().lower()
if getattr(agent, "_fallback_activated", False) and not (primary_provider and current_provider == primary_provider):
return None
backoff_count = getattr(agent, "_rate_limit_backoff_count", 0)
agent._rate_limit_backoff_count = backoff_count + 1
backoff_seconds = min(60 * (2 ** backoff_count), 14400)
agent._rate_limited_until = time.monotonic() + backoff_seconds
logging.info("Rate-limit backoff level %d: cooldown %d s (%.1f min, backoff#%d)", backoff_count, backoff_seconds, backoff_seconds / 60, backoff_count + 1)
return backoff_seconds
# Codex ChatGPT-account entitlement 400 — the account can never use the named slug, so with
# nothing to rotate it is a config error, not a transient failure (#106475).
_CODEX_ACCOUNT_MODEL_ENTITLEMENT_MARKER = "model is not supported when using codex with a chatgpt account"
def _mark_entitlement_rejected_model(agent, api_error) -> bool:
"""Record a Codex ChatGPT-account 400 that rejects the current model for this account.
With a single credential there is no pool to rotate (#71970 covers that case), so the
(provider, model) pair is treated as dead for the session: the fallback walk skips it and
restore_primary_runtime stops switching back — otherwise every turn re-fails on the primary,
announces an unverified "Primary model restored", and oscillates forever (#106475).
"""
if getattr(api_error, "status_code", None) != 400:
return False
pool = getattr(agent, "_credential_pool", None)
if pool is not None and len(pool.entries()) > 1:
return False # another account in the pool may be entitled; leave rotation to it
haystack = str(getattr(api_error, "message", "") or api_error).lower()
if _CODEX_ACCOUNT_MODEL_ENTITLEMENT_MARKER not in haystack:
return False
provider = str(getattr(agent, "provider", "") or "").strip().lower()
model = str(getattr(agent, "model", "") or "").strip()
if not provider or not model:
return False
rejected = getattr(agent, "_entitlement_rejected_models", None)
if rejected is None:
rejected = agent._entitlement_rejected_models = set()
if (provider, model) in rejected:
return True
rejected.add((provider, model))
logger.warning(
"Model entitlement rejection: this account is not entitled to %s via %s; "
"treating it as unavailable for this session",
model, provider,
)
agent._buffer_status(
f"🚫 This account is not entitled to {model} via {provider}; it will be skipped "
"until restart. Switch to an entitled model via /model or `hermes model`."
)
return True
def _is_entitlement_rejected(agent, provider: str, model: str) -> bool:
"""True when (provider, model) — as configured or normalized — was rejected as unentitled
for this account (see _mark_entitlement_rejected_model)."""
rejected = getattr(agent, "_entitlement_rejected_models", None) or ()
if not rejected:
return False
if (provider, model) in rejected:
return True
from hermes_cli.model_normalize import normalize_model_for_provider
return (provider, normalize_model_for_provider(model, provider)) in rejected