Files
hermes-agent/agent/fallback_cooldown.py
teknium1 7958cbf2de fix: rotate Codex pool credentials on a ChatGPT-account model entitlement 400
The exact `The '<model>' model is not supported when using Codex with a
ChatGPT account.` 400 classified as format_error, so a two-entry openai-codex
pool never tried its second account: the turn failed as a malformed request
even though the other account was entitled (#71970). #106475 covered the
single-credential case and explicitly left the multi-entry case to rotation,
but no rotation branch existed for it.

- classify the exact normalized text as FailoverReason.model_entitlement
  (rotate + fallback, never retry); arbitrary 400s stay format_error
- recover_with_credential_pool rotates once on that reason; the pool records
  it through the existing model_cooldowns path (Anthropic per-model 429), so
  only (credential, model) is benched: other models keep using the account,
  and `hermes auth` reset clears the marker with everything else
- _mark_entitlement_rejected_model gates on pool.has_available(model=...)
  instead of entry count, so once every account rejects the model it falls
  back to the #106475 session marker (fallback walk skips it, no oscillation)

Salvages the design of #71973 by @kilhyeonjun on the current classifier
tables and the model-cooldown substrate that landed since.

Co-authored-by: kilhyeonjun <kboxstar@gmail.com>
2026-09-19 09:37:23 -07:00

83 lines
4.2 KiB
Python

"""Primary rate-limit cooldown arming and per-session model rejection markers, shared by the
fallback walk (chat_completion_helpers) and restore_primary_runtime (agent_runtime_helpers)."""
import logging
import time
from agent.error_classifier import FailoverReason
logger = logging.getLogger(__name__)
_RATE_LIMIT_FAILOVER_REASONS = frozenset({FailoverReason.rate_limit, FailoverReason.billing, FailoverReason.upstream_rate_limit})
def _arm_rate_limit_cooldown(agent, reason: "FailoverReason | None") -> int | None:
"""Arm the primary's exponential cooldown (60s → 2m → ... → 4h cap) on CONSECUTIVE rate-limits;
restore_primary_runtime resets the counter. Only when leaving the primary: chain-switching from
an active fallback means the primary was not the 429 source, so its cooldown is left alone.
Return the armed cooldown in seconds, or None when no cooldown was armed."""
if reason not in _RATE_LIMIT_FAILOVER_REASONS:
return None
current_provider = (getattr(agent, "provider", "") or "").strip().lower()
primary_provider = ((agent._primary_runtime or {}).get("provider") or "").strip().lower()
if getattr(agent, "_fallback_activated", False) and not (primary_provider and current_provider == primary_provider):
return None
backoff_count = getattr(agent, "_rate_limit_backoff_count", 0)
agent._rate_limit_backoff_count = backoff_count + 1
backoff_seconds = min(60 * (2 ** backoff_count), 14400)
agent._rate_limited_until = time.monotonic() + backoff_seconds
logging.info("Rate-limit backoff level %d: cooldown %d s (%.1f min, backoff#%d)", backoff_count, backoff_seconds, backoff_seconds / 60, backoff_count + 1)
return backoff_seconds
def _mark_entitlement_rejected_model(agent, api_error) -> bool:
"""Record a Codex ChatGPT-account 400 that rejects the current model for this account.
Pool rotation runs first (recover_with_credential_pool benches (credential, model) and
moves to the next entitled entry, #71970); this runs only once no pool entry is left for
the model, so the (provider, model) pair is treated as dead for the session: the fallback
walk skips it and restore_primary_runtime stops switching back — otherwise every turn
re-fails on the primary, announces an unverified "Primary model restored", and oscillates
forever (#106475).
"""
if getattr(api_error, "status_code", None) != 400:
return False
from agent.error_classifier import CODEX_ACCOUNT_MODEL_ENTITLEMENT_MARKER
haystack = str(getattr(api_error, "message", "") or api_error).lower()
if CODEX_ACCOUNT_MODEL_ENTITLEMENT_MARKER not in haystack:
return False
provider = str(getattr(agent, "provider", "") or "").strip().lower()
model = str(getattr(agent, "model", "") or "").strip()
if not provider or not model:
return False
pool = getattr(agent, "_credential_pool", None)
if pool is not None and pool.has_available(model=model):
return False # another pool entry is still eligible for this model; rotation owns it
rejected = getattr(agent, "_entitlement_rejected_models", None)
if rejected is None:
rejected = agent._entitlement_rejected_models = set()
if (provider, model) in rejected:
return True
rejected.add((provider, model))
logger.warning(
"Model entitlement rejection: this account is not entitled to %s via %s; "
"treating it as unavailable for this session",
model, provider,
)
agent._buffer_diagnostic_status(
f"🚫 This account is not entitled to {model} via {provider}; it will be skipped "
"until restart. Switch to an entitled model via /model or `hermes model`."
)
return True
def _is_entitlement_rejected(agent, provider: str, model: str) -> bool:
"""True when (provider, model) — as configured or normalized — was rejected as unentitled
for this account (see _mark_entitlement_rejected_model)."""
rejected = getattr(agent, "_entitlement_rejected_models", None) or ()
if not rejected:
return False
if (provider, model) in rejected:
return True
from hermes_cli.model_normalize import normalize_model_for_provider
return (provider, normalize_model_for_provider(model, provider)) in rejected