Files
hermes-agent/agent/credential_pool_model_cooldowns.py
teknium1 6de6e6da99 fix(anthropic): model-scoped 429 cooldowns live in a pool sibling; restore path no longer NameErrors
Follow-up to the salvaged #111787 commit (@KoNit-K), same mechanism as #75578 (@adikpb).

What:
- Move the per-model cooldown logic out of the 2.7k-line credential_pool.py facade into
  agent/credential_pool_model_cooldowns.py (mixin + module helpers), keeping only the
  select/has_available/next_available_at hooks and the mark_exhausted_and_rotate branch in the facade.
- The model cooldown uses the same TTL policy as a credential-wide 429 (_exhausted_ttl: provider
  reset_at wins, a sole credential keeps its 60s bench) instead of a flat 1h, so a single-credential
  user is never benched LONGER for the failed model than before.
- Drop the `quota_scope == "account"` check: nothing in the tree produces that key.
- Also keep billing_unverified 429s credential-wide, matching the credential-wide branch.
- _rebind_primary_credential_pool read `rt` that was not in its scope (NameError on every
  post-fallback restore); pass primary_model from the caller instead.
- resolve_anthropic_token(model=...) gates only model-aware callers; model-less diagnostics
  (usage display, model discovery) keep the key as before.
- _anthropic_token_or_raise names the cooled model instead of claiming no credentials exist.
- Simplify the salvaged call sites (unconditional select(model=)/resolve_anthropic_token(model=)),
  widen test stubs that lacked the new kwargs, trim the tests to two invariants on a real temp store.
- Docs: credential-pools.md documents per-model Anthropic 429 cooldowns.

Why: a generic Anthropic 429 is a per-model rate limit; benching the whole credential took every
other Claude model offline while the env/borrowed token path handed the same benched key straight
back (#111769, #61451).

Co-authored-by: KoNit-K <124019182+KoNit-K@users.noreply.github.com>
Co-authored-by: adikpb <67222969+adikpb@users.noreply.github.com>
2026-09-16 17:16:06 -07:00

89 lines
3.9 KiB
Python

"""Model-scoped rate-limit cooldowns for pooled credentials.
Anthropic enforces its API rate limits (requests / tokens per minute) per
model, so a generic 429 for one Claude model says nothing about the same
credential's standing for its sibling models. Such a 429 is recorded as a
cooldown on the requested model only, beside the credential-wide status that
auth, billing and payment failures keep benching the whole credential with.
"""
from __future__ import annotations
import time
from typing import Any, Dict, Optional, TYPE_CHECKING
if TYPE_CHECKING:
from agent.credential_pool import PooledCredential
def model_cooldown_until(entry: "PooledCredential", model: Optional[str]) -> Optional[float]:
"""Active cooldown blocking *entry* for *model*, or ``None``.
Callers that do not know the model stay conservative: any active model
cooldown blocks them, so an unscoped route cannot reuse the credential.
"""
cooldowns = entry.model_cooldowns or {}
values = cooldowns.values() if not model else (cooldowns.get(model),)
now = time.time()
active = [float(until) for until in values if isinstance(until, (int, float)) and until > now]
return max(active) if active else None
def merge_model_cooldowns(*maps: Any) -> Dict[str, float]:
"""Latest reset per model across snapshots — each writer only observed its own model."""
merged: Dict[str, float] = {}
for cooldowns in maps:
if not isinstance(cooldowns, dict):
continue
for model, until in cooldowns.items():
if isinstance(until, (int, float)):
merged[model] = max(float(until), merged.get(model, 0.0))
return merged
class CredentialPoolModelCooldownMixin:
def token_is_blocked(self, token: str, *, model: Optional[str] = None) -> bool:
"""Whether a pool cooldown blocks *token* for *model*.
Closes the paths that hand out a native Anthropic token without
selecting it from the pool (env / borrowed credentials). Tokens the
pool does not know fail open: no row can attribute a cooldown to them.
"""
with self._lock:
return any(
entry.runtime_api_key == token and model_cooldown_until(entry, model) is not None
for entry in self._entries
)
def _is_model_scoped_rate_limit(
self, status_code: Optional[int], model: Optional[str], failure_reason: Optional[str],
) -> bool:
from agent.credential_pool import FAILURE_REASON_BILLING, FAILURE_REASON_BILLING_UNVERIFIED
return (
self.provider == "anthropic" and status_code == 429 and bool(model)
and failure_reason not in (FAILURE_REASON_BILLING, FAILURE_REASON_BILLING_UNVERIFIED)
)
def _cool_down_model(
self, entry: "PooledCredential", model: str, error_context: Optional[Dict[str, Any]],
) -> None:
"""Record a cooldown for *model* on *entry* and every sibling sharing its key.
Same TTL policy as a credential-wide 429 (provider ``reset_at`` wins, a
sole credential keeps its short bench). Siblings matter because a
``model_config`` twin seeded from the same key would otherwise be
re-selected for the very model that just failed. Caller holds the lock.
"""
from agent.credential_pool import _exhausted_ttl, _normalize_error_context
until = _normalize_error_context(error_context).get("reset_at") or (
time.time() + _exhausted_ttl(429, sole_credential=self._is_sole_credential())
)
failed_key = entry.runtime_api_key
for scoped in list(self._entries):
if scoped.id != entry.id and not (failed_key and scoped.runtime_api_key == failed_key):
continue
cooldowns = merge_model_cooldowns(scoped.model_cooldowns, {model: until})
self._adopt(scoped, persist=False, model_cooldowns=cooldowns)
self._persist()