Follow-up to the salvaged #111787 commit (@KoNit-K), same mechanism as #75578 (@adikpb). What: - Move the per-model cooldown logic out of the 2.7k-line credential_pool.py facade into agent/credential_pool_model_cooldowns.py (mixin + module helpers), keeping only the select/has_available/next_available_at hooks and the mark_exhausted_and_rotate branch in the facade. - The model cooldown uses the same TTL policy as a credential-wide 429 (_exhausted_ttl: provider reset_at wins, a sole credential keeps its 60s bench) instead of a flat 1h, so a single-credential user is never benched LONGER for the failed model than before. - Drop the `quota_scope == "account"` check: nothing in the tree produces that key. - Also keep billing_unverified 429s credential-wide, matching the credential-wide branch. - _rebind_primary_credential_pool read `rt` that was not in its scope (NameError on every post-fallback restore); pass primary_model from the caller instead. - resolve_anthropic_token(model=...) gates only model-aware callers; model-less diagnostics (usage display, model discovery) keep the key as before. - _anthropic_token_or_raise names the cooled model instead of claiming no credentials exist. - Simplify the salvaged call sites (unconditional select(model=)/resolve_anthropic_token(model=)), widen test stubs that lacked the new kwargs, trim the tests to two invariants on a real temp store. - Docs: credential-pools.md documents per-model Anthropic 429 cooldowns. Why: a generic Anthropic 429 is a per-model rate limit; benching the whole credential took every other Claude model offline while the env/borrowed token path handed the same benched key straight back (#111769, #61451). Co-authored-by: KoNit-K <124019182+KoNit-K@users.noreply.github.com> Co-authored-by: adikpb <67222969+adikpb@users.noreply.github.com>
89 lines
3.9 KiB
Python
89 lines
3.9 KiB
Python
"""Model-scoped rate-limit cooldowns for pooled credentials.
|
|
|
|
Anthropic enforces its API rate limits (requests / tokens per minute) per
|
|
model, so a generic 429 for one Claude model says nothing about the same
|
|
credential's standing for its sibling models. Such a 429 is recorded as a
|
|
cooldown on the requested model only, beside the credential-wide status that
|
|
auth, billing and payment failures keep benching the whole credential with.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import time
|
|
from typing import Any, Dict, Optional, TYPE_CHECKING
|
|
|
|
if TYPE_CHECKING:
|
|
from agent.credential_pool import PooledCredential
|
|
|
|
|
|
def model_cooldown_until(entry: "PooledCredential", model: Optional[str]) -> Optional[float]:
|
|
"""Active cooldown blocking *entry* for *model*, or ``None``.
|
|
|
|
Callers that do not know the model stay conservative: any active model
|
|
cooldown blocks them, so an unscoped route cannot reuse the credential.
|
|
"""
|
|
cooldowns = entry.model_cooldowns or {}
|
|
values = cooldowns.values() if not model else (cooldowns.get(model),)
|
|
now = time.time()
|
|
active = [float(until) for until in values if isinstance(until, (int, float)) and until > now]
|
|
return max(active) if active else None
|
|
|
|
|
|
def merge_model_cooldowns(*maps: Any) -> Dict[str, float]:
|
|
"""Latest reset per model across snapshots — each writer only observed its own model."""
|
|
merged: Dict[str, float] = {}
|
|
for cooldowns in maps:
|
|
if not isinstance(cooldowns, dict):
|
|
continue
|
|
for model, until in cooldowns.items():
|
|
if isinstance(until, (int, float)):
|
|
merged[model] = max(float(until), merged.get(model, 0.0))
|
|
return merged
|
|
|
|
|
|
class CredentialPoolModelCooldownMixin:
|
|
def token_is_blocked(self, token: str, *, model: Optional[str] = None) -> bool:
|
|
"""Whether a pool cooldown blocks *token* for *model*.
|
|
|
|
Closes the paths that hand out a native Anthropic token without
|
|
selecting it from the pool (env / borrowed credentials). Tokens the
|
|
pool does not know fail open: no row can attribute a cooldown to them.
|
|
"""
|
|
with self._lock:
|
|
return any(
|
|
entry.runtime_api_key == token and model_cooldown_until(entry, model) is not None
|
|
for entry in self._entries
|
|
)
|
|
|
|
def _is_model_scoped_rate_limit(
|
|
self, status_code: Optional[int], model: Optional[str], failure_reason: Optional[str],
|
|
) -> bool:
|
|
from agent.credential_pool import FAILURE_REASON_BILLING, FAILURE_REASON_BILLING_UNVERIFIED
|
|
|
|
return (
|
|
self.provider == "anthropic" and status_code == 429 and bool(model)
|
|
and failure_reason not in (FAILURE_REASON_BILLING, FAILURE_REASON_BILLING_UNVERIFIED)
|
|
)
|
|
|
|
def _cool_down_model(
|
|
self, entry: "PooledCredential", model: str, error_context: Optional[Dict[str, Any]],
|
|
) -> None:
|
|
"""Record a cooldown for *model* on *entry* and every sibling sharing its key.
|
|
|
|
Same TTL policy as a credential-wide 429 (provider ``reset_at`` wins, a
|
|
sole credential keeps its short bench). Siblings matter because a
|
|
``model_config`` twin seeded from the same key would otherwise be
|
|
re-selected for the very model that just failed. Caller holds the lock.
|
|
"""
|
|
from agent.credential_pool import _exhausted_ttl, _normalize_error_context
|
|
|
|
until = _normalize_error_context(error_context).get("reset_at") or (
|
|
time.time() + _exhausted_ttl(429, sole_credential=self._is_sole_credential())
|
|
)
|
|
failed_key = entry.runtime_api_key
|
|
for scoped in list(self._entries):
|
|
if scoped.id != entry.id and not (failed_key and scoped.runtime_api_key == failed_key):
|
|
continue
|
|
cooldowns = merge_model_cooldowns(scoped.model_cooldowns, {model: until})
|
|
self._adopt(scoped, persist=False, model_cooldowns=cooldowns)
|
|
self._persist()
|