Follow-up to the salvaged #111787 commit (@KoNit-K), same mechanism as #75578 (@adikpb). What: - Move the per-model cooldown logic out of the 2.7k-line credential_pool.py facade into agent/credential_pool_model_cooldowns.py (mixin + module helpers), keeping only the select/has_available/next_available_at hooks and the mark_exhausted_and_rotate branch in the facade. - The model cooldown uses the same TTL policy as a credential-wide 429 (_exhausted_ttl: provider reset_at wins, a sole credential keeps its 60s bench) instead of a flat 1h, so a single-credential user is never benched LONGER for the failed model than before. - Drop the `quota_scope == "account"` check: nothing in the tree produces that key. - Also keep billing_unverified 429s credential-wide, matching the credential-wide branch. - _rebind_primary_credential_pool read `rt` that was not in its scope (NameError on every post-fallback restore); pass primary_model from the caller instead. - resolve_anthropic_token(model=...) gates only model-aware callers; model-less diagnostics (usage display, model discovery) keep the key as before. - _anthropic_token_or_raise names the cooled model instead of claiming no credentials exist. - Simplify the salvaged call sites (unconditional select(model=)/resolve_anthropic_token(model=)), widen test stubs that lacked the new kwargs, trim the tests to two invariants on a real temp store. - Docs: credential-pools.md documents per-model Anthropic 429 cooldowns. Why: a generic Anthropic 429 is a per-model rate limit; benching the whole credential took every other Claude model offline while the env/borrowed token path handed the same benched key straight back (#111769, #61451). Co-authored-by: KoNit-K <124019182+KoNit-K@users.noreply.github.com> Co-authored-by: adikpb <67222969+adikpb@users.noreply.github.com>
71 lines
3.2 KiB
Python
71 lines
3.2 KiB
Python
"""A generic Anthropic 429 benches only the model that was rate-limited (#111769, #61451).
|
|
|
|
Real ``agent.credential_pool`` against a real temp auth store: one API-key credential,
|
|
one ``mark_exhausted_and_rotate(status_code=429, model=A)``.
|
|
"""
|
|
import json
|
|
|
|
import pytest
|
|
|
|
KEY = "sk-ant-api03-synthetic-test-key-0000"
|
|
MODEL_A = "claude-sonnet-4-5"
|
|
MODEL_B = "claude-haiku-4-5"
|
|
|
|
|
|
@pytest.fixture
|
|
def pool(tmp_path, monkeypatch):
|
|
root = tmp_path / "hermes-root"
|
|
root.mkdir()
|
|
(tmp_path / "fakehome").mkdir()
|
|
monkeypatch.setenv("HOME", str(tmp_path / "fakehome"))
|
|
monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(tmp_path / "fakehome"))
|
|
for var in ("ANTHROPIC_TOKEN", "ANTHROPIC_API_KEY", "CLAUDE_CODE_OAUTH_TOKEN"):
|
|
monkeypatch.delenv(var, raising=False)
|
|
monkeypatch.setenv("HERMES_HOME", str(root))
|
|
import hermes_constants
|
|
hermes_constants._default_hermes_root_memo = None # type: ignore[attr-defined]
|
|
(root / "auth.json").write_text(json.dumps({"credential_pool": {"anthropic": [{
|
|
"id": "seat", "label": "seat", "auth_type": "api_key", "priority": 0,
|
|
"source": "manual", "access_token": KEY,
|
|
}]}}))
|
|
from agent.credential_pool import load_pool
|
|
return load_pool("anthropic")
|
|
|
|
|
|
def test_generic_429_benches_only_the_rate_limited_model(pool, monkeypatch):
|
|
from agent.anthropic_credentials import resolve_anthropic_token
|
|
from agent.credential_pool import load_pool
|
|
|
|
ctx = {"message": "This request would exceed your account's rate limit. Please try again later."}
|
|
assert pool.mark_exhausted_and_rotate(
|
|
status_code=429, error_context=ctx, api_key_hint=KEY, failure_reason="rate_limit", model=MODEL_A,
|
|
) is None # a sole credential has nothing to rotate to for model A
|
|
|
|
assert pool.entries()[0].last_status is None # credential-wide state untouched
|
|
assert pool.select(model=MODEL_A) is None
|
|
assert pool.select(model=MODEL_B) is not None
|
|
assert pool.select() is None # a caller that names no model honours every active cooldown
|
|
assert pool.next_available_at(model=MODEL_A) is not None
|
|
assert pool.next_available_at(model=MODEL_B) is None
|
|
|
|
# The cooldown is persisted, so another process (and the env/borrowed token resolver,
|
|
# which reads the store fresh) sees the same per-model verdict.
|
|
fresh = load_pool("anthropic")
|
|
assert fresh.select(model=MODEL_A) is None and fresh.select(model=MODEL_B) is not None
|
|
monkeypatch.setenv("ANTHROPIC_API_KEY", KEY)
|
|
assert resolve_anthropic_token(model=MODEL_A) is None
|
|
assert resolve_anthropic_token(model=MODEL_B) == KEY
|
|
assert resolve_anthropic_token() == KEY # model-less diagnostics keep the key
|
|
|
|
|
|
@pytest.mark.parametrize("status_code, failure_reason", [(401, None), (402, "billing"), (429, "billing")])
|
|
def test_auth_and_billing_failures_stay_credential_wide(pool, status_code, failure_reason):
|
|
from agent.credential_pool import STATUS_EXHAUSTED
|
|
|
|
pool.mark_exhausted_and_rotate(
|
|
status_code=status_code, api_key_hint=KEY, failure_reason=failure_reason, model=MODEL_A,
|
|
)
|
|
assert pool.entries()[0].last_status == STATUS_EXHAUSTED
|
|
assert not pool.entries()[0].model_cooldowns
|
|
assert pool.select(model=MODEL_B) is None
|