Files
hermes-agent/tests/agent/test_anthropic_model_cooldowns.py
teknium1 6de6e6da99 fix(anthropic): model-scoped 429 cooldowns live in a pool sibling; restore path no longer NameErrors
Follow-up to the salvaged #111787 commit (@KoNit-K), same mechanism as #75578 (@adikpb).

What:
- Move the per-model cooldown logic out of the 2.7k-line credential_pool.py facade into
  agent/credential_pool_model_cooldowns.py (mixin + module helpers), keeping only the
  select/has_available/next_available_at hooks and the mark_exhausted_and_rotate branch in the facade.
- The model cooldown uses the same TTL policy as a credential-wide 429 (_exhausted_ttl: provider
  reset_at wins, a sole credential keeps its 60s bench) instead of a flat 1h, so a single-credential
  user is never benched LONGER for the failed model than before.
- Drop the `quota_scope == "account"` check: nothing in the tree produces that key.
- Also keep billing_unverified 429s credential-wide, matching the credential-wide branch.
- _rebind_primary_credential_pool read `rt` that was not in its scope (NameError on every
  post-fallback restore); pass primary_model from the caller instead.
- resolve_anthropic_token(model=...) gates only model-aware callers; model-less diagnostics
  (usage display, model discovery) keep the key as before.
- _anthropic_token_or_raise names the cooled model instead of claiming no credentials exist.
- Simplify the salvaged call sites (unconditional select(model=)/resolve_anthropic_token(model=)),
  widen test stubs that lacked the new kwargs, trim the tests to two invariants on a real temp store.
- Docs: credential-pools.md documents per-model Anthropic 429 cooldowns.

Why: a generic Anthropic 429 is a per-model rate limit; benching the whole credential took every
other Claude model offline while the env/borrowed token path handed the same benched key straight
back (#111769, #61451).

Co-authored-by: KoNit-K <124019182+KoNit-K@users.noreply.github.com>
Co-authored-by: adikpb <67222969+adikpb@users.noreply.github.com>
2026-09-16 17:16:06 -07:00

71 lines
3.2 KiB
Python

"""A generic Anthropic 429 benches only the model that was rate-limited (#111769, #61451).
Real ``agent.credential_pool`` against a real temp auth store: one API-key credential,
one ``mark_exhausted_and_rotate(status_code=429, model=A)``.
"""
import json
import pytest
KEY = "sk-ant-api03-synthetic-test-key-0000"
MODEL_A = "claude-sonnet-4-5"
MODEL_B = "claude-haiku-4-5"
@pytest.fixture
def pool(tmp_path, monkeypatch):
root = tmp_path / "hermes-root"
root.mkdir()
(tmp_path / "fakehome").mkdir()
monkeypatch.setenv("HOME", str(tmp_path / "fakehome"))
monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(tmp_path / "fakehome"))
for var in ("ANTHROPIC_TOKEN", "ANTHROPIC_API_KEY", "CLAUDE_CODE_OAUTH_TOKEN"):
monkeypatch.delenv(var, raising=False)
monkeypatch.setenv("HERMES_HOME", str(root))
import hermes_constants
hermes_constants._default_hermes_root_memo = None # type: ignore[attr-defined]
(root / "auth.json").write_text(json.dumps({"credential_pool": {"anthropic": [{
"id": "seat", "label": "seat", "auth_type": "api_key", "priority": 0,
"source": "manual", "access_token": KEY,
}]}}))
from agent.credential_pool import load_pool
return load_pool("anthropic")
def test_generic_429_benches_only_the_rate_limited_model(pool, monkeypatch):
from agent.anthropic_credentials import resolve_anthropic_token
from agent.credential_pool import load_pool
ctx = {"message": "This request would exceed your account's rate limit. Please try again later."}
assert pool.mark_exhausted_and_rotate(
status_code=429, error_context=ctx, api_key_hint=KEY, failure_reason="rate_limit", model=MODEL_A,
) is None # a sole credential has nothing to rotate to for model A
assert pool.entries()[0].last_status is None # credential-wide state untouched
assert pool.select(model=MODEL_A) is None
assert pool.select(model=MODEL_B) is not None
assert pool.select() is None # a caller that names no model honours every active cooldown
assert pool.next_available_at(model=MODEL_A) is not None
assert pool.next_available_at(model=MODEL_B) is None
# The cooldown is persisted, so another process (and the env/borrowed token resolver,
# which reads the store fresh) sees the same per-model verdict.
fresh = load_pool("anthropic")
assert fresh.select(model=MODEL_A) is None and fresh.select(model=MODEL_B) is not None
monkeypatch.setenv("ANTHROPIC_API_KEY", KEY)
assert resolve_anthropic_token(model=MODEL_A) is None
assert resolve_anthropic_token(model=MODEL_B) == KEY
assert resolve_anthropic_token() == KEY # model-less diagnostics keep the key
@pytest.mark.parametrize("status_code, failure_reason", [(401, None), (402, "billing"), (429, "billing")])
def test_auth_and_billing_failures_stay_credential_wide(pool, status_code, failure_reason):
from agent.credential_pool import STATUS_EXHAUSTED
pool.mark_exhausted_and_rotate(
status_code=status_code, api_key_hint=KEY, failure_reason=failure_reason, model=MODEL_A,
)
assert pool.entries()[0].last_status == STATUS_EXHAUSTED
assert not pool.entries()[0].model_cooldowns
assert pool.select(model=MODEL_B) is None