diff --git a/agent/credential_pool.py b/agent/credential_pool.py index a9df6fc866..2cfcb80363 100644 --- a/agent/credential_pool.py +++ b/agent/credential_pool.py @@ -1982,7 +1982,7 @@ class CredentialPool(CredentialPoolAdminMixin, CredentialPoolModelCooldownMixin) # A generic Anthropic 429 (per-model rate limit) or a Codex account model # entitlement rejection: bench this model only, the credential stays # available for its siblings. - self._cool_down_model(entry, model, error_context) + self._cool_down_model(entry, model, error_context, failure_reason=failure_reason) logger.info("credential pool: %s unavailable for model %s; other models stay available", _label, model) self._current_id = None next_entry, _pending = self._select_unlocked(refresh=False, model=model) diff --git a/agent/credential_pool_model_cooldowns.py b/agent/credential_pool_model_cooldowns.py index fa6da883ca..d74d1ded1b 100644 --- a/agent/credential_pool_model_cooldowns.py +++ b/agent/credential_pool_model_cooldowns.py @@ -14,6 +14,10 @@ from typing import Any, Dict, Optional, TYPE_CHECKING if TYPE_CHECKING: from agent.credential_pool import PooledCredential +# A Codex ChatGPT-account model entitlement 400 is a plan property, not a window: bench the +# (credential, model) pair until an explicit ``hermes auth reset`` clears model_cooldowns (#71970). +MODEL_ENTITLEMENT_BENCH_SECONDS = 365 * 24 * 60 * 60 + def model_cooldown_until(entry: "PooledCredential", model: Optional[str]) -> Optional[float]: """Active cooldown blocking *entry* for *model*, or ``None``. @@ -72,19 +76,25 @@ class CredentialPoolModelCooldownMixin: def _cool_down_model( self, entry: "PooledCredential", model: str, error_context: Optional[Dict[str, Any]], + failure_reason: Optional[str] = None, ) -> None: """Record a cooldown for *model* on *entry* and every sibling sharing its key. Same TTL policy as a credential-wide 429 (provider ``reset_at`` wins, a - sole credential keeps its short bench). Siblings matter because a - ``model_config`` twin seeded from the same key would otherwise be - re-selected for the very model that just failed. Caller holds the lock. + sole credential keeps its short bench), except a ``model_entitlement`` + rejection, which stays benched until the explicit reset path clears it. + Siblings matter because a ``model_config`` twin seeded from the same key + would otherwise be re-selected for the very model that just failed. + Caller holds the lock. """ from agent.credential_pool import _exhausted_ttl, _normalize_error_context - until = _normalize_error_context(error_context).get("reset_at") or ( - time.time() + _exhausted_ttl(429, sole_credential=self._is_sole_credential()) - ) + if failure_reason == "model_entitlement": + until = time.time() + MODEL_ENTITLEMENT_BENCH_SECONDS + else: + until = _normalize_error_context(error_context).get("reset_at") or ( + time.time() + _exhausted_ttl(429, sole_credential=self._is_sole_credential()) + ) failed_key = entry.runtime_api_key for scoped in list(self._entries): if scoped.id != entry.id and not (failed_key and scoped.runtime_api_key == failed_key): diff --git a/tests/agent/test_codex_model_entitlement_rotation.py b/tests/agent/test_codex_model_entitlement_rotation.py index 367e7c6f8d..1777a6bb60 100644 --- a/tests/agent/test_codex_model_entitlement_rotation.py +++ b/tests/agent/test_codex_model_entitlement_rotation.py @@ -5,10 +5,13 @@ back; every other 400 stays a plain request failure. Once every entry rejects th single-credential handling from #106475 takes over. """ import json +import time import types +from unittest.mock import MagicMock import pytest +from agent.agent_runtime_helpers import recover_with_credential_pool from agent.error_classifier import FailoverReason, classify_api_error MODEL = "gpt-5.3-codex" @@ -56,16 +59,26 @@ def test_entitlement_400_benches_only_that_model_and_rotates(pool): generic = classify_api_error(_Err(400, {"detail": "Invalid request: bad field"}), provider="openai-codex", model=MODEL) assert generic.reason == FailoverReason.format_error and not generic.should_rotate_credential + # Drive the production recovery entry point (turn recovery -> recover_with_credential_pool), + # not the pool directly: the classifier verdict must reach the model-scoped bench. assert pool.select(model=MODEL).id == "cred-0" - next_entry = pool.mark_exhausted_and_rotate( - status_code=400, api_key_hint=TOKENS[0], credential_id="cred-0", - failure_reason=verdict.reason.value, model=MODEL, + agent = types.SimpleNamespace( + provider="openai-codex", model=MODEL, base_url="https://chatgpt.com/backend-api/codex", + api_key=TOKENS[0], _credential_pool=pool, _credential_pool_entry_id="cred-0", + _swap_credential=MagicMock(return_value=True), ) - assert next_entry is not None and next_entry.id == "cred-1" + assert recover_with_credential_pool( + agent, status_code=400, has_retried_429=False, classified_reason=verdict.reason, + ) == (True, False) + agent._swap_credential.assert_called_once() + assert agent._swap_credential.call_args.args[0].id == "cred-1" first = pool.entries()[0] assert first.last_status is None # credential-wide state untouched: other models stay usable assert set(first.model_cooldowns) == {MODEL} + # An entitlement is a plan property, not a window: no hourly re-probe, only reset clears it. + assert first.model_cooldowns[MODEL] > time.time() + 24 * 3600 assert pool.select(model=OTHER_MODEL).id == "cred-0" + assert pool.reset_statuses() >= 1 and not pool.entries()[0].model_cooldowns def test_all_entries_rejecting_falls_back_to_session_marker(pool): diff --git a/website/docs/user-guide/features/credential-pools.md b/website/docs/user-guide/features/credential-pools.md index 90ff0d5a3b..f8c4505cab 100644 --- a/website/docs/user-guide/features/credential-pools.md +++ b/website/docs/user-guide/features/credential-pools.md @@ -178,7 +178,7 @@ The pool handles different errors differently: | **429 Rate Limit** | Retry same key once (transient). Second consecutive 429 rotates to next key | 1 hour | | **402 Billing/Quota** | Immediately rotate to next key | 1 hour | | **401 Auth Expired** | Try refreshing the OAuth token first. Rotate only if refresh fails | 5 minutes | -| **400 Codex model entitlement** (`The '' model is not supported when using Codex with a ChatGPT account.`) | Bench this key for the rejected model only and rotate to the next key; other models keep using the key. Other 400s never rotate | 1 hour (per model) | +| **400 Codex model entitlement** (`The '' model is not supported when using Codex with a ChatGPT account.`) | Bench this key for the rejected model only and rotate to the next key; other models keep using the key. Other 400s never rotate | Until `hermes auth reset` (per model; an entitlement is a plan property, not a window) | | **All keys exhausted** | Fall through to `fallback_model` if configured | — | Provider-supplied `reset_at` timestamps override these default cooldowns.