fix: pin entitlement rotation through recover_with_credential_pool; bench until reset

The rotation test drove pool.mark_exhausted_and_rotate directly, so reverting the
recover_with_credential_pool model_entitlement branch stayed green. It now drives the
production recovery entry point with the classifier verdict and asserts (True, False),
the swap to cred-1 and the model-only bench on cred-0.

A ChatGPT-account entitlement is a plan property, not a quota window: the model bench
now lasts until the explicit reset path (hermes auth reset) clears model_cooldowns,
instead of re-probing the unentitled account hourly on the 429 TTL (#71970).
This commit is contained in:
teknium1
2026-09-19 01:15:49 -07:00
committed by Teknium
parent 7958cbf2de
commit 4a43cc50ae
4 changed files with 35 additions and 12 deletions

View File

@@ -1982,7 +1982,7 @@ class CredentialPool(CredentialPoolAdminMixin, CredentialPoolModelCooldownMixin)
# A generic Anthropic 429 (per-model rate limit) or a Codex account model
# entitlement rejection: bench this model only, the credential stays
# available for its siblings.
self._cool_down_model(entry, model, error_context)
self._cool_down_model(entry, model, error_context, failure_reason=failure_reason)
logger.info("credential pool: %s unavailable for model %s; other models stay available", _label, model)
self._current_id = None
next_entry, _pending = self._select_unlocked(refresh=False, model=model)

View File

@@ -14,6 +14,10 @@ from typing import Any, Dict, Optional, TYPE_CHECKING
if TYPE_CHECKING:
from agent.credential_pool import PooledCredential
# A Codex ChatGPT-account model entitlement 400 is a plan property, not a window: bench the
# (credential, model) pair until an explicit ``hermes auth reset`` clears model_cooldowns (#71970).
MODEL_ENTITLEMENT_BENCH_SECONDS = 365 * 24 * 60 * 60
def model_cooldown_until(entry: "PooledCredential", model: Optional[str]) -> Optional[float]:
"""Active cooldown blocking *entry* for *model*, or ``None``.
@@ -72,19 +76,25 @@ class CredentialPoolModelCooldownMixin:
def _cool_down_model(
self, entry: "PooledCredential", model: str, error_context: Optional[Dict[str, Any]],
failure_reason: Optional[str] = None,
) -> None:
"""Record a cooldown for *model* on *entry* and every sibling sharing its key.
Same TTL policy as a credential-wide 429 (provider ``reset_at`` wins, a
sole credential keeps its short bench). Siblings matter because a
``model_config`` twin seeded from the same key would otherwise be
re-selected for the very model that just failed. Caller holds the lock.
sole credential keeps its short bench), except a ``model_entitlement``
rejection, which stays benched until the explicit reset path clears it.
Siblings matter because a ``model_config`` twin seeded from the same key
would otherwise be re-selected for the very model that just failed.
Caller holds the lock.
"""
from agent.credential_pool import _exhausted_ttl, _normalize_error_context
until = _normalize_error_context(error_context).get("reset_at") or (
time.time() + _exhausted_ttl(429, sole_credential=self._is_sole_credential())
)
if failure_reason == "model_entitlement":
until = time.time() + MODEL_ENTITLEMENT_BENCH_SECONDS
else:
until = _normalize_error_context(error_context).get("reset_at") or (
time.time() + _exhausted_ttl(429, sole_credential=self._is_sole_credential())
)
failed_key = entry.runtime_api_key
for scoped in list(self._entries):
if scoped.id != entry.id and not (failed_key and scoped.runtime_api_key == failed_key):

View File

@@ -5,10 +5,13 @@ back; every other 400 stays a plain request failure. Once every entry rejects th
single-credential handling from #106475 takes over.
"""
import json
import time
import types
from unittest.mock import MagicMock
import pytest
from agent.agent_runtime_helpers import recover_with_credential_pool
from agent.error_classifier import FailoverReason, classify_api_error
MODEL = "gpt-5.3-codex"
@@ -56,16 +59,26 @@ def test_entitlement_400_benches_only_that_model_and_rotates(pool):
generic = classify_api_error(_Err(400, {"detail": "Invalid request: bad field"}), provider="openai-codex", model=MODEL)
assert generic.reason == FailoverReason.format_error and not generic.should_rotate_credential
# Drive the production recovery entry point (turn recovery -> recover_with_credential_pool),
# not the pool directly: the classifier verdict must reach the model-scoped bench.
assert pool.select(model=MODEL).id == "cred-0"
next_entry = pool.mark_exhausted_and_rotate(
status_code=400, api_key_hint=TOKENS[0], credential_id="cred-0",
failure_reason=verdict.reason.value, model=MODEL,
agent = types.SimpleNamespace(
provider="openai-codex", model=MODEL, base_url="https://chatgpt.com/backend-api/codex",
api_key=TOKENS[0], _credential_pool=pool, _credential_pool_entry_id="cred-0",
_swap_credential=MagicMock(return_value=True),
)
assert next_entry is not None and next_entry.id == "cred-1"
assert recover_with_credential_pool(
agent, status_code=400, has_retried_429=False, classified_reason=verdict.reason,
) == (True, False)
agent._swap_credential.assert_called_once()
assert agent._swap_credential.call_args.args[0].id == "cred-1"
first = pool.entries()[0]
assert first.last_status is None # credential-wide state untouched: other models stay usable
assert set(first.model_cooldowns) == {MODEL}
# An entitlement is a plan property, not a window: no hourly re-probe, only reset clears it.
assert first.model_cooldowns[MODEL] > time.time() + 24 * 3600
assert pool.select(model=OTHER_MODEL).id == "cred-0"
assert pool.reset_statuses() >= 1 and not pool.entries()[0].model_cooldowns
def test_all_entries_rejecting_falls_back_to_session_marker(pool):

View File

@@ -178,7 +178,7 @@ The pool handles different errors differently:
| **429 Rate Limit** | Retry same key once (transient). Second consecutive 429 rotates to next key | 1 hour |
| **402 Billing/Quota** | Immediately rotate to next key | 1 hour |
| **401 Auth Expired** | Try refreshing the OAuth token first. Rotate only if refresh fails | 5 minutes |
| **400 Codex model entitlement** (`The '<model>' model is not supported when using Codex with a ChatGPT account.`) | Bench this key for the rejected model only and rotate to the next key; other models keep using the key. Other 400s never rotate | 1 hour (per model) |
| **400 Codex model entitlement** (`The '<model>' model is not supported when using Codex with a ChatGPT account.`) | Bench this key for the rejected model only and rotate to the next key; other models keep using the key. Other 400s never rotate | Until `hermes auth reset` (per model; an entitlement is a plan property, not a window) |
| **All keys exhausted** | Fall through to `fallback_model` if configured | — |
Provider-supplied `reset_at` timestamps override these default cooldowns.