fix: pin entitlement rotation through recover_with_credential_pool; bench until reset
The rotation test drove pool.mark_exhausted_and_rotate directly, so reverting the recover_with_credential_pool model_entitlement branch stayed green. It now drives the production recovery entry point with the classifier verdict and asserts (True, False), the swap to cred-1 and the model-only bench on cred-0. A ChatGPT-account entitlement is a plan property, not a quota window: the model bench now lasts until the explicit reset path (hermes auth reset) clears model_cooldowns, instead of re-probing the unentitled account hourly on the 429 TTL (#71970).
This commit is contained in:
@@ -1982,7 +1982,7 @@ class CredentialPool(CredentialPoolAdminMixin, CredentialPoolModelCooldownMixin)
|
||||
# A generic Anthropic 429 (per-model rate limit) or a Codex account model
|
||||
# entitlement rejection: bench this model only, the credential stays
|
||||
# available for its siblings.
|
||||
self._cool_down_model(entry, model, error_context)
|
||||
self._cool_down_model(entry, model, error_context, failure_reason=failure_reason)
|
||||
logger.info("credential pool: %s unavailable for model %s; other models stay available", _label, model)
|
||||
self._current_id = None
|
||||
next_entry, _pending = self._select_unlocked(refresh=False, model=model)
|
||||
|
||||
@@ -14,6 +14,10 @@ from typing import Any, Dict, Optional, TYPE_CHECKING
|
||||
if TYPE_CHECKING:
|
||||
from agent.credential_pool import PooledCredential
|
||||
|
||||
# A Codex ChatGPT-account model entitlement 400 is a plan property, not a window: bench the
|
||||
# (credential, model) pair until an explicit ``hermes auth reset`` clears model_cooldowns (#71970).
|
||||
MODEL_ENTITLEMENT_BENCH_SECONDS = 365 * 24 * 60 * 60
|
||||
|
||||
|
||||
def model_cooldown_until(entry: "PooledCredential", model: Optional[str]) -> Optional[float]:
|
||||
"""Active cooldown blocking *entry* for *model*, or ``None``.
|
||||
@@ -72,19 +76,25 @@ class CredentialPoolModelCooldownMixin:
|
||||
|
||||
def _cool_down_model(
|
||||
self, entry: "PooledCredential", model: str, error_context: Optional[Dict[str, Any]],
|
||||
failure_reason: Optional[str] = None,
|
||||
) -> None:
|
||||
"""Record a cooldown for *model* on *entry* and every sibling sharing its key.
|
||||
|
||||
Same TTL policy as a credential-wide 429 (provider ``reset_at`` wins, a
|
||||
sole credential keeps its short bench). Siblings matter because a
|
||||
``model_config`` twin seeded from the same key would otherwise be
|
||||
re-selected for the very model that just failed. Caller holds the lock.
|
||||
sole credential keeps its short bench), except a ``model_entitlement``
|
||||
rejection, which stays benched until the explicit reset path clears it.
|
||||
Siblings matter because a ``model_config`` twin seeded from the same key
|
||||
would otherwise be re-selected for the very model that just failed.
|
||||
Caller holds the lock.
|
||||
"""
|
||||
from agent.credential_pool import _exhausted_ttl, _normalize_error_context
|
||||
|
||||
until = _normalize_error_context(error_context).get("reset_at") or (
|
||||
time.time() + _exhausted_ttl(429, sole_credential=self._is_sole_credential())
|
||||
)
|
||||
if failure_reason == "model_entitlement":
|
||||
until = time.time() + MODEL_ENTITLEMENT_BENCH_SECONDS
|
||||
else:
|
||||
until = _normalize_error_context(error_context).get("reset_at") or (
|
||||
time.time() + _exhausted_ttl(429, sole_credential=self._is_sole_credential())
|
||||
)
|
||||
failed_key = entry.runtime_api_key
|
||||
for scoped in list(self._entries):
|
||||
if scoped.id != entry.id and not (failed_key and scoped.runtime_api_key == failed_key):
|
||||
|
||||
@@ -5,10 +5,13 @@ back; every other 400 stays a plain request failure. Once every entry rejects th
|
||||
single-credential handling from #106475 takes over.
|
||||
"""
|
||||
import json
|
||||
import time
|
||||
import types
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
from agent.agent_runtime_helpers import recover_with_credential_pool
|
||||
from agent.error_classifier import FailoverReason, classify_api_error
|
||||
|
||||
MODEL = "gpt-5.3-codex"
|
||||
@@ -56,16 +59,26 @@ def test_entitlement_400_benches_only_that_model_and_rotates(pool):
|
||||
generic = classify_api_error(_Err(400, {"detail": "Invalid request: bad field"}), provider="openai-codex", model=MODEL)
|
||||
assert generic.reason == FailoverReason.format_error and not generic.should_rotate_credential
|
||||
|
||||
# Drive the production recovery entry point (turn recovery -> recover_with_credential_pool),
|
||||
# not the pool directly: the classifier verdict must reach the model-scoped bench.
|
||||
assert pool.select(model=MODEL).id == "cred-0"
|
||||
next_entry = pool.mark_exhausted_and_rotate(
|
||||
status_code=400, api_key_hint=TOKENS[0], credential_id="cred-0",
|
||||
failure_reason=verdict.reason.value, model=MODEL,
|
||||
agent = types.SimpleNamespace(
|
||||
provider="openai-codex", model=MODEL, base_url="https://chatgpt.com/backend-api/codex",
|
||||
api_key=TOKENS[0], _credential_pool=pool, _credential_pool_entry_id="cred-0",
|
||||
_swap_credential=MagicMock(return_value=True),
|
||||
)
|
||||
assert next_entry is not None and next_entry.id == "cred-1"
|
||||
assert recover_with_credential_pool(
|
||||
agent, status_code=400, has_retried_429=False, classified_reason=verdict.reason,
|
||||
) == (True, False)
|
||||
agent._swap_credential.assert_called_once()
|
||||
assert agent._swap_credential.call_args.args[0].id == "cred-1"
|
||||
first = pool.entries()[0]
|
||||
assert first.last_status is None # credential-wide state untouched: other models stay usable
|
||||
assert set(first.model_cooldowns) == {MODEL}
|
||||
# An entitlement is a plan property, not a window: no hourly re-probe, only reset clears it.
|
||||
assert first.model_cooldowns[MODEL] > time.time() + 24 * 3600
|
||||
assert pool.select(model=OTHER_MODEL).id == "cred-0"
|
||||
assert pool.reset_statuses() >= 1 and not pool.entries()[0].model_cooldowns
|
||||
|
||||
|
||||
def test_all_entries_rejecting_falls_back_to_session_marker(pool):
|
||||
|
||||
@@ -178,7 +178,7 @@ The pool handles different errors differently:
|
||||
| **429 Rate Limit** | Retry same key once (transient). Second consecutive 429 rotates to next key | 1 hour |
|
||||
| **402 Billing/Quota** | Immediately rotate to next key | 1 hour |
|
||||
| **401 Auth Expired** | Try refreshing the OAuth token first. Rotate only if refresh fails | 5 minutes |
|
||||
| **400 Codex model entitlement** (`The '<model>' model is not supported when using Codex with a ChatGPT account.`) | Bench this key for the rejected model only and rotate to the next key; other models keep using the key. Other 400s never rotate | 1 hour (per model) |
|
||||
| **400 Codex model entitlement** (`The '<model>' model is not supported when using Codex with a ChatGPT account.`) | Bench this key for the rejected model only and rotate to the next key; other models keep using the key. Other 400s never rotate | Until `hermes auth reset` (per model; an entitlement is a plan property, not a window) |
|
||||
| **All keys exhausted** | Fall through to `fallback_model` if configured | — |
|
||||
|
||||
Provider-supplied `reset_at` timestamps override these default cooldowns.
|
||||
|
||||
Reference in New Issue
Block a user