fix: route 404 insufficient_credits_for_paid_model to fallback_model chain

A 404 carrying insufficient_credits_for_paid_model (paid model ungated
by credits) classified as unknown: retryable with no fallback, burning
retries and never switching models. Treat it like 429-exhaustion --
billing with rotate+fallback -- and log an actionable ERROR naming the
credits and the fallback target on activation.

Fixes #115702

(cherry picked from commit a1f7f9996bb82230c945340dcb279ffa923e90a1)
This commit is contained in:
Yagna Vudathu
2026-09-19 02:03:54 -04:00
committed by Teknium
parent 40bfcbda34
commit d861bca178
5 changed files with 170 additions and 1 deletions

View File

@@ -136,6 +136,9 @@ _BILLING_ERROR_CODES = frozenset({
# terminal for this credential until limits are raised.
"credit_balance_exhausted", "organization_spend_limit_exceeded",
"organization_usage_limit_exceeded", "project_spend_limit_exceeded",
# Paid-model credit wall arriving as a 404 (the paid model "vanishes"
# without credits): also matched explicitly by _status_404. (#115702)
"insufficient_credits_for_paid_model",
})
# Transient rate limiting. Bedrock "Throttling error: Too many tokens" also
@@ -585,6 +588,10 @@ _OVERFLOW_AS_5XX_RULES = (
(_CONTEXT_OVERFLOW_PATTERNS, _V_CONTEXT_OVERFLOW),
)
# 404 credit-exhaustion code (paid model without credits): matched as a
# structured code AND in flattened SDK error text by _status_404. (#115702)
_CREDIT_EXHAUSTION_404_CODES = frozenset({"insufficient_credits_for_paid_model"})
# 404: Nous API surfaces credit depletion as a paid model vanishing from the
# Free Tier (billing, not missing model); policy block before model_not_found.
_404_RULES = (
@@ -996,6 +1003,17 @@ def _status_403(c: _Ctx) -> Verdict:
def _status_404(c: _Ctx) -> Verdict:
# Paid-model credit wall: route like 429-exhaustion (billing with the
# fallback chain armed). A structured billing code is decisive here, mirroring
# _status_429 — the status handler always returns, so _by_error_code never
# sees it. The marker lets the fallback switch log an actionable ERROR
# naming the credits and the fallback target.
credit_code = c.code if c.code in _CREDIT_EXHAUSTION_404_CODES else next(
(code for code in _CREDIT_EXHAUSTION_404_CODES if code in c.msg), None,
)
if credit_code is not None:
return _v(_R.billing, retryable=False, **_ROTATE_FALLBACK,
error_context={"credit_exhaustion_code": credit_code})
verdict = _first_match(c.msg, _404_RULES)
if verdict is not None:
return verdict

View File

@@ -19,7 +19,7 @@ from agent.error_classifier import FailoverReason, classify_api_error
from agent.turn_overflow import recover_from_overflow
from agent.turn_recovery import (
_NONRETRYABLE_LABELS, abort_turn_on_interrupt, compute_error_backoff, interruptible_backoff_sleep,
log_api_error_attempt,
log_api_error_attempt, log_credit_exhaustion_fallback,
max_retries_exhausted_result, nonretryable_client_error_result, recover_after_classification,
recover_before_classification, route_classified_error,
)
@@ -339,6 +339,7 @@ def settle_unrecovered_error(
_label = _NONRETRYABLE_LABELS.get(classified.reason, f"Non-retryable error (HTTP {status_code})")
agent._buffer_diagnostic_status(f"⚠️ {_label} — trying fallback...")
if agent._try_activate_fallback():
log_credit_exhaustion_fallback(agent, classified)
# Direct ``return _verdict("break")`` is load-bearing: the restart handler
# re-runs the pre-API preflight against the fallback's context window.
active_system_prompt = _arm_fallback_restart(agent, api_messages, active_system_prompt, _retry)

View File

@@ -1603,6 +1603,24 @@ def activate_codex_app_server_fallback(agent: Any, result: Dict[str, Any]) -> bo
return bool(agent._try_activate_fallback(reason=classified.reason))
def log_credit_exhaustion_fallback(agent: Any, classified: Any) -> None:
"""ERROR naming a 404 credit-exhaustion fallback switch and its target.
Narrow (#115702): only fires when the classifier marked the verdict with
``credit_exhaustion_code`` -- every other fallback keeps its existing copy.
"""
code = (getattr(classified, "error_context", None) or {}).get("credit_exhaustion_code")
if not code:
return
logger.error(
"%sInsufficient credits for paid model (404 %s on %s via %s) -- "
"using fallback %s via %s. Top up credits or pick a free model.",
getattr(agent, "log_prefix", ""),
code, getattr(classified, "model", None), getattr(classified, "provider", None),
getattr(agent, "model", None), getattr(agent, "provider", None),
)
def _is_genuine_nous_rate_limit(agent: Any, api_error: Exception, error_context: Any, classified: Any = None) -> bool:
"""Record a genuine account-level Nous 429 to the cross-session breaker; upstream
capacity 429s (no exhausted bucket in headers or last-known state) are left alone.
@@ -1779,6 +1797,7 @@ def route_classified_error(
if not pool_may_recover:
agent._buffer_diagnostic_status(_eager_fallback_status(classified, _is_upstream, _is_transport_failure))
if agent._try_activate_fallback(reason=classified.reason):
log_credit_exhaustion_fallback(agent, classified)
return _fallback_break()
# A 401/403 surviving credential refresh means a broken credential or endpoint:

View File

@@ -0,0 +1,94 @@
"""404 'insufficient_credits_for_paid_model' triggers the fallback chain (#115702)."""
import logging
from types import SimpleNamespace
from agent.error_classifier import FailoverReason, classify_api_error
from agent.turn_recovery import log_credit_exhaustion_fallback, route_classified_error
from agent.turn_retry_state import TurnRetryState
class MockAPIError(Exception):
"""Simulates an OpenAI SDK APIStatusError."""
def __init__(self, message, status_code=None, body=None):
super().__init__(message)
self.status_code = status_code
self.body = body or {}
def _classified_credit_404():
e = MockAPIError(
"Not Found",
status_code=404,
body={"error": {"code": "insufficient_credits_for_paid_model", "message": "Not Found"}},
)
return classify_api_error(e, provider="nous", model="openai/gpt-5.5-pro")
def test_credit_404_classifies_like_429_exhaustion():
result = _classified_credit_404()
assert result.reason == FailoverReason.billing
assert result.retryable is False
assert result.should_fallback is True
def test_credit_404_fallback_logs_actionable_error_naming_credits_and_target(caplog):
agent = SimpleNamespace(model="openai/gpt-5.5-free", provider="nous", log_prefix="")
with caplog.at_level(logging.ERROR, logger="agent.conversation_loop"):
log_credit_exhaustion_fallback(agent, _classified_credit_404())
errors = [r for r in caplog.records if r.levelno >= logging.ERROR]
assert len(errors) == 1
text = errors[0].getMessage().lower()
assert "credit" in text
assert "insufficient_credits_for_paid_model" in text
assert "openai/gpt-5.5-free" in text
def test_credit_404_fallback_log_silent_without_marker(caplog):
agent = SimpleNamespace(model="x", provider="nous", log_prefix="")
classified = SimpleNamespace(reason=FailoverReason.billing, error_context={})
with caplog.at_level(logging.ERROR, logger="agent.conversation_loop"):
log_credit_exhaustion_fallback(agent, classified)
assert [r for r in caplog.records if r.levelno >= logging.ERROR] == []
def test_credit_404_routes_to_fallback_chain(caplog):
"""End to end through the classifier + eager-fallback routing: the 404
credit error attempts the fallback_model chain and logs the ERROR."""
classified = _classified_credit_404()
calls = []
agent = SimpleNamespace(
model="openai/gpt-5.5-pro",
provider="nous",
log_prefix="",
_fallback_index=0,
_fallback_chain=[{"provider": "nous", "model": "openai/gpt-5.5-free"}],
_credential_pool=None,
)
def _activate(reason=None):
calls.append(reason)
agent.model = "openai/gpt-5.5-free"
return True
agent._try_activate_fallback = _activate
agent._buffer_diagnostic_status = lambda msg: calls.append(msg)
with caplog.at_level(logging.ERROR, logger="agent.conversation_loop"):
verdict = route_classified_error(
agent, MockAPIError("Not Found", 404), classified, TurnRetryState(),
error_msg="Not Found", error_context={}, recovered_with_pool=False,
base_url="", model="openai/gpt-5.5-pro", messages=[], api_messages=[],
system_message=None, active_system_prompt="sys", conversation_history=[],
retry_count=0, max_retries=3, compression_attempts=0,
max_compression_attempts=2, api_call_count=1, effective_task_id=None,
)
assert verdict.action == "break"
assert FailoverReason.billing in calls
assert agent.model == "openai/gpt-5.5-free"
errors = [r for r in caplog.records if r.levelno >= logging.ERROR]
assert len(errors) == 1
text = errors[0].getMessage().lower()
assert "credit" in text and "openai/gpt-5.5-free" in text

View File

@@ -270,6 +270,43 @@ class TestClassifyApiError:
assert result.retryable is False
assert result.should_fallback is True
def test_404_insufficient_credits_for_paid_model_code_is_billing(self):
# A 404 carrying the structured 'insufficient_credits_for_paid_model'
# code (paid model ungated by credits) must route like 429-exhaustion:
# billing with the fallback chain armed -- not 'unknown' burning
# retries and never falling back. (#115702)
e = MockAPIError(
"Not Found",
status_code=404,
body={
"error": {
"code": "insufficient_credits_for_paid_model",
"message": "Not Found",
},
},
)
result = classify_api_error(e, provider="nous", model="openai/gpt-5.5-pro")
assert result.reason == FailoverReason.billing
assert result.retryable is False
assert result.should_rotate_credential is True
assert result.should_fallback is True
assert result.error_context.get("credit_exhaustion_code") == "insufficient_credits_for_paid_model"
def test_404_insufficient_credits_for_paid_model_in_message_is_billing(self):
# Same code embedded in the exception text (the SDK str() flattens the
# body) with no billing wording in the message -- still billing.
e = MockAPIError(
"Error code: 404 - {'error': {'code': 'insufficient_credits_for_paid_model', "
"'message': 'request failed'}}",
status_code=404,
)
result = classify_api_error(e, provider="nous", model="openai/gpt-5.5-pro")
assert result.reason == FailoverReason.billing
assert result.retryable is False
assert result.should_fallback is True
assert result.error_context.get("credit_exhaustion_code") == "insufficient_credits_for_paid_model"
def test_wrapped_402_uses_nested_body_message(self):
inner = MockAPIError(
"inner",