From d861bca17848ea8bb843e38c36aca16b8e7efa0d Mon Sep 17 00:00:00 2001 From: Yagna Vudathu Date: Sat, 19 Sep 2026 02:03:54 -0400 Subject: [PATCH] fix: route 404 insufficient_credits_for_paid_model to fallback_model chain A 404 carrying insufficient_credits_for_paid_model (paid model ungated by credits) classified as unknown: retryable with no fallback, burning retries and never switching models. Treat it like 429-exhaustion -- billing with rotate+fallback -- and log an actionable ERROR naming the credits and the fallback target on activation. Fixes #115702 (cherry picked from commit a1f7f9996bb82230c945340dcb279ffa923e90a1) --- agent/error_classifier.py | 18 +++++ agent/turn_api_error.py | 3 +- agent/turn_recovery.py | 19 +++++ tests/agent/test_credit_404_fallback.py | 94 +++++++++++++++++++++++++ tests/agent/test_error_classifier.py | 37 ++++++++++ 5 files changed, 170 insertions(+), 1 deletion(-) create mode 100644 tests/agent/test_credit_404_fallback.py diff --git a/agent/error_classifier.py b/agent/error_classifier.py index 1cdc7bcd95..b90801d161 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -136,6 +136,9 @@ _BILLING_ERROR_CODES = frozenset({ # terminal for this credential until limits are raised. "credit_balance_exhausted", "organization_spend_limit_exceeded", "organization_usage_limit_exceeded", "project_spend_limit_exceeded", + # Paid-model credit wall arriving as a 404 (the paid model "vanishes" + # without credits): also matched explicitly by _status_404. (#115702) + "insufficient_credits_for_paid_model", }) # Transient rate limiting. Bedrock "Throttling error: Too many tokens" also @@ -585,6 +588,10 @@ _OVERFLOW_AS_5XX_RULES = ( (_CONTEXT_OVERFLOW_PATTERNS, _V_CONTEXT_OVERFLOW), ) +# 404 credit-exhaustion code (paid model without credits): matched as a +# structured code AND in flattened SDK error text by _status_404. (#115702) +_CREDIT_EXHAUSTION_404_CODES = frozenset({"insufficient_credits_for_paid_model"}) + # 404: Nous API surfaces credit depletion as a paid model vanishing from the # Free Tier (billing, not missing model); policy block before model_not_found. _404_RULES = ( @@ -996,6 +1003,17 @@ def _status_403(c: _Ctx) -> Verdict: def _status_404(c: _Ctx) -> Verdict: + # Paid-model credit wall: route like 429-exhaustion (billing with the + # fallback chain armed). A structured billing code is decisive here, mirroring + # _status_429 — the status handler always returns, so _by_error_code never + # sees it. The marker lets the fallback switch log an actionable ERROR + # naming the credits and the fallback target. + credit_code = c.code if c.code in _CREDIT_EXHAUSTION_404_CODES else next( + (code for code in _CREDIT_EXHAUSTION_404_CODES if code in c.msg), None, + ) + if credit_code is not None: + return _v(_R.billing, retryable=False, **_ROTATE_FALLBACK, + error_context={"credit_exhaustion_code": credit_code}) verdict = _first_match(c.msg, _404_RULES) if verdict is not None: return verdict diff --git a/agent/turn_api_error.py b/agent/turn_api_error.py index 464801c0cc..8fdf475ade 100644 --- a/agent/turn_api_error.py +++ b/agent/turn_api_error.py @@ -19,7 +19,7 @@ from agent.error_classifier import FailoverReason, classify_api_error from agent.turn_overflow import recover_from_overflow from agent.turn_recovery import ( _NONRETRYABLE_LABELS, abort_turn_on_interrupt, compute_error_backoff, interruptible_backoff_sleep, - log_api_error_attempt, + log_api_error_attempt, log_credit_exhaustion_fallback, max_retries_exhausted_result, nonretryable_client_error_result, recover_after_classification, recover_before_classification, route_classified_error, ) @@ -339,6 +339,7 @@ def settle_unrecovered_error( _label = _NONRETRYABLE_LABELS.get(classified.reason, f"Non-retryable error (HTTP {status_code})") agent._buffer_diagnostic_status(f"⚠️ {_label} — trying fallback...") if agent._try_activate_fallback(): + log_credit_exhaustion_fallback(agent, classified) # Direct ``return _verdict("break")`` is load-bearing: the restart handler # re-runs the pre-API preflight against the fallback's context window. active_system_prompt = _arm_fallback_restart(agent, api_messages, active_system_prompt, _retry) diff --git a/agent/turn_recovery.py b/agent/turn_recovery.py index 32324ecd79..34b0a53d51 100644 --- a/agent/turn_recovery.py +++ b/agent/turn_recovery.py @@ -1603,6 +1603,24 @@ def activate_codex_app_server_fallback(agent: Any, result: Dict[str, Any]) -> bo return bool(agent._try_activate_fallback(reason=classified.reason)) +def log_credit_exhaustion_fallback(agent: Any, classified: Any) -> None: + """ERROR naming a 404 credit-exhaustion fallback switch and its target. + + Narrow (#115702): only fires when the classifier marked the verdict with + ``credit_exhaustion_code`` -- every other fallback keeps its existing copy. + """ + code = (getattr(classified, "error_context", None) or {}).get("credit_exhaustion_code") + if not code: + return + logger.error( + "%sInsufficient credits for paid model (404 %s on %s via %s) -- " + "using fallback %s via %s. Top up credits or pick a free model.", + getattr(agent, "log_prefix", ""), + code, getattr(classified, "model", None), getattr(classified, "provider", None), + getattr(agent, "model", None), getattr(agent, "provider", None), + ) + + def _is_genuine_nous_rate_limit(agent: Any, api_error: Exception, error_context: Any, classified: Any = None) -> bool: """Record a genuine account-level Nous 429 to the cross-session breaker; upstream capacity 429s (no exhausted bucket in headers or last-known state) are left alone. @@ -1779,6 +1797,7 @@ def route_classified_error( if not pool_may_recover: agent._buffer_diagnostic_status(_eager_fallback_status(classified, _is_upstream, _is_transport_failure)) if agent._try_activate_fallback(reason=classified.reason): + log_credit_exhaustion_fallback(agent, classified) return _fallback_break() # A 401/403 surviving credential refresh means a broken credential or endpoint: diff --git a/tests/agent/test_credit_404_fallback.py b/tests/agent/test_credit_404_fallback.py new file mode 100644 index 0000000000..ea4fbdb925 --- /dev/null +++ b/tests/agent/test_credit_404_fallback.py @@ -0,0 +1,94 @@ +"""404 'insufficient_credits_for_paid_model' triggers the fallback chain (#115702).""" + +import logging +from types import SimpleNamespace + +from agent.error_classifier import FailoverReason, classify_api_error +from agent.turn_recovery import log_credit_exhaustion_fallback, route_classified_error +from agent.turn_retry_state import TurnRetryState + + +class MockAPIError(Exception): + """Simulates an OpenAI SDK APIStatusError.""" + + def __init__(self, message, status_code=None, body=None): + super().__init__(message) + self.status_code = status_code + self.body = body or {} + + +def _classified_credit_404(): + e = MockAPIError( + "Not Found", + status_code=404, + body={"error": {"code": "insufficient_credits_for_paid_model", "message": "Not Found"}}, + ) + return classify_api_error(e, provider="nous", model="openai/gpt-5.5-pro") + + +def test_credit_404_classifies_like_429_exhaustion(): + result = _classified_credit_404() + assert result.reason == FailoverReason.billing + assert result.retryable is False + assert result.should_fallback is True + + +def test_credit_404_fallback_logs_actionable_error_naming_credits_and_target(caplog): + agent = SimpleNamespace(model="openai/gpt-5.5-free", provider="nous", log_prefix="") + with caplog.at_level(logging.ERROR, logger="agent.conversation_loop"): + log_credit_exhaustion_fallback(agent, _classified_credit_404()) + errors = [r for r in caplog.records if r.levelno >= logging.ERROR] + assert len(errors) == 1 + text = errors[0].getMessage().lower() + assert "credit" in text + assert "insufficient_credits_for_paid_model" in text + assert "openai/gpt-5.5-free" in text + + +def test_credit_404_fallback_log_silent_without_marker(caplog): + agent = SimpleNamespace(model="x", provider="nous", log_prefix="") + classified = SimpleNamespace(reason=FailoverReason.billing, error_context={}) + with caplog.at_level(logging.ERROR, logger="agent.conversation_loop"): + log_credit_exhaustion_fallback(agent, classified) + assert [r for r in caplog.records if r.levelno >= logging.ERROR] == [] + + +def test_credit_404_routes_to_fallback_chain(caplog): + """End to end through the classifier + eager-fallback routing: the 404 + credit error attempts the fallback_model chain and logs the ERROR.""" + classified = _classified_credit_404() + calls = [] + + agent = SimpleNamespace( + model="openai/gpt-5.5-pro", + provider="nous", + log_prefix="", + _fallback_index=0, + _fallback_chain=[{"provider": "nous", "model": "openai/gpt-5.5-free"}], + _credential_pool=None, + ) + + def _activate(reason=None): + calls.append(reason) + agent.model = "openai/gpt-5.5-free" + return True + + agent._try_activate_fallback = _activate + agent._buffer_diagnostic_status = lambda msg: calls.append(msg) + + with caplog.at_level(logging.ERROR, logger="agent.conversation_loop"): + verdict = route_classified_error( + agent, MockAPIError("Not Found", 404), classified, TurnRetryState(), + error_msg="Not Found", error_context={}, recovered_with_pool=False, + base_url="", model="openai/gpt-5.5-pro", messages=[], api_messages=[], + system_message=None, active_system_prompt="sys", conversation_history=[], + retry_count=0, max_retries=3, compression_attempts=0, + max_compression_attempts=2, api_call_count=1, effective_task_id=None, + ) + assert verdict.action == "break" + assert FailoverReason.billing in calls + assert agent.model == "openai/gpt-5.5-free" + errors = [r for r in caplog.records if r.levelno >= logging.ERROR] + assert len(errors) == 1 + text = errors[0].getMessage().lower() + assert "credit" in text and "openai/gpt-5.5-free" in text diff --git a/tests/agent/test_error_classifier.py b/tests/agent/test_error_classifier.py index 5c0f941b15..6658881081 100644 --- a/tests/agent/test_error_classifier.py +++ b/tests/agent/test_error_classifier.py @@ -270,6 +270,43 @@ class TestClassifyApiError: assert result.retryable is False assert result.should_fallback is True + def test_404_insufficient_credits_for_paid_model_code_is_billing(self): + # A 404 carrying the structured 'insufficient_credits_for_paid_model' + # code (paid model ungated by credits) must route like 429-exhaustion: + # billing with the fallback chain armed -- not 'unknown' burning + # retries and never falling back. (#115702) + e = MockAPIError( + "Not Found", + status_code=404, + body={ + "error": { + "code": "insufficient_credits_for_paid_model", + "message": "Not Found", + }, + }, + ) + result = classify_api_error(e, provider="nous", model="openai/gpt-5.5-pro") + assert result.reason == FailoverReason.billing + assert result.retryable is False + assert result.should_rotate_credential is True + assert result.should_fallback is True + assert result.error_context.get("credit_exhaustion_code") == "insufficient_credits_for_paid_model" + + def test_404_insufficient_credits_for_paid_model_in_message_is_billing(self): + # Same code embedded in the exception text (the SDK str() flattens the + # body) with no billing wording in the message -- still billing. + e = MockAPIError( + "Error code: 404 - {'error': {'code': 'insufficient_credits_for_paid_model', " + "'message': 'request failed'}}", + status_code=404, + ) + result = classify_api_error(e, provider="nous", model="openai/gpt-5.5-pro") + assert result.reason == FailoverReason.billing + assert result.retryable is False + assert result.should_fallback is True + assert result.error_context.get("credit_exhaustion_code") == "insufficient_credits_for_paid_model" + + def test_wrapped_402_uses_nested_body_message(self): inner = MockAPIError( "inner",