From 92df11f81df82dd8d708ec2eeb2dcf3f83d3ca23 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 9 Sep 2026 02:38:20 -0700 Subject: [PATCH] fix(classifier): terminal_quota_exhausted 429s classify as billing, not rate_limit MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Port from code-yeongyu/oh-my-openagent#6677 (credit: @niStee). LiteLLM proxies stamp a structured `terminal_quota_exhausted` code on hard-cap 429s. Hermes' `_status_429` handler always returns a verdict, so `_by_error_code` (which maps _BILLING_ERROR_CODES to billing) never saw the code: the exhausted key classified as rate_limit, earned the 429 cooldown, and got retried against a wall that cannot clear until someone pays. Upstream this respawned duplicate subagent sessions. - `_status_429` now honors a structured billing code first (decisive signal outranks message heuristics). - `terminal_quota_exhausted` joins _BILLING_ERROR_CODES so every path (429, 402, status-less) agrees. - "hard billing limit" free text joins _BILLING_PATTERNS ("billing hard limit" was already there; providers use both orders). "terminal billing limit" text is deliberately NOT matched: substring rules cannot negate the "non-terminal billing limit" wording — the structured code covers it. --- agent/error_classifier.py | 12 +++++++++++- tests/agent/test_error_classifier.py | 24 ++++++++++++++++++++++++ 2 files changed, 35 insertions(+), 1 deletion(-) diff --git a/agent/error_classifier.py b/agent/error_classifier.py index 59b244b7b9..b49ff9b7b3 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -92,6 +92,11 @@ _BILLING_PATTERNS = ( "billing hard limit", "exceeded your current quota", "account is deactivated", "plan does not include", "out of extra usage", "out of funds", "run out of funds", "balance_depleted", "model_not_supported_on_free_tier", "not available on the free tier", + # LiteLLM proxies word a hard cap as "hard billing limit" (structured twin: + # ``terminal_quota_exhausted`` in _BILLING_ERROR_CODES). "terminal billing + # limit" free text is NOT matched: substring rules can't negate the + # "non-terminal billing limit" wording, and the structured code covers it. + "hard billing limit", ) # Not proof of exhaustion: Anthropic returns the same "out of extra usage" body @@ -107,7 +112,7 @@ _XAI_SPENDING_LIMIT_ERROR_CODE = "personal-team-blocked:spending-limit" _BILLING_ERROR_CODES = frozenset({ "insufficient_quota", "billing_not_active", "payment_required", "insufficient_credits", "no_usable_credits", "balance_depleted", "model_not_supported_on_free_tier", - "member_spend_cap_exceeded", _XAI_SPENDING_LIMIT_ERROR_CODE, + "member_spend_cap_exceeded", "terminal_quota_exhausted", _XAI_SPENDING_LIMIT_ERROR_CODE, }) # Transient rate limiting. Bedrock "Throttling error: Too many tokens" also @@ -725,6 +730,11 @@ def _status_404(c: _Ctx) -> Verdict: def _status_429(c: _Ctx) -> Verdict: + # A structured billing code is decisive: LiteLLM stamps + # ``terminal_quota_exhausted`` (a hard cap, not throttling) on 429s, and + # this handler always returns, so _by_error_code never sees the code. + if c.code in _BILLING_ERROR_CODES: + return _V_BILLING # Z.AI/Zhipu reuse 429 for server-wide overload: back off on the same # key instead of burning the pool (#14038). if any(p in c.msg for p in _OVERLOADED_PATTERNS): diff --git a/tests/agent/test_error_classifier.py b/tests/agent/test_error_classifier.py index 13cb0ab873..995a91fb40 100644 --- a/tests/agent/test_error_classifier.py +++ b/tests/agent/test_error_classifier.py @@ -539,6 +539,30 @@ class TestClassifyApiError: assert result.reason == FailoverReason.rate_limit assert result.should_rotate_credential is True + def test_429_with_structured_terminal_quota_code_is_billing(self): + """LiteLLM stamps ``terminal_quota_exhausted`` on a hard-cap 429. The + 429 handler always returns a verdict, so the structured billing code + must be honored inside it — otherwise the exhausted key is retried + (upstream this respawned duplicate subagents; ported from + code-yeongyu/oh-my-openagent#6677).""" + e = MockAPIError( + "request failed", status_code=429, + body={"error": {"code": "terminal_quota_exhausted", "message": "request failed"}}, + ) + result = classify_api_error(e) + assert result.reason == FailoverReason.billing + assert result.retryable is False + assert result.should_fallback is True + + def test_429_hard_billing_limit_text_is_billing(self): + """The free-text twin: "hard billing limit" is exhaustion wording, not + throttling, even though it contains no reset signal to disambiguate.""" + result = classify_api_error( + MockAPIError("hard billing limit reached for this key", status_code=429) + ) + assert result.reason == FailoverReason.billing + assert result.retryable is False + # ── 5xx that are actually request-validation errors ── # Some OpenAI-compatible gateways (e.g. codex.nekos.me) return # request-validation failures with a 5xx status. These are