"""Tests for agent.error_classifier — structured API error classification.""" from types import SimpleNamespace import pytest from agent.error_classifier import ( ClassifiedError, FailoverReason, PROVIDER_STREAM_NON_JSON_ERROR_CODE, classify_api_error, is_reasoning_field_rejection, _extract_status_code, _extract_error_body, _extract_error_code, _classify_402, ) from tests.hermes_cli.anon_portal import make_jwt # ── Helper: mock API errors ──────────────────────────────────────────── class MockAPIError(Exception): """Simulates an OpenAI SDK APIStatusError.""" def __init__(self, message, status_code=None, body=None, headers=None): super().__init__(message) self.status_code = status_code self.body = body or {} self.response = SimpleNamespace(headers=headers or {}) class MockTransportError(Exception): """Simulates a transport-level error with a specific type name.""" pass class ReadTimeout(MockTransportError): pass class ConnectError(MockTransportError): pass class RemoteProtocolError(MockTransportError): pass class ServerDisconnectedError(MockTransportError): pass # ── Test: FailoverReason enum ────────────────────────────────────────── # ── Test: ClassifiedError ────────────────────────────────────────────── class TestClassifiedError: def test_is_auth_property(self): e1 = ClassifiedError(reason=FailoverReason.auth) assert e1.is_auth is True e2 = ClassifiedError(reason=FailoverReason.auth_permanent) assert e2.is_auth is True e3 = ClassifiedError(reason=FailoverReason.billing) assert e3.is_auth is False # ── Test: Status code extraction ─────────────────────────────────────── class TestExtractStatusCode: def test_from_status_attr(self): class ErrWithStatus(Exception): status = 503 assert _extract_status_code(ErrWithStatus()) == 503 def test_from_cause_chain(self): inner = MockAPIError("inner", status_code=401) outer = Exception("outer") outer.__cause__ = inner assert _extract_status_code(outer) == 401 # ── Test: Error body extraction ──────────────────────────────────────── class TestExtractErrorBody: def test_from_cause_chain_body_attr(self): inner = MockAPIError( "inner", status_code=402, body={"error": {"message": "Usage limit reached, try again in 5 minutes"}}, ) outer = Exception("outer") outer.__cause__ = inner assert _extract_error_body(outer) == { "error": {"message": "Usage limit reached, try again in 5 minutes"}, } def test_empty_when_no_body(self): assert _extract_error_body(Exception("generic")) == {} # ── Test: Error code extraction ──────────────────────────────────────── class TestExtractErrorCode: def test_from_top_level_code(self): body = {"code": "model_not_found"} assert _extract_error_code(body) == "model_not_found" def test_empty_when_no_code(self): assert _extract_error_code({}) == "" assert _extract_error_code({"error": {"message": "oops"}}) == "" # ── Test: 402 disambiguation ─────────────────────────────────────────── class TestClassify402: """The critical 402 billing vs rate_limit disambiguation.""" def test_billing_exhaustion(self): """Plain 402 = billing.""" result = _classify_402( "payment required", lambda reason, **kw: ClassifiedError(reason=reason, **kw), ) assert result.reason == FailoverReason.billing assert result.should_rotate_credential is True def test_quota_with_retry(self): """402 with 'quota' + 'retry' = rate limit.""" result = _classify_402( "quota exceeded, please retry after the window resets", lambda reason, **kw: ClassifiedError(reason=reason, **kw), ) assert result.reason == FailoverReason.rate_limit # ── Test: Full classification pipeline ───────────────────────────────── class TestClassifyApiError: """End-to-end classification tests.""" # ── Auth errors ── def test_401_classified_as_auth(self): e = MockAPIError("Unauthorized", status_code=401) result = classify_api_error(e, provider="openrouter") assert result.reason == FailoverReason.auth assert result.should_rotate_credential is True # 401 is non-retryable on its own — credential rotation runs # before the retryability check in the agent loop. assert result.retryable is False assert result.should_fallback is True def test_403_classified_as_auth(self): e = MockAPIError("Forbidden", status_code=403) result = classify_api_error(e, provider="anthropic") assert result.reason == FailoverReason.auth assert result.should_fallback is True def test_403_upstream_unavailable_code_is_transient_not_auth(self): """A gateway 403 stamped ``code=upstream_unavailable`` is a transient upstream outage: retried with backoff, credential untouched (#75388).""" body = {"error": {"message": "Upstream service temporarily unavailable. Please retry later.", "type": "upstream_unavailable", "code": "upstream_unavailable"}} result = classify_api_error(MockAPIError("Forbidden", status_code=403, body=body), provider="custom") assert result.reason == FailoverReason.overloaded assert result.retryable is True assert result.should_rotate_credential is False # ── Billing ── def test_402_plain_billing(self): e = MockAPIError("Payment Required", status_code=402) result = classify_api_error(e) assert result.reason == FailoverReason.billing assert result.retryable is False def test_404_free_tier_model_block_is_billing(self): e = MockAPIError( "Not Found", status_code=404, body={ "status": 404, "message": ( "Model 'gpt-5' is not available on the Free Tier. " "Upgrade at https://portal.nousresearch.com or pick a free model." ), }, ) result = classify_api_error(e, provider="nous", model="gpt-5") assert result.reason == FailoverReason.billing assert result.retryable is False assert result.should_fallback is True def test_404_requires_available_credits_is_billing(self): e = MockAPIError( "Not Found", status_code=404, body={ "status": 404, "message": ( "Model 'openai/gpt-5.5-pro' requires available credits. " "Your account balance is too low to use paid models — " "add credits at https://portal.nousresearch.com or pick a free model." ), }, ) result = classify_api_error(e, provider="nous", model="openai/gpt-5.5-pro") assert result.reason == FailoverReason.billing assert result.retryable is False assert result.should_fallback is True def test_wrapped_402_uses_nested_body_message(self): inner = MockAPIError( "inner", status_code=402, body={"error": {"message": "Usage limit reached, try again in 5 minutes"}}, ) outer = Exception("outer") outer.__cause__ = inner result = classify_api_error(outer) assert result.reason == FailoverReason.rate_limit assert result.retryable is True assert result.message == "Usage limit reached, try again in 5 minutes" # ── Rate limit ── def test_429_rate_limit(self): e = MockAPIError("Too Many Requests", status_code=429) result = classify_api_error(e) assert result.reason == FailoverReason.rate_limit assert result.should_fallback is True @pytest.mark.parametrize("spelling", [ "resource exhausted", "RESOURCE_EXHAUSTED", "ResourceExhausted", "resource-exhausted", ]) def test_resource_exhausted_separator_variants_without_status(self, spelling): result = classify_api_error( Exception(f"{spelling}: Worker local total request limit reached (32/32)"), provider="nvidia", model="nvidia/nemotron-3-ultra-550b-a55b", ) assert result.reason == FailoverReason.rate_limit assert result.retryable is True assert result.should_rotate_credential is True assert result.should_fallback is True def test_anthropic_429_usage_limit_without_reset_is_billing(self): e = MockAPIError( "usage limit reached", status_code=429, body={ "error": { "type": "usage_limit_reached", "message": "Your account has reached its usage limit.", } }, ) result = classify_api_error(e, provider="anthropic", model="claude-opus-5") assert result.reason == FailoverReason.billing assert result.retryable is False assert result.should_fallback is True def test_anthropic_429_usage_limit_with_reset_stays_rate_limit(self): e = MockAPIError( "usage limit reached; resets at 2026-08-24T10:00:00Z", status_code=429, ) result = classify_api_error(e, provider="anthropic", model="claude-opus-5") assert result.reason == FailoverReason.rate_limit assert result.retryable is True @pytest.mark.parametrize( ("reset_field", "reset_value"), [ ("resets_in_seconds", 3600), ("resets_at", "2026-08-24T10:00:00Z"), ("reset_at", "2026-08-24T10:00:00Z"), ("retry_after", 3600), ], ) def test_anthropic_429_usage_limit_with_structured_reset_stays_rate_limit( self, reset_field, reset_value, ): e = MockAPIError( "usage limit reached", status_code=429, body={ "error": { "type": "usage_limit_reached", "message": "Your account has reached its usage limit.", reset_field: reset_value, } }, ) result = classify_api_error(e, provider="anthropic", model="claude-opus-5") assert result.reason == FailoverReason.rate_limit assert result.retryable is True @pytest.mark.parametrize("header", ["Retry-After", "x-ratelimit-reset"]) def test_anthropic_429_usage_limit_with_reset_header_stays_rate_limit(self, header): e = MockAPIError( "usage limit reached", status_code=429, body={ "error": { "type": "usage_limit_reached", "message": "Your account has reached its usage limit.", } }, headers={header: "3600"}, ) result = classify_api_error(e, provider="anthropic", model="claude-opus-5") assert result.reason == FailoverReason.rate_limit assert result.retryable is True def test_429_generic_quota_wall_is_billing(self): # Broadened from the narrow "usage limit" core to the full # _USAGE_LIMIT_PATTERNS: a bare "quota" / "limit exceeded" 429 with no # reset signal is a hard wall, not a retryable throttle. (credit #39441) for msg in ("Monthly quota reached.", "API key limit exceeded."): e = MockAPIError(msg, status_code=429) result = classify_api_error(e, provider="groq", model="llama-3") assert result.reason == FailoverReason.billing, msg assert result.retryable is False, msg def test_429_insufficient_credits_is_billing(self): e = MockAPIError("Insufficient credits remaining.", status_code=429) result = classify_api_error(e, provider="openrouter", model="x") assert result.reason == FailoverReason.billing assert result.retryable is False @pytest.mark.parametrize( "code", [ "credit_balance_exhausted", "organization_spend_limit_exceeded", "project_spend_limit_exceeded", "organization_usage_limit_exceeded", ], ) @pytest.mark.parametrize("status_code", [None, 429]) def test_openai_spend_usage_limit_codes_are_billing(self, code, status_code): # OpenAI documents these structured codes on HTTP 429 when a credit # balance or org/project spend/usage cap is exhausted. They must # classify as billing (rotate + fallback) on the 429 path AND on the # status-less path (SSE/stream-surfaced errors carry only the body), # never as a retryable rate limit. (clean-room port of # zed-industries/zed#63208) e = MockAPIError( "request rejected", status_code=status_code, body={"error": {"code": code, "message": "request rejected"}}, ) result = classify_api_error(e, provider="openai", model="gpt-5") assert result.reason == FailoverReason.billing assert result.retryable is False assert result.should_rotate_credential is True assert result.should_fallback is True def test_429_rate_limit_phrase_never_promotes_to_billing(self): # The exclusion guard: "Rate limit exceeded" contains the # "limit exceeded" usage-limit substring, but an explicit rate-limit # phrase must stay a retryable rate limit. (guard credit #39441) for msg in ( "Rate limit exceeded, please slow down.", "Too many requests; rate_limit hit.", ): e = MockAPIError(msg, status_code=429) result = classify_api_error(e, provider="anthropic", model="claude-opus-5") assert result.reason == FailoverReason.rate_limit, msg assert result.retryable is True, msg def test_codex_weekly_usage_limit_resets_in_stays_rate_limit(self): # Codex surfaces "Weekly usage limit reached. Resets in 6hr 29min." # "resets in" was NOT a transient signal before, so this wrongly read # as terminal billing. (transient-signal credit #63021) e = MockAPIError( "Weekly usage limit reached. Resets in 6hr 29min.", status_code=429, ) result = classify_api_error(e, provider="openai-codex", model="gpt-5-codex") assert result.reason == FailoverReason.rate_limit assert result.retryable is True @pytest.mark.parametrize( "phrase", [ "usage limit reached, reset after 3600s", "usage limit reached, available in 42 minutes", "usage limit reached; 20 requests per minute", ], ) def test_429_usage_limit_with_extra_transient_phrases_stays_rate_limit(self, phrase): # Additional transient signals. (credit #74785) e = MockAPIError(phrase, status_code=429) result = classify_api_error(e, provider="anthropic", model="claude-opus-5") assert result.reason == FailoverReason.rate_limit assert result.retryable is True def test_alibaba_rate_increased_too_quickly(self): """Alibaba/DashScope returns a unique throttling message. Port from anomalyco/opencode#21355. """ msg = ( "Upstream error from Alibaba: Request rate increased too quickly. " "To ensure system stability, please adjust your client logic to " "scale requests more smoothly over time." ) e = MockAPIError(msg, status_code=400) result = classify_api_error(e) assert result.reason == FailoverReason.rate_limit assert result.retryable is True assert result.should_rotate_credential is True # ── Server errors ── def test_500_server_error(self): e = MockAPIError("Internal Server Error", status_code=500) result = classify_api_error(e) assert result.reason == FailoverReason.server_error assert result.retryable is True def test_502_server_error(self): e = MockAPIError("Bad Gateway", status_code=502) result = classify_api_error(e) assert result.reason == FailoverReason.server_error def test_503_overloaded(self): e = MockAPIError("Service Unavailable", status_code=503) result = classify_api_error(e) assert result.reason == FailoverReason.overloaded def test_408_request_timeout_is_retryable_timeout(self): """HTTP 408 Request Timeout is a transient timing failure the server itself flags as safe to retry (RFC 9110 §15.5.9) — commonly emitted by reverse proxies in front of self-hosted backends (llama.cpp / Ollama / vLLM) when a long generation outruns the proxy's request-read window. It must NOT fall into the generic 4xx bucket as a non-retryable format_error, which would abort the turn on a retry-safe error.""" e = MockAPIError("Request Timeout", status_code=408) result = classify_api_error(e, provider="vllm") assert result.reason == FailoverReason.timeout assert result.retryable is True def test_400_bad_request_still_non_retryable_format_error(self): """Guard the boundary: a genuine 400 Bad Request must remain a non-retryable format_error and must not be swept up by the 408 branch.""" e = MockAPIError("Bad Request", status_code=400) result = classify_api_error(e) assert result.reason == FailoverReason.format_error assert result.retryable is False def test_message_only_overloaded_without_status_is_overloaded(self): """Some Anthropic-compatible proxies surface 'overloaded' in the message with no 503/529 status_code. It must classify as overloaded (transient backoff+retry), not unknown / credential rotation. (#14261)""" e = MockAPIError( "Anthropic API error: Overloaded - the service is temporarily overloaded" ) # no status_code result = classify_api_error(e, provider="anthropic") assert result.reason == FailoverReason.overloaded assert result.retryable is True assert result.should_rotate_credential is False def test_429_with_overloaded_body_is_overloaded_not_rate_limit(self): """Z.AI / Zhipu reuse HTTP 429 for server-wide overload. The credential is valid — the server is just busy — so it must classify as overloaded (back off + retry the same key), NOT rate_limit (which would rotate and exhaust the pool, doing nothing for a single-key user). (#14038)""" e = MockAPIError( "The service may be temporarily overloaded, please try again later", status_code=429, ) result = classify_api_error(e, provider="zai") assert result.reason == FailoverReason.overloaded assert result.retryable is True assert result.should_rotate_credential is False def test_429_server_overload_is_overloaded_not_rate_limit(self): """Novita returns HTTP 429 with message 'server overload, please try again later' and error type 'server_overload' for a genuinely busy server (not a credential quota). Neither phrase was in the overload tuple, so it fell through to rate_limit and would rotate the credential / fall back early instead of retrying the same key. (#106205)""" e = MockAPIError( "server overload, please try again later", status_code=429, body={"error": {"message": "server overload, please try again later", "type": "server_overload"}}, ) result = classify_api_error(e, provider="custom:novita") assert result.reason == FailoverReason.overloaded assert result.retryable is True assert result.should_fallback is False assert result.should_rotate_credential is False def test_429_normal_rate_limit_still_rotates(self): """Guard: a genuine 429 rate limit (no overload language) must still classify as rate_limit and rotate the credential. (#14038)""" e = MockAPIError( "Rate limit exceeded: too many requests", status_code=429 ) result = classify_api_error(e, provider="zai") assert result.reason == FailoverReason.rate_limit assert result.should_rotate_credential is True def test_429_with_structured_terminal_quota_code_is_billing(self): """LiteLLM stamps ``terminal_quota_exhausted`` on a hard-cap 429. The 429 handler always returns a verdict, so the structured billing code must be honored inside it — otherwise the exhausted key is retried (upstream this respawned duplicate subagents; ported from code-yeongyu/oh-my-openagent#6677).""" e = MockAPIError( "request failed", status_code=429, body={"error": {"code": "terminal_quota_exhausted", "message": "request failed"}}, ) result = classify_api_error(e) assert result.reason == FailoverReason.billing assert result.retryable is False assert result.should_fallback is True def test_429_hard_billing_limit_text_is_billing(self): """The free-text twin: "hard billing limit" is exhaustion wording, not throttling, even though it contains no reset signal to disambiguate.""" result = classify_api_error( MockAPIError("hard billing limit reached for this key", status_code=429) ) assert result.reason == FailoverReason.billing assert result.retryable is False # ── 5xx that are actually request-validation errors ── # Some OpenAI-compatible gateways (e.g. codex.nekos.me) return # request-validation failures with a 5xx status. These are # deterministic, so they must NOT be retried — otherwise the retry # loop hammers the identical bad request into a flood. def test_non_json_stream_validation_error_is_non_retryable(self): e = MockAPIError( "Provider stream returned non-JSON SSE data", body={ "error": { "code": PROVIDER_STREAM_NON_JSON_ERROR_CODE, "message": ( "request validation failed: unsupported reasoning_effort" ), } }, ) result = classify_api_error(e) assert result.status_code is None assert result.reason == FailoverReason.format_error assert result.retryable is False assert result.should_fallback is True def test_non_json_stream_unknown_error_remains_retryable(self): e = MockAPIError( "Provider stream returned non-JSON SSE data", body={ "error": { "code": PROVIDER_STREAM_NON_JSON_ERROR_CODE, "message": "upstream sent opaque plain-text stream data", } }, ) result = classify_api_error(e) assert result.status_code is None assert result.reason == FailoverReason.unknown assert result.retryable is True assert result.should_fallback is False # ── 5xx that are actually context overflow ── # Some local inference servers (llama.cpp / llama-server, and vLLM/Ollama # behind a Cloudflare/Tailscale hop) report context overflow with a 5xx # status instead of the standard 400/413. These must route into the # compression-and-retry path, not the blind server_error/overloaded retry # that exhausts and drops the turn. # ── Model not found ── def test_404_model_not_found(self): e = MockAPIError("model not found", status_code=404) result = classify_api_error(e) assert result.reason == FailoverReason.model_not_found assert result.should_fallback is True assert result.retryable is False def test_404_generic(self): # Generic 404 with no "model not found" signal — common for local # llama.cpp/Ollama/vLLM endpoints with slightly wrong paths. Treat # as unknown (retryable) so the real error surfaces, rather than # claiming the model is missing and silently falling back. e = MockAPIError("Not Found", status_code=404) result = classify_api_error(e) assert result.reason == FailoverReason.unknown assert result.retryable is True assert result.should_fallback is False def test_404_bare_model_id_missing_prefix_is_model_not_found(self): """A bare id the provider only serves as ``vendor/id`` is malformed. Regression for #78796: NVIDIA NIM answers a prefix-less ``nemotron-3-ultra-550b-a55b`` with a naked ``404 page not found``. Without the catalogue check this fell into the generic branch and burned three retries on a deterministic failure, reporting what looked like an outage. """ e = MockAPIError("404 page not found", status_code=404) result = classify_api_error( e, provider="nvidia", model="nemotron-3-ultra-550b-a55b" ) assert result.reason == FailoverReason.model_not_found assert result.retryable is False def test_404_correctly_prefixed_model_stays_generic(self): """A properly prefixed id hitting a 404 is a real endpoint problem — it must keep the retryable generic classification.""" e = MockAPIError("404 page not found", status_code=404) result = classify_api_error( e, provider="nvidia", model="nvidia/nemotron-3-ultra-550b-a55b" ) assert result.reason == FailoverReason.unknown assert result.retryable is True def test_404_unknown_bare_model_stays_generic(self): """A local NIM container isn't in the catalogue — no verdict invented.""" e = MockAPIError("404 page not found", status_code=404) result = classify_api_error(e, provider="nvidia", model="my-local-nim") assert result.reason == FailoverReason.unknown assert result.retryable is True # ── Provider policy-block (OpenRouter privacy/guardrail) ── # ── Provider content-policy block (per-prompt safety filter) ── # # Distinct from ``provider_policy_blocked`` above — these are upstream # model-provider safety refusals for THIS prompt, not OpenRouter # account-level data policy. Recovery is fallback model, not config fix. # See issue #18028 — OpenAI Codex was burning 3 retries on identical # refusals before users saw "API failed after 3 retries" on Telegram. def test_message_only_cyber_content_policy_blocked(self): # OpenAI Codex returns this without an HTTP status. Retrying the # same prompt three times only repeats the same policy decision, so # the classifier must jump straight to fallback / abort instead of # leaving it in the retryable ``unknown`` bucket. e = Exception( "This content was flagged for possible cybersecurity risk. If this " "seems wrong, try rephrasing your request. To get authorized for " "security work, join the Trusted Access for Cyber program." ) result = classify_api_error(e, provider="openai-codex", model="gpt-5.5") assert result.reason == FailoverReason.content_policy_blocked assert result.retryable is False assert result.should_fallback is True assert result.should_compress is False def test_400_content_exists_risk_commandcode_moderation(self): # CommandCode gateway (OpenAI-compatible aggregator fronting DeepSeek) # rejects filtered prompts with HTTP 400 "Content Exists Risk" and a # nested param envelope marking isRetryable=false — deterministic for # the unchanged request, so the recovery is the fallback chain, not a # same-provider retry. Without the pattern the 400 fell through to # format_error and the surfaced copy blamed a malformed request. See # #115218. body = { "error": { "message": "Content Exists Risk", "type": "AI_APICallError", "param": { "error": "Content Exists Risk", "statusCode": 400, "name": "AI_APICallError", "message": "Content Exists Risk", "isRetryable": False, "type": "AI_APICallError", }, } } e = MockAPIError( "Error code: 400 - {'error': {'message': 'Content Exists Risk'}}", status_code=400, body=body, ) result = classify_api_error( e, provider="commandcode", model="deepseek/deepseek-v4.1-flash" ) assert result.reason == FailoverReason.content_policy_blocked assert result.retryable is False assert result.should_fallback is True assert result.should_compress is False # ── Payload too large ── def test_413_payload_too_large(self): e = MockAPIError("Request Entity Too Large", status_code=413) result = classify_api_error(e) assert result.reason == FailoverReason.payload_too_large assert result.should_compress is True # ── Context overflow ── # ── Local-inference memory ceiling (oMLX/MLX prefill guard, #52261) ── @pytest.mark.parametrize("message, status_code, body", [ # 0.5.6 prefill guard: a memory peak in BYTES whose remediation tail says "Reduce context # length" — the phrase that used to route it into the compress loop. ("Prefill memory guard rejected request: Prefill would require ~13.87 GB peak, " "dynamic ceiling is 13.50 GB. Reduce context length or lower memory_guard_tier.", 400, None), # 0.5.7 rewording ("predicted peak would require"); cap names survive the verb change. ("process memory limit exceeded: predicted peak would require ~78.57 GB, prefill " "safety cap is 77.76 GB (90% of metal_cap ceiling 86.40 GB). Reduce context size.", 400, None), # Mid-stream the guard exits as a generic 500 (streaming generator drops the code). ("predicted peak would exceed prefill safety cap 77.8GB. Reduce context length.", 500, None), # Status-less: "memory limit exceeded" contains "limit exceeded" and would otherwise read # as billing in the usage-limit disambiguation — the memory rule runs in the message HEAD. ("process memory limit exceeded: predicted peak would require ~78.57 GB. " "Reduce context size.", None, None), # Proxy flattened the wording; only the structured code survives (400 must read it, since # _by_status runs before _by_error_code). ("Request failed.", 400, {"error": {"message": "Request failed.", "code": "prefill_memory_exceeded"}}), ]) def test_memory_ceiling_rejection_is_overloaded_not_overflow(self, message, status_code, body): kwargs = {"status_code": status_code} if status_code is not None else {} if body is not None: kwargs["body"] = body result = classify_api_error(MockAPIError(message, **kwargs), provider="omlx") assert result.reason == FailoverReason.overloaded assert result.should_compress is False assert result.should_rotate_credential is False def test_genuine_context_overflow_still_compresses(self): """Guard against over-reach: a real window overflow must keep its compression recovery.""" e = MockAPIError( "This model's maximum context length is 200000 tokens. However, your " "messages resulted in 250000 tokens.", status_code=400, ) result = classify_api_error(e, provider="omlx") assert result.reason == FailoverReason.context_overflow assert result.should_compress is True # ── Server disconnect + large session ── # ── Provider-specific: Anthropic thinking signature ── @pytest.mark.parametrize("error_code", ["Invalid_Encrypted_Content", "INVALID_ENCRYPTED_CONTENT"]) def test_invalid_encrypted_content_code_is_case_insensitive_for_400(self, error_code): e = MockAPIError( "Error code: 400 - bad request", status_code=400, body={"error": {"code": error_code, "message": "Bad request"}}, ) result = classify_api_error(e, provider="custom", model="gpt-5.4") assert result.reason == FailoverReason.invalid_encrypted_content assert result.retryable is True assert result.should_fallback is False def test_opencode_zen_wrapped_replay_rejection_reaches_replay_strip(self): """OpenCode Zen wraps the rejected encrypted replay in a generic 400.""" e = MockAPIError( "HTTP 400: Error from provider (Console): Upstream request failed: " "[invalid_request_error] reasoning `encrypted_content` was not issued to this caller", status_code=400, ) result = classify_api_error(e, provider="opencode-zen", model="muse-spark-1.3-contributor-free") assert result.reason == FailoverReason.invalid_encrypted_content assert result.retryable is True assert result.should_fallback is False # ── Codex masked encrypted-reasoning replay rejection (#92353) ── _CODEX_MASKED = {"message": "Request blocked.", "type": "invalid_request_error", "param": None, "code": "invalid_prompt"} @pytest.mark.parametrize("error", [ MockAPIError("Error code: 400 - Request blocked.", status_code=400, body=_CODEX_MASKED), # SDK unwraps body["error"] MockAPIError("Request blocked.", status_code=None, body={"error": _CODEX_MASKED}), # SSE ``error`` frame RuntimeError("invalid_prompt: Request blocked."), # ``response.failed`` terminal frame ], ids=["http400", "sse-frame", "response-failed"]) def test_codex_masked_replay_rejection_reaches_replay_strip(self, error): result = classify_api_error(error, provider="openai-codex", model="gpt-5.5") assert result.reason == FailoverReason.invalid_encrypted_content assert result.retryable is False and result.should_fallback is True # format_error's terminal hints kept @pytest.mark.parametrize(("provider", "body", "expected"), [ ("custom", _CODEX_MASKED, FailoverReason.format_error), # same envelope, other provider ("openai-codex", {**_CODEX_MASKED, "message": "Invalid prompt: too long."}, FailoverReason.format_error), ("openai-codex", {**_CODEX_MASKED, "code": "server_error"}, FailoverReason.format_error), ("openai-codex", {**_CODEX_MASKED, "message": "Request blocked. Your request was flagged by our safety system."}, FailoverReason.content_policy_blocked), # #18028 refusal still wins ], ids=["other-provider", "other-message", "other-code", "safety-refusal"]) def test_codex_masked_replay_rejection_stays_narrow(self, provider, body, expected): e = MockAPIError("Error code: 400 - " + body["message"], status_code=400, body=body) assert classify_api_error(e, provider=provider, model="gpt-5.5").reason == expected @pytest.mark.parametrize(("provider", "body", "expected"), [ ("openai-codex", {"detail": "Unsupported content type"}, FailoverReason.invalid_encrypted_content), # Some SDK paths surface only the wrapped message text, no parsed body. ("openai-codex", None, FailoverReason.invalid_encrypted_content), ("openai", {"detail": "Unsupported content type"}, FailoverReason.format_error), # elsewhere a genuine shape 400 ], ids=["codex-dict-body", "codex-message-only", "other-provider"]) def test_codex_unsupported_content_type_detail_reaches_replay_strip(self, provider, body, expected): """#51512: the ChatGPT Codex backend rejects a replayed encrypted-reasoning item as a bare ``{"detail": "Unsupported content type"}`` 400; only the codex provider maps it to the replay strip.""" e = MockAPIError("Error code: 400 - {'detail': 'Unsupported content type'}", status_code=400, body=body) assert classify_api_error(e, provider=provider, model="gpt-5.5").reason == expected def test_thinking_signature_invalid_uses_encrypted_replay_recovery(self): """#70595: the OpenAI code contains "thinking" + "signature", so it must beat the Anthropic thinking-block heuristic and reach the one-shot encrypted-replay strip (retry, no fallback).""" body = {"error": {"code": "thinking_signature_invalid", "message": "The reasoning signature is no longer valid."}} e = MockAPIError(f"Error code: 400 - {body}", status_code=400, body=body) result = classify_api_error(e, provider="openai", model="gpt-5.5") assert result.reason == FailoverReason.invalid_encrypted_content assert result.retryable is True and result.should_fallback is False @pytest.mark.parametrize(("provider", "model", "message", "code"), [ ("azure-foundry", "gpt-6-astra", "Conflicting authenticated continuation identities.", "invalid_value"), # Custom Responses endpoint wraps the replay rejection in a generic bad_request (#95834). ("custom", "gpt-5.6", "The encrypted content could not be decrypted or parsed.", "bad_request"), ], ids=["azure-continuation-identities", "custom-decrypted-or-parsed"]) def test_message_only_replay_rejection_is_invalid_encrypted_content(self, provider, model, message, code): """Endpoints whose ``code`` is generic; the message wording alone must decide.""" e = MockAPIError( f"Error code: 400 - {{'error': {{'message': '{message}', 'type': 'invalid_request_error', " f"'param': 'input', 'code': '{code}'}}", status_code=400, body={"error": {"message": message, "type": "invalid_request_error", "param": "input", "code": code}}, ) result = classify_api_error(e, provider=provider, model=model) assert result.reason == FailoverReason.invalid_encrypted_content assert result.retryable is True assert result.should_fallback is False # ── Reasoning-mandatory route rejecting a disable ── def test_reasoning_mandatory_400_is_retryable_not_format_error(self): e = MockAPIError( "Error code: 400 - This request is not valid. Check the model name " "and other parameters. Additional info: Reasoning is mandatory for " "this endpoint and cannot be disabled.", status_code=400, ) result = classify_api_error(e, provider="nous", model="z-ai/glm-5.3-flash") assert result.reason == FailoverReason.reasoning_mandatory assert result.retryable is True assert result.should_fallback is False assert result.should_compress is False def test_reasoning_field_rejection_is_reasoning_mandatory(self): """A 400 rejecting a reasoning wire control by name — reversed ("reasoning_effort 'none' unsupported; use ...", #114460), forward ("Unrecognized request argument supplied: reasoning_effort"), or an enum rejection whose only field name sits in the structured 'param' tail (commandcode.ai, #115277) — takes the drop-the-disable rung, not the format_error abort; a model-id segment (kimi-k2-thinking) stays route gating.""" for msg in ( "Error code: 400 - reasoning_effort 'none' unsupported; use minimal|low|medium|high|xhigh", "Unrecognized request argument supplied: reasoning_effort", "Error code: 400 - {'error': {'message': 'Invalid option: expected one of " "\"low\"|\"medium\"|\"high\"|\"xhigh\"|\"max\"', 'type': 'invalid_request_error', " "'param': 'reasoning_effort'}}", ): result = classify_api_error(MockAPIError(msg, status_code=400), provider="custom", model="m") assert result.reason == FailoverReason.reasoning_mandatory, msg assert result.retryable is True and result.should_fallback is False gated = classify_api_error( MockAPIError("The model kimi-k2-thinking is not supported when using this account", status_code=400), provider="custom", model="kimi-k2-thinking", ) assert gated.reason != FailoverReason.reasoning_mandatory def test_structured_invalid_reasoning_effort_400_never_compresses(self): """A custom Responses relay rejects an unsupported ``reasoning.effort`` with a message-less structured 400 (``param`` + ``error_code: invalid_reasoning_effort``, #100536). No wording rule can match it; before, the empty message fell to the large-session overflow heuristic and the loop compressed a tiny conversation. Now it is a reasoning-field rejection with ``should_compress`` off on every session size; a genuine context-window 400 still compresses.""" body = {"error": {"param": "reasoning.effort", "error_code": "invalid_reasoning_effort", "retryable": False}} for approx_tokens, num_messages in ((77, 3), (90000, 100)): result = classify_api_error( MockAPIError(f"Error code: 400 - {body}", status_code=400, body=body), provider="custom", model="m", approx_tokens=approx_tokens, context_length=200000, num_messages=num_messages, ) assert result.reason == FailoverReason.reasoning_mandatory, approx_tokens assert result.should_compress is False overflow = classify_api_error( MockAPIError("This model's maximum context length is 128000 tokens. Please reduce the length " "of the messages.", status_code=400), provider="custom", model="m", approx_tokens=77, num_messages=3, ) assert overflow.reason == FailoverReason.context_overflow and overflow.should_compress is True def test_openai_unsupported_none_effort_body_is_reasoning_mandatory(self): """OpenAI's real 400 for ``reasoning.effort: none`` on a model whose ladder has no ``none`` (o3/o4-mini, gpt-5/gpt-5-codex; ``none`` is gpt-5.1+): the SDK message carries the body — ``param: reasoning.effort`` plus ``code: unsupported_value`` — and must take the drop-the-disable retry rung, not a format abort.""" body = {"error": {"message": "Unsupported value: 'none' is not supported with this model. Supported values " "are: 'low', 'medium', and 'high'.", "type": "invalid_request_error", "param": "reasoning.effort", "code": "unsupported_value"}} msg = f"Error code: 400 - {body}" assert is_reasoning_field_rejection(msg) result = classify_api_error(MockAPIError(msg, status_code=400, body=body), provider="openai-api", model="o4-mini") assert result.reason == FailoverReason.reasoning_mandatory assert result.retryable is True and result.should_fallback is False # ── Provider-specific: llama.cpp grammar-parse ── def test_llama_cpp_unable_to_generate_parser_template(self): e = MockAPIError( "Unable to generate parser for this template. " "Automatic parser generation failed: error parsing grammar", status_code=400, ) result = classify_api_error(e, provider="custom", model="local-llama") assert result.reason == FailoverReason.llama_cpp_grammar_pattern assert result.retryable is True assert result.should_compress is False def test_openai_regex_lookaround_rejection_strips_pattern_and_retries(self): """Strict OpenAI-compatible endpoints reject ``pattern`` lookaround with a 400 (#42631). Driven through the production path (classifier → ``recover_after_classification``): the lookaround ``pattern`` must be stripped from ``agent.tools`` and the turn retried.""" from agent.turn_recovery import recover_after_classification from agent.turn_retry_state import TurnRetryState class _Agent: log_prefix = "" api_mode = "chat_completions" provider = "custom" model = "gpt-5.5" base_url = "http://relay.example/v1" tools = [{ "type": "function", "function": { "name": "send", "parameters": { "type": "object", "properties": {"email": {"type": "string", "pattern": r"^(?!no-reply).+@.+$"}}, }, }, }] def _recover_with_credential_pool(self, **kwargs): return False, False def __getattr__(self, name): return lambda *args, **kwargs: None e = MockAPIError( "Invalid JSON schema: regex lookaround is not supported. Found at $.properties.email.pattern.", status_code=400, ) classified = classify_api_error(e, provider="custom", model="gpt-5.5") assert classified.reason == FailoverReason.llama_cpp_grammar_pattern agent = _Agent() retry_now, _ = recover_after_classification( agent, e, classified, TurnRetryState(), status_code=400, error_context=None, messages=[], api_messages=[], ) assert retry_now is True assert "pattern" not in agent.tools[0]["function"]["parameters"]["properties"]["email"] # A generic schema 400 without the lookaround sentence stays a plain client error. other = classify_api_error( MockAPIError("Invalid JSON schema: regex syntax error in pattern", status_code=400), provider="custom" ) assert other.reason != FailoverReason.llama_cpp_grammar_pattern def test_qwen_apply_prompt_template_no_user_query_not_llama_cpp_grammar(self): """Local engines wrap Qwen raise_exception as applyPromptTemplate 400. Must NOT classify as llama_cpp_grammar_pattern (which strips tool schema keywords and retries). Fail fast as format_error so the user sees a request-shape failure instead of a misleading template/parser loop — typical after context overflow + failed compression. """ e = MockAPIError( "Engine protocol applyPromptTemplate request returned 400: " '{"error":{"code":400,"message":"Unable to generate parser for ' "this template. Automatic parser generation failed: " "While executing CallExpression ... multi_step_tool %} " "{{- raise_exception('No user query found in messages')", status_code=400, ) result = classify_api_error( e, provider="custom", model="qwen/qwen3.6-35b-a3b", approx_tokens=226_000, context_length=100_864, ) assert result.reason == FailoverReason.format_error assert result.retryable is False assert result.should_compress is False assert result.should_fallback is True def test_bare_no_user_query_found_is_format_error_even_on_large_session(self): e = MockAPIError("No user query found in messages", status_code=400) result = classify_api_error( e, approx_tokens=226_000, context_length=100_864, ) assert result.reason == FailoverReason.format_error assert result.retryable is False assert result.should_compress is False # ── Provider-specific: Anthropic long-context tier ── def test_anthropic_long_context_tier(self): e = MockAPIError( "Extra usage is required for long context requests over 200k tokens", status_code=429, ) result = classify_api_error(e, provider="anthropic", model="claude-sonnet-4") assert result.reason == FailoverReason.long_context_tier assert result.should_compress is True # ── Provider-specific: Anthropic OAuth 1M-context beta forbidden ── # ── Transport errors ── def test_read_timeout(self): e = ReadTimeout("Read timed out") result = classify_api_error(e) assert result.reason == FailoverReason.timeout assert result.retryable is True def test_connect_error(self): e = ConnectError("Connection refused") result = classify_api_error(e) assert result.reason == FailoverReason.timeout def test_connection_error_builtin(self): e = ConnectionError("Connection reset by peer") result = classify_api_error(e) assert result.reason == FailoverReason.timeout def test_timeout_error_builtin(self): e = TimeoutError("timed out") result = classify_api_error(e) assert result.reason == FailoverReason.timeout # ── Error code classification ── # ── Message-only patterns (no status code) ── def test_message_account_id_token_extraction_failure_is_auth(self): """Codex 'Failed to extract accountId from token' without a status is an auth failure: no retry on the same credential, rotate, fall back (#72911).""" e = Exception("Failed to extract accountId from token") result = classify_api_error(e, provider="openai-codex") assert result.reason == FailoverReason.auth assert result.retryable is False assert result.should_rotate_credential is True assert result.should_fallback is True # ── Message-only usage limit disambiguation (no status code) ── # ── Unknown / fallback ── # ── Format error ── def test_400_litellm_invalid_request_body_shape(self, caplog): """litellm/Bedrock proxy shape (errorMessage/errorCode) → format_error. The proxy in front of Anthropic surfaces the empty-content rejection as {"errorMessage": "...non-empty content...", "errorCode": "INVALID_REQUEST_BODY", "errorArgs": {"reason": "..."}}. Those keys are not the standard error.message / message, so err_body_msg used to come back empty → is_generic=True → mis-routed into compression on a large session. Both the message pattern and the errorCode must be recognized, and a distinct warning must be logged so the condition is observable in the field. """ import logging proxy_msg = ("The provided request body is invalid: claude " "messages.208: all messages must have non-empty content " "except for the optional final assistant message") e = MockAPIError( proxy_msg, status_code=400, body={ "errorMessage": proxy_msg, "errorCode": "INVALID_REQUEST_BODY", "statusCode": 400, "errorArgs": {"reason": "claude messages.208: ..."}, }, ) with caplog.at_level(logging.WARNING, logger="agent.error_classifier"): result = classify_api_error( e, approx_tokens=66000, context_length=200000, num_messages=219, ) assert result.reason == FailoverReason.format_error assert result.retryable is False assert result.should_compress is not True def test_400_top_level_detail_body_is_not_a_bare_400_on_large_session(self): """FastAPI-style ``{"detail": "..."}`` bodies (Codex gateway, Starlette relays) → the descriptive text is read, so the large-session heuristic does not route a model entitlement/retirement rejection into compression (#81558, #106475). ``str(error)`` is the SDK's ``Error code: 400 - {...}`` form, exactly as on the wire. Salvaged from #100783 (@i-Hun).""" detail = "The 'gpt-5.5-codex' model is not supported when using Codex with a ChatGPT account." large = dict(provider="openai-codex", model="gpt-5.5-codex", approx_tokens=109_962, context_length=272_000, num_messages=223) for body in ({"detail": detail}, {"detail": {"message": detail}}): e = MockAPIError(f"Error code: 400 - {body!r}", status_code=400, body=body) result = classify_api_error(e, **large) # type: ignore[arg-type] assert result.reason is not FailoverReason.context_overflow, body assert result.should_compress is False assert result.should_fallback is True assert result.message == detail # Control: the genuinely bare body the heuristic exists for still compresses. bare = classify_api_error( MockAPIError("Error code: 400 - {'error': {'message': 'Error'}}", status_code=400, body={"error": {"message": "Error"}}), **large) # type: ignore[arg-type] assert bare.reason is FailoverReason.context_overflow # ── Peer closed + large session ── # ── Chinese error messages ── # ── Z.AI / Zhipu GLM error messages ── def test_zai_glm_token_limit_overflow(self): """Z.AI GLM's 'tokens in request more than max tokens allowed' (error code 1210) → context_overflow, so the agent compresses instead of blindly retrying. Port of anomalyco/opencode#35671.""" e = MockAPIError( '{"error": {"code": "1210", "message": ' '"tokens in request more than max tokens allowed"}}', status_code=400, ) result = classify_api_error(e, provider="zai") assert result.reason == FailoverReason.context_overflow # ── vLLM / local inference server error messages ── # ── Result metadata ── def test_message_extracted(self): e = MockAPIError( "outer", status_code=500, body={"error": {"message": "Internal server error occurred"}}, ) result = classify_api_error(e) assert result.message == "Internal server error occurred" # ── Test: Adversarial / edge cases (from live testing) ───────────────── class TestAdversarialEdgeCases: """Edge cases discovered during live testing with real SDK objects.""" def test_500_with_none_body(self): e = MockAPIError("fail", status_code=500, body=None) result = classify_api_error(e) assert result.reason == FailoverReason.server_error def test_non_dict_body(self): """Some providers return strings instead of JSON.""" class StringBodyError(Exception): status_code = 400 body = "just a string" result = classify_api_error(StringBodyError("bad")) assert result.reason == FailoverReason.format_error def test_three_level_cause_chain(self): inner = MockAPIError("inner", status_code=429) middle = Exception("middle") middle.__cause__ = inner outer = RuntimeError("outer") outer.__cause__ = middle result = classify_api_error(outer) assert result.status_code == 429 assert result.reason == FailoverReason.rate_limit def test_400_with_rate_limit_text(self): """Some providers send rate limits as 400 instead of 429.""" e = MockAPIError( "rate limit policy", status_code=400, body={"error": {"message": "rate limit exceeded on this model"}}, ) result = classify_api_error(e, provider="openrouter") assert result.reason == FailoverReason.rate_limit def test_400_anthropic_extra_usage_exhausted(self): """Anthropic returns 400 with 'out of extra usage' when the user's extra-usage allowance is depleted. Must classify as billing so the fallback chain engages (with credential rotation) instead of the generic format_error path, which never rotates. (#11736, #13170) #82154: the identical body is ALSO returned when Anthropic's content filter rejects part of the request on a subscription OAuth token, so the billing verdict must be marked unverified — downstream surfaces hedge instead of asserting exhaustion, and the credential pool skips the one-hour billing bench.""" e = MockAPIError( "You're out of extra usage. Add more at claude.ai/settings/usage and keep going.", status_code=400, body={"error": { "type": "invalid_request_error", "message": "You're out of extra usage. Add more at claude.ai/settings/usage and keep going.", }}, ) result = classify_api_error(e, provider="anthropic") assert result.reason == FailoverReason.billing assert result.should_fallback is True assert result.retryable is False assert result.should_rotate_credential is True assert result.billing_unverified is True assert result.error_context.get("possible_content_filter") is True def test_400_unambiguous_billing_body_is_not_marked_unverified(self): """A 400 whose billing evidence is NOT the ambiguous 'out of extra usage' body keeps a confirmed verdict (#82154).""" e = MockAPIError( "Your credit balance is too low to access the Anthropic API.", status_code=400, body={"error": { "type": "invalid_request_error", "message": "Your credit balance is too low to access the Anthropic API.", }}, ) result = classify_api_error(e, provider="anthropic") assert result.reason == FailoverReason.billing assert result.billing_unverified is False def test_statusless_extra_usage_is_marked_unverified(self): """Adapters can strip the HTTP status from the Anthropic 400; the message-only path must carry the same ambiguity marking (#82154).""" e = Exception( "You're out of extra usage. Add more at claude.ai/settings/usage and keep going." ) result = classify_api_error(e, provider="anthropic") assert result.reason == FailoverReason.billing assert result.billing_unverified is True def test_200_with_error_body(self): """200 status with error in body — should be unknown, not crash.""" class WeirdSuccess(Exception): status_code = 200 body = {"error": {"message": "loading"}} result = classify_api_error(WeirdSuccess("model loading")) assert result.reason == FailoverReason.unknown def test_connection_refused_error(self): e = ConnectionRefusedError("Connection refused: localhost:11434") result = classify_api_error(e, provider="ollama") assert result.reason == FailoverReason.timeout def test_disconnect_pattern_ordering(self): """Disconnect + large session must beat generic transport catch.""" class FakeRemoteProtocol(Exception): pass # Type name isn't in _TRANSPORT_ERROR_TYPES but message has disconnect pattern e = Exception("peer closed connection without sending complete message") result = classify_api_error(e, approx_tokens=150000, context_length=200000) assert result.reason == FailoverReason.context_overflow assert result.should_compress is True def test_deepseek_402_chinese(self): """Chinese billing message should still match billing patterns.""" # "余额不足" doesn't match English billing patterns, but 402 defaults to billing e = MockAPIError("余额不足", status_code=402) result = classify_api_error(e, provider="deepseek") assert result.reason == FailoverReason.billing # ── Regression: dict-typed message field (Issue #11233) ── # Broader non-string type guards — defense against other provider quirks. # ── Test: SSL/TLS transient errors ───────────────────────────────────── class TestSSLTransientPatterns: """SSL/TLS alerts mid-stream should retry as timeout, not unknown, and should NOT trigger context compression even on a large session. Motivation: OpenSSL 3.x changed TLS alert error code format (`SSLV3_ALERT_BAD_RECORD_MAC` → `SSL/TLS_ALERT_BAD_RECORD_MAC`), breaking string-exact matching in downstream retry logic. We match stable substrings instead. """ def test_bad_record_mac_classifies_as_timeout(self): """OpenSSL 3.x mid-stream bad record mac alert.""" e = Exception("[SSL: BAD_RECORD_MAC] sslv3 alert bad record mac (_ssl.c:2580)") result = classify_api_error(e) assert result.reason == FailoverReason.timeout assert result.retryable is True assert result.should_compress is False def test_plain_disconnect_on_large_session_still_compresses(self): """Regression guard: the context-overflow-via-disconnect path (non-SSL disconnects on large sessions) must still trigger compression. Only SSL-specific disconnects skip it. """ e = Exception("Server disconnected without sending a response") result = classify_api_error( e, approx_tokens=180000, context_length=200000, num_messages=300, ) assert result.reason == FailoverReason.context_overflow assert result.should_compress is True # ── Test: SSL certificate verification failures (fail fast) ──────────── class TestSSLCertVerificationFailFast: """Certificate verification failures are deterministic for the host — a TLS-inspecting proxy, missing custom CA, expired or self-signed cert fails identically on every retry. They must classify as non-retryable ``ssl_cert_verification`` so the user sees the fix hint immediately, instead of matching the transient "[ssl:" pattern and retrying forever. Inspired by Claude Code v2.1.199 (July 2026). """ def test_python_cert_verify_failed_is_non_retryable(self): import ssl e = ssl.SSLCertVerificationError( 1, "[SSL: CERTIFICATE_VERIFY_FAILED] certificate verify failed: " "unable to get local issuer certificate (_ssl.c:1006)", ) result = classify_api_error(e) assert result.reason == FailoverReason.ssl_cert_verification assert result.retryable is False assert result.should_compress is False def test_transient_ssl_alert_still_retries(self): """Regression guard: genuine transient alerts keep retrying.""" e = Exception("[SSL: BAD_RECORD_MAC] sslv3 alert bad record mac") result = classify_api_error(e) assert result.reason == FailoverReason.timeout assert result.retryable is True # ── Test: RateLimitError without status_code (Copilot/GitHub Models) ────────── class TestProviderCodeOnlyErrors: """Bare ``{"error": {"code": …}}`` bodies with no HTTP status map to the provider's structured reason instead of ``unknown`` (#70414).""" @pytest.mark.parametrize("provider, code, reason", [ ("gemini", "UNAVAILABLE", FailoverReason.overloaded), ("google", "DEADLINE_EXCEEDED", FailoverReason.timeout), ("vertex", "INTERNAL", FailoverReason.server_error), ("anthropic", "API_ERROR", FailoverReason.server_error), ("openai-codex", "SERVER_ERROR", FailoverReason.server_error), ]) def test_provider_native_code_maps_to_structured_reason(self, provider, code, reason): e = MockAPIError(code, body={"error": {"code": code}}) result = classify_api_error(e, provider=provider) assert result.reason == reason assert result.retryable is True assert result.should_rotate_credential is False def test_code_meaning_does_not_leak_across_providers(self): e = MockAPIError("UNAVAILABLE", body={"error": {"code": "UNAVAILABLE"}}) assert classify_api_error(e, provider="openai").reason == FailoverReason.unknown def test_gemini_wire_body_numeric_code_falls_back_to_status(self): """Gemini's real body carries the HTTP status in ``error.code`` and the symbolic code in ``error.status``; the numeric code must not shadow it.""" body = {"error": {"code": 503, "status": "UNAVAILABLE", "message": "Service unavailable."}} e = MockAPIError("Service unavailable.", body=body) assert classify_api_error(e, provider="gemini").reason == FailoverReason.overloaded def test_anthropic_rate_limit_error_code_rotates_credential(self): e = MockAPIError("rate limited", body={"error": {"code": "rate_limit_error"}}) result = classify_api_error(e, provider="anthropic") assert result.reason == FailoverReason.rate_limit assert result.should_rotate_credential is True assert result.should_fallback is True class TestRateLimitErrorWithoutStatusCode: """Regression tests for the Copilot/GitHub Models edge case where the OpenAI SDK raises RateLimitError but does not populate .status_code.""" def _make_rate_limit_error(self, status_code=None): """Create an exception whose class name is 'RateLimitError' with an optionally missing status_code, mirroring the OpenAI SDK shape.""" cls = type("RateLimitError", (Exception,), {}) e = cls("You have exceeded your rate limit.") e.status_code = status_code # None simulates the Copilot case return e def test_rate_limit_error_without_status_code_classified_as_rate_limit(self): """RateLimitError with status_code=None must classify as rate_limit.""" e = self._make_rate_limit_error(status_code=None) result = classify_api_error(e, provider="copilot", model="gpt-4o") assert result.reason == FailoverReason.rate_limit def test_rate_limit_error_with_status_code_429_classified_as_rate_limit(self): """RateLimitError that does set status_code=429 still classifies correctly.""" e = self._make_rate_limit_error(status_code=429) result = classify_api_error(e, provider="copilot", model="gpt-4o") assert result.reason == FailoverReason.rate_limit def test_other_error_without_status_code_not_forced_to_rate_limit(self): """A non-RateLimitError with missing status_code must NOT be forced to 429.""" cls = type("APIError", (Exception,), {}) e = cls("something went wrong") e.status_code = None result = classify_api_error(e, provider="copilot", model="gpt-4o") assert result.reason != FailoverReason.rate_limit # ── Test: multimodal_tool_content_unsupported pattern ─────────────────── class TestMultimodalToolContentUnsupported: """Issue #27344 — providers that reject list-type tool message content should be classified as ``multimodal_tool_content_unsupported`` so the retry loop can downgrade screenshots to text and try again. """ def test_xiaomi_mimo_text_is_not_set_pattern(self): """The actual Xiaomi MiMo 400 wording from the bug report.""" e = MockAPIError( "Error code: 400 - {'error': {'code': '400', 'message': 'Param Incorrect', 'param': 'text is not set', 'type': ''}}", status_code=400, ) result = classify_api_error(e, provider="xiaomi", model="mimo-v2.5") assert result.reason == FailoverReason.multimodal_tool_content_unsupported assert result.retryable is True def test_unrelated_400_is_not_misclassified(self): """Make sure the patterns don't false-positive on normal 400s.""" e = MockAPIError("bad request: missing field 'model'", status_code=400) result = classify_api_error(e, provider="openrouter", model="anthropic/claude-sonnet-4") assert result.reason != FailoverReason.multimodal_tool_content_unsupported class TestOpenRouterUpstreamRateLimit: """Distinguish upstream-provider 429 from account-level 429 on OpenRouter. When an upstream model (DeepSeek, Anthropic, etc.) rate-limits OpenRouter's aggregate traffic, OpenRouter returns 429 with the outer message "Provider returned error". The user's key is healthy — we must fall back to a different model, NOT mark the credential exhausted. """ def test_openrouter_upstream_429_classified_as_upstream_rate_limit(self): """OpenRouter 429 with 'Provider returned error' → upstream_rate_limit.""" e = MockAPIError( "Provider returned error", status_code=429, body={ "error": { "message": "Provider returned error", "code": 429, "metadata": { "provider_name": "DeepSeek", "raw": '{"error":{"message":"Rate limit exceeded"}}', }, } }, ) result = classify_api_error(e, provider="openrouter", model="deepseek/deepseek-v4-flash") assert result.reason == FailoverReason.upstream_rate_limit assert result.should_rotate_credential is False assert result.should_fallback is True assert result.error_context.get("upstream_provider") == "DeepSeek" def test_account_level_429_still_rotates_credential(self): """A real account-level 429 (no upstream wrapper) → rate_limit, rotates.""" e = MockAPIError( "Rate limit exceeded: 200 requests per minute", status_code=429, body={ "error": { "message": "Rate limit exceeded: 200 requests per minute", "code": 429, } }, ) result = classify_api_error(e, provider="openrouter", model="deepseek/deepseek-v4-flash") assert result.reason == FailoverReason.rate_limit assert result.should_rotate_credential is True class TestCommandCodeUpstreamUnavailable: """An explicit upstream outage is not a credential rate limit.""" @pytest.mark.parametrize( ("provider", "status_code"), [ ("commandcode", 429), ("commandcode-anthropic", 429), ("commandcode", None), ("other-gateway", 429), ], ) def test_upstream_unavailable_keeps_credential_healthy(self, provider, status_code): e = MockAPIError( "Upstream model provider is temporarily unavailable. Please try again in a moment.", status_code=status_code, ) result = classify_api_error(e, provider=provider, model="deepseek/deepseek-v4-flash") assert result.reason == FailoverReason.overloaded assert result.should_rotate_credential is False @pytest.mark.parametrize( "message", [ "Rate limit exceeded: 200 requests per minute", "Upstream model provider is temporarily unavailable because this account is rate limited.", ], ) def test_non_outage_rate_limits_still_rotate_credential(self, message): e = MockAPIError(message, status_code=429) result = classify_api_error( e, provider="commandcode", model="deepseek/deepseek-v4-flash" ) assert result.reason == FailoverReason.rate_limit assert result.should_rotate_credential is True # ── HTTP 408 request timeout ──────────────────────────────────────────── class Test408RequestTimeout: """HTTP 408 must never fall through to the non-retryable 'other 4xx' bucket (that abort persists an empty assistant turn — the "disappeared conversation" / blank-bubble symptom). ALL 408s are classified as a transient ``timeout``: retryable, and explicitly NOT should_compress. Design decision (field 2026-07-02): even the GitHub Copilot ``user_request_timeout`` / "Timed out reading request body ... use a smaller request size" case is a plain retry, NOT auto-compression. Real data showed the 408 is probabilistic jitter well below the hard prompt ceiling — the same ~785k-token request that 408'd once succeeded on the next attempt at ~786k — so retrying the same body usually works, and auto-compaction would silently delete conversation history for a merely transient timeout. Genuine over-window prompts surface as 413 / context_overflow (their own compression path); users compact 408-prone long sessions deliberately via ``/compress``. """ def test_copilot_oversized_body_408_retries_as_timeout_not_compress(self): # The exact shape GitHub Copilot returns on a long session. It must # retry (timeout), and must NOT auto-compress. e = MockAPIError( "Error code: 408 - {'error': {'message': 'Timed out reading " "request body. Try again, or use a smaller request size.', " "'code': 'user_request_timeout'}}", status_code=408, body={"error": {"message": "Timed out reading request body. " "Try again, or use a smaller request size.", "code": "user_request_timeout"}}, ) result = classify_api_error(e, provider="copilot", model="claude-opus-4.8") assert result.reason == FailoverReason.timeout assert result.retryable is True assert result.should_compress is False def test_stale_breaker_runtime_error_triggers_fallback_not_retry(self): # The cross-turn stale-call circuit breaker (_check_stale_giveup in # chat_completion_helpers.py) raises a RuntimeError when the provider # has been unresponsive for N consecutive stale attempts. This must # be classified as non-retryable + should_fallback so the retry loop # activates the fallback provider immediately instead of burning all # max_retries against the same dead provider (each retry hitting the # circuit breaker instantly with zero network overhead). e = RuntimeError( "Provider has been unresponsive (no response received) for " "6 consecutive stale attempts — aborting this call to " "avoid an indefinite stall. Switch models or start a new " "session, then retry." ) result = classify_api_error( e, provider="openrouter", model="anthropic/claude-fable-5", approx_tokens=126327, context_length=200000, num_messages=274, ) assert result.reason == FailoverReason.timeout assert result.retryable is False assert result.should_fallback is True assert result.should_compress is False # ── Test: connection/DNS failure message patterns on generic exception types ── # Port of anomalyco/opencode#40707 (expand retryable error patterns): errors # whose TYPE is generic (RuntimeError/Exception from local shims, MCP bridges, # re-raising SDKs) but whose MESSAGE carries a connection-establishment or DNS # failure must classify as retryable transport, not FailoverReason.unknown. class TestConnectionMessagePatterns: """Generic-typed connect/DNS failures route to the transport bucket.""" @pytest.mark.parametrize("message", [ "connect ECONNREFUSED 127.0.0.1:11434", "Connection refused by proxy", "getaddrinfo failed", "getaddrinfo ENOTFOUND api.example.com", "[Errno -3] Temporary failure in name resolution", "[Errno 8] nodename nor servname provided, or not known", "getaddrinfo EAI_AGAIN openrouter.ai", "Name or service not known", "No route to host", "[Errno 101] Network is unreachable", "fetch failed", "TypeError: Failed to fetch", "upstream connect error or disconnect/reset before headers", ]) def test_generic_exception_with_connect_failure_message_is_timeout(self, message): # RuntimeError — NOT in _TRANSPORT_ERROR_TYPES, not a ConnectionError # subclass, no status code. Without message matching this falls to # FailoverReason.unknown and misses the eager transport fallback. result = classify_api_error(RuntimeError(message)) assert result.reason == FailoverReason.timeout, message assert result.retryable is True assert result.should_compress is False def test_connect_failure_never_routes_to_compression_on_large_session(self): # A connection that was never established is not an overflow signal, # even when the session is huge (the disconnect+large-session # heuristic must not apply to connect-phase failures). result = classify_api_error( RuntimeError("connect ECONNREFUSED 10.0.0.5:443"), approx_tokens=180000, context_length=200000, num_messages=400, ) assert result.reason == FailoverReason.timeout assert result.should_compress is False def test_midstream_disconnect_patterns_still_use_disconnect_path(self): # "connection reset by peer" is deliberately NOT in the connect-phase # list — it stays on the _SERVER_DISCONNECT_PATTERNS path, which # routes large sessions to context-overflow compression. result = classify_api_error( RuntimeError("Connection reset by peer"), approx_tokens=180000, context_length=200000, num_messages=400, ) assert result.reason == FailoverReason.context_overflow assert result.should_compress is True def test_plain_unknown_error_still_unknown(self): # Guard against over-matching: an unrelated message stays unknown. result = classify_api_error(RuntimeError("something exploded")) assert result.reason == FailoverReason.unknown # ── Test: throttle vs overflow disambiguation + new overflow shapes ───── # Port of anomalyco/opencode#37848 (expand context overflow patterns + # rate-limit exclusion guard). class TestThrottleVsOverflowDisambiguation: """Throttle messages that mention tokens must NOT route to compression.""" def test_bedrock_throttling_too_many_tokens_is_rate_limit(self): # AWS Bedrock (and some proxies) surface throttling as # "Throttling error: Too many tokens, please wait before trying # again." — the "too many tokens" fragment sits in # _CONTEXT_OVERFLOW_PATTERNS, so before the "throttling" rate-limit # pattern this compressed a healthy session on every throttle. e = Exception( "Throttling error: Too many tokens, please wait before trying again." ) result = classify_api_error(e, provider="bedrock", model="claude") assert result.reason == FailoverReason.rate_limit assert result.should_compress is False def test_plain_too_many_tokens_still_overflow(self): # Without any throttle wording, "Too many tokens" remains a # context-overflow signal (Z.AI / GLM family wording). e = Exception("Too many tokens") result = classify_api_error(e, provider="zai", model="glm-5") assert result.reason == FailoverReason.context_overflow assert result.should_compress is True class TestExpandedOverflowPatterns: """New provider overflow wordings route into compression recovery.""" def test_maximum_allowed_input_length_is_overflow(self): # Together/Fireworks-style wording — matched no pattern before. e = Exception( "Input length 131393 exceeds the maximum allowed input length " "of 131040 tokens." ) result = classify_api_error(e, provider="together", model="m") assert result.reason == FailoverReason.context_overflow assert result.should_compress is True def test_request_too_large_message_only_is_payload_too_large(self): # Anthropic's structured 413 type re-wrapped by a proxy with no # status attribute — was falling through to `unknown`. e = Exception( '{"error":{"type":"request_too_large",' '"message":"Request exceeds the maximum size"}}' ) result = classify_api_error(e, provider="anthropic", model="m") assert result.reason == FailoverReason.payload_too_large assert result.should_compress is True def test_longer_than_context_length_still_overflow(self): # Regression guard for wordings that already matched. e = Exception( "The input (516368 tokens) is longer than the model's context " "length (262144 tokens)." ) result = classify_api_error(e, provider="openrouter", model="m") assert result.reason == FailoverReason.context_overflow class TestServerInjectedParameterRejection: """A 400 blaming a parameter the client never sent is a server-side flake. The Codex backend (chatgpt.com/backend-api/codex) intermittently adds ``prompt_cache_retention`` to its own upstream call and then rejects it, so an identical request succeeds on retry ~80% of the time. Hermes never sends that field on this route, so the 400 is not a deterministic request-shape error and must stay retryable instead of aborting the turn. """ RETENTION_BODY = { "message": "prompt_cache_retention is not supported on this model", "type": "invalid_request_error", "param": "prompt_cache_retention", "code": "invalid_parameter", } def test_codex_retention_400_is_retryable_server_error(self): e = MockAPIError( "Error code: 400 - {'error': {'message': 'prompt_cache_retention " "is not supported on this model', 'type': 'invalid_request_error', " "'param': 'prompt_cache_retention', 'code': 'invalid_parameter'}}", status_code=400, body=dict(self.RETENTION_BODY), ) result = classify_api_error( e, provider="openai-codex", model="gpt-5.6-sol", approx_tokens=546912, context_length=272000, num_messages=576, ) assert result.reason == FailoverReason.server_error assert result.retryable is True # Retrying the identical request is the recovery — do NOT enter the # compression loop (the context was never the problem). assert result.should_compress is False def test_codex_retention_400_nested_error_body_is_retryable(self): """The same rejection arrives wrapped in an ``error`` envelope too.""" e = MockAPIError( "prompt_cache_retention is not supported on this model", status_code=400, body={"error": dict(self.RETENTION_BODY)}, ) result = classify_api_error( e, provider="openai-codex", model="gpt-5.6-sol", ) assert result.reason == FailoverReason.server_error assert result.retryable is True def test_codex_gateway_terse_retention_400_is_retryable(self): """The Codex gateway's own validator uses a bare ``detail`` body.""" e = MockAPIError( "Unsupported parameter: prompt_cache_retention", status_code=400, body={"detail": "Unsupported parameter: prompt_cache_retention"}, ) result = classify_api_error( e, provider="openai-codex", model="gpt-5.6-sol", ) assert result.reason == FailoverReason.server_error assert result.retryable is True def test_small_session_retention_400_is_still_retryable(self): """Must not depend on the context-size heuristic — a tiny request gets the identical spontaneous rejection (reproduced live).""" e = MockAPIError( "prompt_cache_retention is not supported on this model", status_code=400, body=dict(self.RETENTION_BODY), ) result = classify_api_error( e, provider="openai-codex", model="gpt-5.6-sol", approx_tokens=50, num_messages=1, ) assert result.reason == FailoverReason.server_error assert result.retryable is True def test_other_unsupported_parameter_400_stays_non_retryable(self): """Boundary: a genuine client-sent bad parameter is deterministic and must keep failing fast as a format_error (the existing behaviour).""" e = MockAPIError( "Unsupported parameter: 'max_tokens' is not supported with this " "model. Use 'max_completion_tokens' instead.", status_code=400, body={ "message": "Unsupported parameter: 'max_tokens' is not supported.", "type": "invalid_request_error", "param": "max_tokens", "code": "unsupported_parameter", }, ) result = classify_api_error( e, provider="openai-codex", model="gpt-5.6-sol", ) assert result.reason == FailoverReason.format_error assert result.retryable is False def test_retention_rejection_from_meta_host_stays_non_retryable(self): """Boundary: on api.meta.ai / Bedrock Mantle Hermes DOES send ``prompt_cache_retention`` deliberately, so a rejection there is a real client-side request error and must not be retried blindly.""" e = MockAPIError( "prompt_cache_retention is not supported on this model", status_code=400, body=dict(self.RETENTION_BODY), ) result = classify_api_error( e, provider="meta-ai", model="muse-spark-1.2", ) assert result.reason == FailoverReason.format_error assert result.retryable is False @pytest.mark.parametrize("status_code", [500, 502]) def test_retention_rejection_via_5xx_proxy_is_retryable(self, status_code): """Sibling path: a proxy in front of the route can surface the same injected-parameter rejection as 5xx, where the request-validation guard would also wrongly fail it fast as a format_error.""" e = MockAPIError( "Unsupported parameter: prompt_cache_retention", status_code=status_code, body={"error": dict(self.RETENTION_BODY)}, ) result = classify_api_error( e, provider="openai-codex", model="gpt-5.6-sol", ) assert result.reason == FailoverReason.server_error assert result.retryable is True @pytest.mark.parametrize("status_code", [500, 502]) def test_other_bad_parameter_via_5xx_stays_non_retryable(self, status_code): """Boundary for the sibling path: the codex.nekos.me 502-on-bad-param behaviour must keep failing fast (regression guard for that fix).""" e = MockAPIError( "Unknown parameter: 'frequency_penalty'", status_code=status_code, body={"error": {"message": "Unknown parameter: 'frequency_penalty'", "code": "unknown_parameter"}}, ) result = classify_api_error(e, provider="custom", model="m") assert result.reason == FailoverReason.format_error assert result.retryable is False # ── Test: Nous welcome tier (free tier) refusals ─────────────────────── class TestNousWelcomeTier: """The Nous gateway's welcome-tier contract: a structured 429 body carries ``reason`` / ``retry_after`` / ``alternates`` / ``upgrade_url``; a 400/403 names the wrong host or a dark tier in its message. The parsed refusal rides ``error_context``.""" @staticmethod def _refusal(reason, retry_after=0, **extra): body = {"status": 429, "message": "refused", "reason": reason, "retry_after": retry_after, **extra} return MockAPIError(f"Error code: 429 - {body}", status_code=429, body=body, headers={"retry-after": str(retry_after)}) def test_model_not_free_is_a_non_retryable_gate_with_fallback(self): err = self._refusal("model_not_free", alternates=["nous/welcome"], upgrade_url="https://portal.example/upgrade") result = classify_api_error(err, provider="nous", api_key=make_jwt(), model="gpt-5") assert result.reason == FailoverReason.model_not_found assert result.retryable is False assert result.should_fallback is True assert result.should_rotate_credential is False refusal = result.error_context["welcome_refusal"] assert refusal["reason"] == "model_not_free" assert refusal["alternates"] == ["nous/welcome"] assert refusal["upgrade_url"] == "https://portal.example/upgrade" def test_feature_not_free_is_the_same_gate(self): result = classify_api_error(self._refusal("feature_not_free"), provider="nous", api_key=make_jwt()) assert result.reason == FailoverReason.model_not_found assert result.retryable is False @pytest.mark.parametrize("reason", ["at_capacity", "admission_closed", "rate_limited"]) def test_capacity_refusals_are_rate_limits_that_honour_retry_after(self, reason): result = classify_api_error(self._refusal(reason, retry_after=30), provider="nous", api_key=make_jwt(), model="nous/welcome") assert result.reason == FailoverReason.rate_limit assert result.retryable is True assert result.should_fallback is True ctx = result.error_context assert ctx["welcome_refusal"]["retry_after"] == 30 assert ctx["reset_at"] > 0 def test_retry_after_zero_carries_no_reset(self): result = classify_api_error(self._refusal("at_capacity", retry_after=0), provider="nous", api_key=make_jwt()) assert "reset_at" not in result.error_context def test_unknown_reason_is_not_the_welcome_shape(self): err = MockAPIError("Error code: 429", status_code=429, body={"status": 429, "message": "x", "reason": "something_else", "retry_after": 5}) result = classify_api_error(err, provider="nous", api_key=make_jwt()) assert "welcome_refusal" not in result.error_context def test_anonymous_jwt_on_the_paid_host_is_deterministic(self): body = {"status": 400, "message": "Anonymous accounts must use https://welcome-api.nousresearch.com for inference."} err = MockAPIError(f"Error code: 400 - {body}", status_code=400, body=body) result = classify_api_error(err, provider="nous", api_key=make_jwt(), model="nous/welcome") assert result.reason == FailoverReason.format_error assert result.retryable is False and result.should_fallback is True assert result.error_context["welcome_route"] == "anon_on_paid_host" def test_named_caller_on_the_welcome_host_is_deterministic(self): body = {"status": 400, "message": "This endpoint serves anonymous Hermes Agent accounts only. Use https://inference-api.nousresearch.com with your API key or signed-in account."} err = MockAPIError(f"Error code: 400 - {body}", status_code=400, body=body) result = classify_api_error(err, provider="nous", api_key=make_jwt(account_tier="free")) assert result.error_context["welcome_route"] == "named_on_welcome_host" assert result.retryable is False def test_dark_tier_403_never_triggers_a_credential_refresh(self): body = {"status": 403, "message": "Anonymous accounts are not accepted by this API right now."} err = MockAPIError(f"Error code: 403 - {body}", status_code=403, body=body) result = classify_api_error(err, provider="nous", api_key=make_jwt(), model="nous/welcome") assert result.reason == FailoverReason.auth_permanent assert result.retryable is False and result.should_fallback is True assert result.should_rotate_credential is False assert result.error_context["welcome_route"] == "tier_disabled" def test_ordinary_403_is_untouched(self): result = classify_api_error(MockAPIError("forbidden", status_code=403, body={"message": "forbidden"}), provider="nous", api_key=make_jwt()) assert result.reason == FailoverReason.auth assert "welcome_route" not in result.error_context class TestAuthErrorNamesOffRouteEndpoint: """#113719: an auth refusal from a route that is not the provider's own endpoint names the host.""" _BODY = {"error": {"code": "api_key_not_supported", "message": "API keys are not supported by this endpoint."}} def test_stale_base_url_names_contacted_host(self): e = MockAPIError("Unauthorized", status_code=401, body=self._BODY) result = classify_api_error(e, provider="anthropic", model="claude", base_url="https://chatgpt.com/backend-api/codex") assert result.reason == FailoverReason.auth assert result.message == "API keys are not supported by this endpoint. (endpoint: chatgpt.com)" def test_stock_endpoint_and_no_base_url_keep_plain_message(self): e = MockAPIError("Unauthorized", status_code=401, body=self._BODY) for base_url in ("", "https://api.anthropic.com/v1"): result = classify_api_error(e, provider="anthropic", model="claude", base_url=base_url) assert result.message == "API keys are not supported by this endpoint.", base_url class TestBodyCarriedStatus: """An in-stream SSE error object's numeric ``code`` classifies like the equivalent HTTP response (#121270).""" def test_top_level_code_and_status_keys_also_count(self): assert _extract_status_code(MockAPIError("x", body={"code": 503})) == 503 assert _extract_status_code(MockAPIError("x", body={"error": {"http_status": 502}})) == 502 def test_403_ban_is_auth_not_transient_retry(self): body = {"error": {"code": 403, "message": "Your account has been banned by the upstream provider", "metadata": {"provider_name": "acme"}}} result = classify_api_error(MockAPIError("Error code: 403", body=body), provider="custom") assert result.status_code == 403 assert result.reason == FailoverReason.auth assert result.retryable is False assert result.should_fallback is True class TestStreamingRenderFormatError: """Status-less Jinja render failures (LM Studio / llama.cpp) fail over; see #62662.""" def test_error_rendering_no_status_is_format_error(self): e = MockAPIError("Error rendering prompt with jinja template: ...") result = classify_api_error(e, provider="lm-studio", model="x") assert result.reason == FailoverReason.format_error assert result.retryable is False assert result.should_fallback is True def test_render_message_with_status_uses_http_path(self): e = MockAPIError("Error rendering prompt with jinja template: ...", status_code=500) result = classify_api_error(e, provider="lm-studio", model="x") assert result.reason != FailoverReason.format_error