fix: read a top-level detail error body so a descriptive 400 is not a bare 400

`agent/error_classifier.py::_body_message_candidates` never yielded the
FastAPI-style top-level `detail` key (string, or nested `{"message": ...}`),
so the Codex gateway's `{"detail": "The '<model>' model is not supported when
using Codex with a ChatGPT account."}` 400 read as a *bare* 400. On a large
session `_classify_400`'s generic-400 heuristic then classified it as
context_overflow: the loop burned compression attempts ("Context length
exceeded (109,962 tokens). Cannot compress further.") and, because
`is_client_error` excludes overflow, the entitlement marker from #106549 never
ran. Reading `detail` makes it a descriptive rejection (format_error: abort +
fall back, no compression) and surfaces the provider's text as the message
instead of `Error code: 400 - {...}`. The pydantic list shape of `detail` is
still handled by `_oversized_message_content_rejection` and is not yielded.

Slim re-port of #100783's detail-body half onto the rule-table classifier;
the session/weekly usage-limit half is a separate class and was dropped.

Refs #81558
Refs #106475
Co-authored-by: Oleg Nagornyy <nagornyy.o@gmail.com>
This commit is contained in:
teknium1
2026-09-19 00:44:49 -07:00
committed by Teknium
parent 3934fe551c
commit 05c95d216e
3 changed files with 30 additions and 1 deletions

View File

@@ -1257,12 +1257,18 @@ def _build_error_msg(error: Exception, body: Any) -> str:
def _body_message_candidates(body: dict) -> Iterator[Any]:
"""Body message fields in priority order (OpenAI, flat, litellm/Bedrock proxy shapes)."""
"""Body message fields in priority order (OpenAI, flat, litellm/Bedrock proxy, FastAPI shapes)."""
yield _error_obj(body).get("message")
yield body.get("message")
yield body.get("errorMessage")
args = body.get("errorArgs")
yield args.get("reason") if isinstance(args, dict) else None
# FastAPI/Starlette relays and the Codex gateway answer {"detail": "..."} (or a nested
# OpenAI-ish object); without it a descriptive rejection reads as a bare 400 and the
# large-session heuristic sends it into compression (#81558). A list here is pydantic's
# validation shape, read by _oversized_message_content_rejection.
detail = body.get("detail")
yield detail.get("message") if isinstance(detail, dict) else detail if isinstance(detail, str) else None
def _from_cause_chain(error: Exception, pick: Callable[[Any], Any], default: Any) -> Any:

View File

@@ -0,0 +1 @@
i-Hun

View File

@@ -1181,6 +1181,28 @@ class TestClassifyApiError:
for r in caplog.records
), "Expected a distinct warning identifying the malformed-body 400"
def test_400_top_level_detail_body_is_not_a_bare_400_on_large_session(self):
"""FastAPI-style ``{"detail": "..."}`` bodies (Codex gateway, Starlette relays) →
the descriptive text is read, so the large-session heuristic does not route a
model entitlement/retirement rejection into compression (#81558, #106475).
``str(error)`` is the SDK's ``Error code: 400 - {...}`` form, exactly as on the wire.
Salvaged from #100783 (@i-Hun)."""
detail = "The 'gpt-5.5-codex' model is not supported when using Codex with a ChatGPT account."
large = dict(provider="openai-codex", model="gpt-5.5-codex",
approx_tokens=109_962, context_length=272_000, num_messages=223)
for body in ({"detail": detail}, {"detail": {"message": detail}}):
e = MockAPIError(f"Error code: 400 - {body!r}", status_code=400, body=body)
result = classify_api_error(e, **large) # type: ignore[arg-type]
assert result.reason is not FailoverReason.context_overflow, body
assert result.should_compress is False
assert result.should_fallback is True
assert result.message == detail
# Control: the genuinely bare body the heuristic exists for still compresses.
bare = classify_api_error(
MockAPIError("Error code: 400 - {'error': {'message': 'Error'}}", status_code=400,
body={"error": {"message": "Error"}}), **large) # type: ignore[arg-type]
assert bare.reason is FailoverReason.context_overflow
# ── Peer closed + large session ──