fix(agent): classify local-inference memory-ceiling rejections as overloaded

oMLX/MLX prefill memory-guard rejections name an allocation peak in BYTES but
close with "Reduce context length", so _CONTEXT_OVERFLOW_PATTERNS claims them
and the turn enters the compress-and-shrink loop. Compression cannot lower a
prefill peak — the prompt is usually far below the window — so it burns the
compression budget, re-hits the wedged server on every attempt and ends in
"Cannot compress further" plus a destructive session reset.

Classify them as `overloaded` instead: retry with backoff, no compression, no
session reset (mirrors 503/529 recovery).

The guard runs before the overflow check AND before the usage-limit
disambiguation. The second ordering matters more than it looks: "memory limit
exceeded" contains "limit exceeded", so a status-less rejection — a proxy that
flattened the body — is currently classified as `billing` and rotates a
healthy credential.

Sites covered:
  - _OVERFLOW_AS_5XX_RULES, which _400_TAIL_RULES extends → 400, 500, 502,
    503, 529
  - _MESSAGE_HEAD_RULES for the status-less path (ahead of usage-limit)
  - _ERROR_CODE_VERDICTS for the structured oMLX codes
  - _classify_400, because _by_status runs before _by_error_code, so a body
    whose wording a proxy stripped would otherwise fall through to
    format_error

Every pattern names memory/allocation in bytes, never a token or window count,
so the list stays disjoint from _CONTEXT_OVERFLOW_PATTERNS. Both oMLX wordings
are kept: 0.5.6 says "Prefill would require ~13.87 GB peak", 0.5.7 reworded it
to "predicted peak would require/exceed" and both are in the field. A test
pins that a genuine window overflow still compresses.

Refs #52261. Supersedes #52289, which predates the classifier rewrite and can
no longer be merged.
This commit is contained in:
Tim Kaufmann
2026-09-08 02:27:13 +02:00
committed by kshitij
parent 2b0b03ec58
commit b7e4712029
2 changed files with 104 additions and 2 deletions

View File

@@ -172,6 +172,30 @@ _MULTIMODAL_TOOL_CONTENT_PATTERNS = (
"tool_call.content must be string",
)
# Local-inference memory/resource-ceiling rejections (oMLX/MLX memory guard,
# llama.cpp/vLLM OOM, Metal/CUDA allocation ceilings). The server aborts on a
# prefill memory PEAK, not a window limit, yet its remediation tail says
# "reduce context length" — so without this list the request routes into
# compression, which cannot lower a prefill peak: it burns the compression
# budget, re-hits the wedged server each attempt and ends in a session reset.
# Every token names memory/allocation in BYTES, never a token count, so the
# list is disjoint from _CONTEXT_OVERFLOW_PATTERNS. Must be checked BEFORE
# both overflow AND the usage-limit disambiguation ("memory limit exceeded"
# contains "limit exceeded", which would otherwise read as billing). oMLX
# reworded the accounting sentence in 0.5.7 ("predicted peak would require /
# exceed"); the 0.5.6 wording is still in the field, so both stay. (#52261)
_MEMORY_CEILING_PATTERNS = (
"memory guard", "memory limit exceeded", "memory_guard_tier", "dynamic ceiling",
"memory ceiling", "available memory", "out of memory", "insufficient memory",
"prefill would require", "predicted peak would", "prefill safety cap", "metal_cap",
)
# Structured codes identifying the same rejection at the source, before an
# OpenAI-compatible proxy flattens the body and drops the wording.
_MEMORY_CEILING_ERROR_CODES = frozenset({
"prefill_memory_exceeded", "prefill_memory_aborted", "omlx_prefill_memory_exceeded",
})
# Bare "max_tokens" is load-bearing: the output-cap-retry path keys off it;
# empty-response advisories mentioning it are intercepted earlier. Groups:
# generic; vLLM; Ollama; llama.cpp; Chinese; Z.AI (1210); Bedrock; Together.
@@ -380,7 +404,8 @@ _IMAGE_TOOL_RULES = (
# Overflow signals arriving as 5xx (llama.cpp reports overflow as 500; busy /
# model-load OOM as 503). Empty-response advisories must not enter compression.
_OVERFLOW_AS_5XX_RULES = (
(_EMPTY_PROVIDER_RESPONSE_PATTERNS, _V_SERVER_ERROR), (_CONTEXT_OVERFLOW_PATTERNS, _V_CONTEXT_OVERFLOW),
(_EMPTY_PROVIDER_RESPONSE_PATTERNS, _V_SERVER_ERROR), (_MEMORY_CEILING_PATTERNS, _V_OVERLOADED),
(_CONTEXT_OVERFLOW_PATTERNS, _V_CONTEXT_OVERFLOW),
)
# 404: Nous API surfaces credit depletion as a paid model vanishing from the
@@ -398,7 +423,8 @@ _400_TAIL_RULES = _OVERFLOW_AS_5XX_RULES + (
)
# Status-less message path, head (before usage-limit disambiguation).
_MESSAGE_HEAD_RULES = ((_PAYLOAD_TOO_LARGE_PATTERNS, _V_PAYLOAD_TOO_LARGE),) + _IMAGE_TOOL_RULES
_MESSAGE_HEAD_RULES = ((_MEMORY_CEILING_PATTERNS, _V_OVERLOADED),
(_PAYLOAD_TOO_LARGE_PATTERNS, _V_PAYLOAD_TOO_LARGE)) + _IMAGE_TOOL_RULES
# Status-less tail. Overload before rate_limit/billing so "overloaded" backs off
# instead of rotating; policy block before model_not_found; timeout/connection
@@ -419,6 +445,7 @@ _ERROR_CODE_VERDICTS: Dict[str, Verdict] = {
**dict.fromkeys(_BILLING_ERROR_CODES, _V_BILLING),
**dict.fromkeys(("model_not_found", "model_not_available", "invalid_model"), _V_MODEL_NOT_FOUND),
**dict.fromkeys(("context_length_exceeded", "max_tokens_exceeded"), _V_CONTEXT_OVERFLOW),
**dict.fromkeys(_MEMORY_CEILING_ERROR_CODES, _V_OVERLOADED),
"invalid_encrypted_content": _V_INVALID_ENCRYPTED,
}
@@ -713,6 +740,10 @@ def _classify_400(c: _Ctx) -> Verdict:
"error=%.200s", c.num_messages, c.approx_tokens, msg,
)
return _V_FORMAT_ERROR
# Memory ceiling by code: _by_status runs before _by_error_code, so a
# 400 whose wording a proxy stripped would fall through to format_error.
if code in _MEMORY_CEILING_ERROR_CODES:
return _V_OVERLOADED
verdict = _first_match(msg, _400_TAIL_RULES)
if verdict is not None:
return verdict

View File

@@ -681,6 +681,77 @@ class TestClassifyApiError:
# ── Local-inference memory ceiling (oMLX/MLX prefill guard, #52261) ──
def test_omlx_prefill_guard_is_overloaded_not_overflow(self):
"""oMLX 0.5.6 prefill guard → overloaded, never compression. The body
names a memory peak in BYTES, but its remediation tail says "Reduce
context length" — which used to route it into the compress loop, where
shrinking history cannot lower a prefill peak."""
e = MockAPIError(
"Prefill memory guard rejected request: Prefill would require ~13.87 GB peak, "
"dynamic ceiling is 13.50 GB. Reduce context length or lower memory_guard_tier.",
status_code=400,
)
result = classify_api_error(e, provider="omlx")
assert result.reason == FailoverReason.overloaded
assert result.should_compress is False
assert result.retryable is True
def test_omlx_057_reworded_peak_accounting(self):
"""0.5.7 rewrote the sentence to "predicted peak would require"; the
cap names (prefill safety cap, metal_cap) survive the verb change."""
e = MockAPIError(
"process memory limit exceeded: predicted peak would require ~78.57 GB, prefill "
"safety cap is 77.76 GB (90% of metal_cap ceiling 86.40 GB). Reduce context size.",
status_code=400,
)
assert classify_api_error(e, provider="omlx").reason == FailoverReason.overloaded
def test_omlx_memory_ceiling_mid_stream_500(self):
"""Mid-stream the guard exits as a generic 500 (the streaming generator
catches bare Exception and drops the structured code)."""
e = MockAPIError(
"predicted peak would exceed prefill safety cap 77.8GB. Reduce context length.",
status_code=500,
)
assert classify_api_error(e, provider="omlx").reason == FailoverReason.overloaded
def test_omlx_memory_ceiling_without_status_is_not_billing(self):
""""memory limit exceeded" contains "limit exceeded", so a flattened,
status-less rejection would hit the usage-limit disambiguation and read
as billing — rotating a healthy credential. The memory rules run in the
message HEAD, before that disambiguation."""
e = MockAPIError(
"process memory limit exceeded: predicted peak would require ~78.57 GB. "
"Reduce context size."
)
result = classify_api_error(e, provider="omlx")
assert result.reason == FailoverReason.overloaded
assert result.should_rotate_credential is False
def test_omlx_memory_ceiling_by_error_code_alone(self):
"""A proxy that flattens the body and drops the wording still leaves
the structured code; _by_status runs before _by_error_code, so the 400
handler has to read it or the request falls through to format_error."""
e = MockAPIError(
"Request failed.", status_code=400,
body={"error": {"message": "Request failed.", "code": "prefill_memory_exceeded"}},
)
assert classify_api_error(e, provider="omlx").reason == FailoverReason.overloaded
def test_genuine_context_overflow_still_compresses(self):
"""Guard against over-reach: a real window overflow must keep its
compression recovery."""
e = MockAPIError(
"This model's maximum context length is 200000 tokens. However, your "
"messages resulted in 250000 tokens.",
status_code=400,
)
result = classify_api_error(e, provider="omlx")
assert result.reason == FailoverReason.context_overflow
assert result.should_compress is True
# ── Server disconnect + large session ──