fix(agent): classify local-inference memory-ceiling rejections as overloaded
oMLX/MLX prefill memory-guard rejections name an allocation peak in BYTES but
close with "Reduce context length", so _CONTEXT_OVERFLOW_PATTERNS claims them
and the turn enters the compress-and-shrink loop. Compression cannot lower a
prefill peak — the prompt is usually far below the window — so it burns the
compression budget, re-hits the wedged server on every attempt and ends in
"Cannot compress further" plus a destructive session reset.
Classify them as `overloaded` instead: retry with backoff, no compression, no
session reset (mirrors 503/529 recovery).
The guard runs before the overflow check AND before the usage-limit
disambiguation. The second ordering matters more than it looks: "memory limit
exceeded" contains "limit exceeded", so a status-less rejection — a proxy that
flattened the body — is currently classified as `billing` and rotates a
healthy credential.
Sites covered:
- _OVERFLOW_AS_5XX_RULES, which _400_TAIL_RULES extends → 400, 500, 502,
503, 529
- _MESSAGE_HEAD_RULES for the status-less path (ahead of usage-limit)
- _ERROR_CODE_VERDICTS for the structured oMLX codes
- _classify_400, because _by_status runs before _by_error_code, so a body
whose wording a proxy stripped would otherwise fall through to
format_error
Every pattern names memory/allocation in bytes, never a token or window count,
so the list stays disjoint from _CONTEXT_OVERFLOW_PATTERNS. Both oMLX wordings
are kept: 0.5.6 says "Prefill would require ~13.87 GB peak", 0.5.7 reworded it
to "predicted peak would require/exceed" and both are in the field. A test
pins that a genuine window overflow still compresses.
Refs #52261. Supersedes #52289, which predates the classifier rewrite and can
no longer be merged.
This commit is contained in:
@@ -172,6 +172,30 @@ _MULTIMODAL_TOOL_CONTENT_PATTERNS = (
|
||||
"tool_call.content must be string",
|
||||
)
|
||||
|
||||
# Local-inference memory/resource-ceiling rejections (oMLX/MLX memory guard,
|
||||
# llama.cpp/vLLM OOM, Metal/CUDA allocation ceilings). The server aborts on a
|
||||
# prefill memory PEAK, not a window limit, yet its remediation tail says
|
||||
# "reduce context length" — so without this list the request routes into
|
||||
# compression, which cannot lower a prefill peak: it burns the compression
|
||||
# budget, re-hits the wedged server each attempt and ends in a session reset.
|
||||
# Every token names memory/allocation in BYTES, never a token count, so the
|
||||
# list is disjoint from _CONTEXT_OVERFLOW_PATTERNS. Must be checked BEFORE
|
||||
# both overflow AND the usage-limit disambiguation ("memory limit exceeded"
|
||||
# contains "limit exceeded", which would otherwise read as billing). oMLX
|
||||
# reworded the accounting sentence in 0.5.7 ("predicted peak would require /
|
||||
# exceed"); the 0.5.6 wording is still in the field, so both stay. (#52261)
|
||||
_MEMORY_CEILING_PATTERNS = (
|
||||
"memory guard", "memory limit exceeded", "memory_guard_tier", "dynamic ceiling",
|
||||
"memory ceiling", "available memory", "out of memory", "insufficient memory",
|
||||
"prefill would require", "predicted peak would", "prefill safety cap", "metal_cap",
|
||||
)
|
||||
|
||||
# Structured codes identifying the same rejection at the source, before an
|
||||
# OpenAI-compatible proxy flattens the body and drops the wording.
|
||||
_MEMORY_CEILING_ERROR_CODES = frozenset({
|
||||
"prefill_memory_exceeded", "prefill_memory_aborted", "omlx_prefill_memory_exceeded",
|
||||
})
|
||||
|
||||
# Bare "max_tokens" is load-bearing: the output-cap-retry path keys off it;
|
||||
# empty-response advisories mentioning it are intercepted earlier. Groups:
|
||||
# generic; vLLM; Ollama; llama.cpp; Chinese; Z.AI (1210); Bedrock; Together.
|
||||
@@ -380,7 +404,8 @@ _IMAGE_TOOL_RULES = (
|
||||
# Overflow signals arriving as 5xx (llama.cpp reports overflow as 500; busy /
|
||||
# model-load OOM as 503). Empty-response advisories must not enter compression.
|
||||
_OVERFLOW_AS_5XX_RULES = (
|
||||
(_EMPTY_PROVIDER_RESPONSE_PATTERNS, _V_SERVER_ERROR), (_CONTEXT_OVERFLOW_PATTERNS, _V_CONTEXT_OVERFLOW),
|
||||
(_EMPTY_PROVIDER_RESPONSE_PATTERNS, _V_SERVER_ERROR), (_MEMORY_CEILING_PATTERNS, _V_OVERLOADED),
|
||||
(_CONTEXT_OVERFLOW_PATTERNS, _V_CONTEXT_OVERFLOW),
|
||||
)
|
||||
|
||||
# 404: Nous API surfaces credit depletion as a paid model vanishing from the
|
||||
@@ -398,7 +423,8 @@ _400_TAIL_RULES = _OVERFLOW_AS_5XX_RULES + (
|
||||
)
|
||||
|
||||
# Status-less message path, head (before usage-limit disambiguation).
|
||||
_MESSAGE_HEAD_RULES = ((_PAYLOAD_TOO_LARGE_PATTERNS, _V_PAYLOAD_TOO_LARGE),) + _IMAGE_TOOL_RULES
|
||||
_MESSAGE_HEAD_RULES = ((_MEMORY_CEILING_PATTERNS, _V_OVERLOADED),
|
||||
(_PAYLOAD_TOO_LARGE_PATTERNS, _V_PAYLOAD_TOO_LARGE)) + _IMAGE_TOOL_RULES
|
||||
|
||||
# Status-less tail. Overload before rate_limit/billing so "overloaded" backs off
|
||||
# instead of rotating; policy block before model_not_found; timeout/connection
|
||||
@@ -419,6 +445,7 @@ _ERROR_CODE_VERDICTS: Dict[str, Verdict] = {
|
||||
**dict.fromkeys(_BILLING_ERROR_CODES, _V_BILLING),
|
||||
**dict.fromkeys(("model_not_found", "model_not_available", "invalid_model"), _V_MODEL_NOT_FOUND),
|
||||
**dict.fromkeys(("context_length_exceeded", "max_tokens_exceeded"), _V_CONTEXT_OVERFLOW),
|
||||
**dict.fromkeys(_MEMORY_CEILING_ERROR_CODES, _V_OVERLOADED),
|
||||
"invalid_encrypted_content": _V_INVALID_ENCRYPTED,
|
||||
}
|
||||
|
||||
@@ -713,6 +740,10 @@ def _classify_400(c: _Ctx) -> Verdict:
|
||||
"error=%.200s", c.num_messages, c.approx_tokens, msg,
|
||||
)
|
||||
return _V_FORMAT_ERROR
|
||||
# Memory ceiling by code: _by_status runs before _by_error_code, so a
|
||||
# 400 whose wording a proxy stripped would fall through to format_error.
|
||||
if code in _MEMORY_CEILING_ERROR_CODES:
|
||||
return _V_OVERLOADED
|
||||
verdict = _first_match(msg, _400_TAIL_RULES)
|
||||
if verdict is not None:
|
||||
return verdict
|
||||
|
||||
@@ -681,6 +681,77 @@ class TestClassifyApiError:
|
||||
|
||||
|
||||
|
||||
# ── Local-inference memory ceiling (oMLX/MLX prefill guard, #52261) ──
|
||||
|
||||
def test_omlx_prefill_guard_is_overloaded_not_overflow(self):
|
||||
"""oMLX 0.5.6 prefill guard → overloaded, never compression. The body
|
||||
names a memory peak in BYTES, but its remediation tail says "Reduce
|
||||
context length" — which used to route it into the compress loop, where
|
||||
shrinking history cannot lower a prefill peak."""
|
||||
e = MockAPIError(
|
||||
"Prefill memory guard rejected request: Prefill would require ~13.87 GB peak, "
|
||||
"dynamic ceiling is 13.50 GB. Reduce context length or lower memory_guard_tier.",
|
||||
status_code=400,
|
||||
)
|
||||
result = classify_api_error(e, provider="omlx")
|
||||
assert result.reason == FailoverReason.overloaded
|
||||
assert result.should_compress is False
|
||||
assert result.retryable is True
|
||||
|
||||
def test_omlx_057_reworded_peak_accounting(self):
|
||||
"""0.5.7 rewrote the sentence to "predicted peak would require"; the
|
||||
cap names (prefill safety cap, metal_cap) survive the verb change."""
|
||||
e = MockAPIError(
|
||||
"process memory limit exceeded: predicted peak would require ~78.57 GB, prefill "
|
||||
"safety cap is 77.76 GB (90% of metal_cap ceiling 86.40 GB). Reduce context size.",
|
||||
status_code=400,
|
||||
)
|
||||
assert classify_api_error(e, provider="omlx").reason == FailoverReason.overloaded
|
||||
|
||||
def test_omlx_memory_ceiling_mid_stream_500(self):
|
||||
"""Mid-stream the guard exits as a generic 500 (the streaming generator
|
||||
catches bare Exception and drops the structured code)."""
|
||||
e = MockAPIError(
|
||||
"predicted peak would exceed prefill safety cap 77.8GB. Reduce context length.",
|
||||
status_code=500,
|
||||
)
|
||||
assert classify_api_error(e, provider="omlx").reason == FailoverReason.overloaded
|
||||
|
||||
def test_omlx_memory_ceiling_without_status_is_not_billing(self):
|
||||
""""memory limit exceeded" contains "limit exceeded", so a flattened,
|
||||
status-less rejection would hit the usage-limit disambiguation and read
|
||||
as billing — rotating a healthy credential. The memory rules run in the
|
||||
message HEAD, before that disambiguation."""
|
||||
e = MockAPIError(
|
||||
"process memory limit exceeded: predicted peak would require ~78.57 GB. "
|
||||
"Reduce context size."
|
||||
)
|
||||
result = classify_api_error(e, provider="omlx")
|
||||
assert result.reason == FailoverReason.overloaded
|
||||
assert result.should_rotate_credential is False
|
||||
|
||||
def test_omlx_memory_ceiling_by_error_code_alone(self):
|
||||
"""A proxy that flattens the body and drops the wording still leaves
|
||||
the structured code; _by_status runs before _by_error_code, so the 400
|
||||
handler has to read it or the request falls through to format_error."""
|
||||
e = MockAPIError(
|
||||
"Request failed.", status_code=400,
|
||||
body={"error": {"message": "Request failed.", "code": "prefill_memory_exceeded"}},
|
||||
)
|
||||
assert classify_api_error(e, provider="omlx").reason == FailoverReason.overloaded
|
||||
|
||||
def test_genuine_context_overflow_still_compresses(self):
|
||||
"""Guard against over-reach: a real window overflow must keep its
|
||||
compression recovery."""
|
||||
e = MockAPIError(
|
||||
"This model's maximum context length is 200000 tokens. However, your "
|
||||
"messages resulted in 250000 tokens.",
|
||||
status_code=400,
|
||||
)
|
||||
result = classify_api_error(e, provider="omlx")
|
||||
assert result.reason == FailoverReason.context_overflow
|
||||
assert result.should_compress is True
|
||||
|
||||
# ── Server disconnect + large session ──
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user