fix(model-metadata): parse 'exceeds model maximum output tokens' cap errors

Recognizes the DeepSeek/OpenAI-compatible relay wording
  max_tokens (98304) exceeds model's maximum output tokens (65536)
in both parse_available_output_tokens_from_error (returns the cap) and
is_output_cap_error (keeps the 400 out of the compression death-loop).

Salvaged from PR #72283; the conversation_loop early-clamp block was
dropped in favor of routing through the existing output-cap handler
(follow-up commit).
This commit is contained in:
ekinnee
2026-08-20 11:09:48 +05:30
committed by kshitij
parent 7f2733b71c
commit 99c980f466
2 changed files with 36 additions and 0 deletions

View File

@@ -1659,10 +1659,28 @@ def parse_available_output_tokens_from_error(error_msg: str) -> Optional[int]:
# The input itself fits — this is purely an output-cap error, so reduce
# max_tokens and retry; do NOT compress.
"range of max_tokens should be" in error_lower
) or (
# OpenAI-compatible relays may reject a request whose output cap exceeds
# the model's separate completion-token limit, e.g.
# "max_tokens (98304) exceeds model's maximum output tokens (65536)"
# This is independent of the input context window.
"exceeds model" in error_lower
and "maximum output tokens" in error_lower
)
if not is_output_cap_error:
return None
# Generic model-output-cap form:
# "max_tokens (98304) exceeds model's maximum output tokens (65536)"
_m_max_output = re.search(
r'exceeds model(?:\'s)? maximum output tokens\s*\(?\s*(\d+)\s*\)?',
error_lower,
)
if _m_max_output:
_cap = int(_m_max_output.group(1))
if _cap >= 1:
return _cap
# DashScope / Alibaba range form: "Range of max_tokens should be [1, 65536]".
# The upper bound is the available output cap.
_m_range = re.search(
@@ -1799,6 +1817,8 @@ def is_output_cap_error(error_msg: str) -> bool:
or "should be" in error_lower # generic "max_tokens should be <= N"
or "less than or equal" in error_lower
or "must be" in error_lower
or ("exceeds model" in error_lower
and "maximum output tokens" in error_lower)
)
if not output_cap_signal:
return False

View File

@@ -68,6 +68,22 @@ class TestParseDashScopeOutputCap:
assert parse_available_output_tokens_from_error(msg) == 32768
class TestParseMaximumOutputTokensCap:
"""Some OpenAI-compatible relays report the model's separate output cap."""
def test_parenthesized_max_output_cap(self):
msg = (
"API call failed after 3 retries: [400]: max_tokens (98304) "
"exceeds model's maximum output tokens (65536)"
)
assert parse_available_output_tokens_from_error(msg) == 65536
def test_parenthesized_max_output_cap_is_output_cap(self):
assert is_output_cap_error(
"max_tokens (98304) exceeds model's maximum output tokens (65536)"
) is True
class TestIsOutputCapError:
"""`is_output_cap_error` is the broader yes/no gate that keeps an
output-cap 400 out of the compression death-loop even when we can't parse