fix(model-metadata): parse 'exceeds model maximum output tokens' cap errors
Recognizes the DeepSeek/OpenAI-compatible relay wording max_tokens (98304) exceeds model's maximum output tokens (65536) in both parse_available_output_tokens_from_error (returns the cap) and is_output_cap_error (keeps the 400 out of the compression death-loop). Salvaged from PR #72283; the conversation_loop early-clamp block was dropped in favor of routing through the existing output-cap handler (follow-up commit).
This commit is contained in:
@@ -1659,10 +1659,28 @@ def parse_available_output_tokens_from_error(error_msg: str) -> Optional[int]:
|
||||
# The input itself fits — this is purely an output-cap error, so reduce
|
||||
# max_tokens and retry; do NOT compress.
|
||||
"range of max_tokens should be" in error_lower
|
||||
) or (
|
||||
# OpenAI-compatible relays may reject a request whose output cap exceeds
|
||||
# the model's separate completion-token limit, e.g.
|
||||
# "max_tokens (98304) exceeds model's maximum output tokens (65536)"
|
||||
# This is independent of the input context window.
|
||||
"exceeds model" in error_lower
|
||||
and "maximum output tokens" in error_lower
|
||||
)
|
||||
if not is_output_cap_error:
|
||||
return None
|
||||
|
||||
# Generic model-output-cap form:
|
||||
# "max_tokens (98304) exceeds model's maximum output tokens (65536)"
|
||||
_m_max_output = re.search(
|
||||
r'exceeds model(?:\'s)? maximum output tokens\s*\(?\s*(\d+)\s*\)?',
|
||||
error_lower,
|
||||
)
|
||||
if _m_max_output:
|
||||
_cap = int(_m_max_output.group(1))
|
||||
if _cap >= 1:
|
||||
return _cap
|
||||
|
||||
# DashScope / Alibaba range form: "Range of max_tokens should be [1, 65536]".
|
||||
# The upper bound is the available output cap.
|
||||
_m_range = re.search(
|
||||
@@ -1799,6 +1817,8 @@ def is_output_cap_error(error_msg: str) -> bool:
|
||||
or "should be" in error_lower # generic "max_tokens should be <= N"
|
||||
or "less than or equal" in error_lower
|
||||
or "must be" in error_lower
|
||||
or ("exceeds model" in error_lower
|
||||
and "maximum output tokens" in error_lower)
|
||||
)
|
||||
if not output_cap_signal:
|
||||
return False
|
||||
|
||||
@@ -68,6 +68,22 @@ class TestParseDashScopeOutputCap:
|
||||
assert parse_available_output_tokens_from_error(msg) == 32768
|
||||
|
||||
|
||||
class TestParseMaximumOutputTokensCap:
|
||||
"""Some OpenAI-compatible relays report the model's separate output cap."""
|
||||
|
||||
def test_parenthesized_max_output_cap(self):
|
||||
msg = (
|
||||
"API call failed after 3 retries: [400]: max_tokens (98304) "
|
||||
"exceeds model's maximum output tokens (65536)"
|
||||
)
|
||||
assert parse_available_output_tokens_from_error(msg) == 65536
|
||||
|
||||
def test_parenthesized_max_output_cap_is_output_cap(self):
|
||||
assert is_output_cap_error(
|
||||
"max_tokens (98304) exceeds model's maximum output tokens (65536)"
|
||||
) is True
|
||||
|
||||
|
||||
class TestIsOutputCapError:
|
||||
"""`is_output_cap_error` is the broader yes/no gate that keeps an
|
||||
output-cap 400 out of the compression death-loop even when we can't parse
|
||||
|
||||
Reference in New Issue
Block a user