fix(vision): recover from generic image content rejection

This commit is contained in:
yoma
2026-07-04 14:20:23 +08:00
committed by Teknium
parent 628a414d29
commit 98a84783c7
4 changed files with 87 additions and 90 deletions

View File

@@ -59,6 +59,7 @@ from agent.message_sanitization import (
_sanitize_structure_surrogates,
_sanitize_surrogates,
_sanitize_tools_non_ascii,
_looks_like_image_content_rejection,
_strip_images_from_messages,
_strip_non_ascii,
)
@@ -4642,71 +4643,7 @@ def run_conversation(
except Exception:
pass
_err_status = getattr(api_error, "status_code", None)
_IMAGE_REJECTION_PHRASES = (
"only 'text' content type is supported",
"only text content type is supported",
"image_url is not supported",
"image content is not supported",
"multimodal is not supported",
"multimodal content is not supported",
"multimodal input is not supported",
"vision is not supported",
"vision input is not supported",
"does not support images",
"does not support image input",
"does not support multimodal",
"does not support vision",
"model does not support image",
# ChatGPT-account Codex backend
# (https://chatgpt.com/backend-api/codex) rejects
# data:image/...base64 URLs in input_image fields
# with HTTP 400 "Invalid 'input[N].content[K].image_url'.
# Expected a valid URL, but got a value with an
# invalid format." The OpenAI Responses API on the
# public endpoint accepts data URLs, but the
# ChatGPT-account variant does not. Without this
# phrase the agent cascaded into compression /
# context-too-large recovery instead of just
# stripping the images. Match is narrow on
# purpose — keyed on the field-path apostrophe so
# we don't false-trip on other URL validation
# errors. (issue #23570)
"image_url'. expected",
# ChatGPT-account Codex can also reject corrupt/unsupported
# native image payloads with this wording. Treat it like a
# provider image rejection so the loop strips images and
# retries text-only instead of aborting the session.
"image data you provided does not represent a valid image",
# DeepSeek's OpenAI-compatible API reports text-only
# request-body variants as:
# "unknown variant `image_url`, expected `text`".
"unknown variant `image_url`, expected `text`",
"unknown variant image_url, expected text",
# OpenRouter routes a request to upstream endpoints and,
# when none of the candidate endpoints for the model accept
# image input, returns HTTP 404 "No endpoints found that
# support image input". Without this phrase the agent never
# strips the images, the retry loop re-sends the same
# rejected request until exhaustion, and the gateway leaves
# every subsequent message queued behind the stuck turn —
# the P1 in issue #21160. The 404 passes the 4xx gate below.
"no endpoints found that support image input",
# Kimi / Moonshot / other OpenAI-compatible Chinese
# providers reject truncated or corrupt image bytes with
# HTTP 400 "Invalid request: prepare image failed ...
# failed to decode image: invalid or unsupported image
# format". Like the Codex case above, the bad bytes are
# baked into immutable conversation history and re-sent on
# every retry, wedging the session. Strip the images so the
# turn recovers instead of exhausting retries. (issue
# #76884; complements the proactive full-decode validation
# in tools/vision_tools._normalize_to_supported_image)
"failed to decode image",
)
_err_lower = _err_body.lower()
_looks_like_image_rejection = any(
p in _err_lower for p in _IMAGE_REJECTION_PHRASES
)
_looks_like_image_rejection = _looks_like_image_content_rejection(_err_body)
# 4xx-only gate: never interpret 5xx/timeout as "server
# said no to images" — those are transient and must
# route to the normal retry path.

View File

@@ -448,6 +448,81 @@ def _strip_images_from_messages(messages: list) -> bool:
return found
_IMAGE_REJECTION_PHRASES = (
"only 'text' content type is supported",
"only text content type is supported",
"image_url is not supported",
"image content is not supported",
"multimodal is not supported",
"multimodal content is not supported",
"multimodal input is not supported",
"vision is not supported",
"vision input is not supported",
"does not support images",
"does not support image input",
"does not support multimodal",
"does not support vision",
"model does not support image",
# Some OpenAI-compatible endpoints (e.g. Alibaba/DashScope-style
# gateways) reject non-text content blocks with this generic body
# instead of naming image_url or vision support explicitly.
# (issue #57948)
"unexpected item type in content",
# ChatGPT-account Codex backend
# (https://chatgpt.com/backend-api/codex) rejects
# data:image/...base64 URLs in input_image fields
# with HTTP 400 "Invalid 'input[N].content[K].image_url'.
# Expected a valid URL, but got a value with an
# invalid format." The OpenAI Responses API on the
# public endpoint accepts data URLs, but the
# ChatGPT-account variant does not. Without this
# phrase the agent cascaded into compression /
# context-too-large recovery instead of just
# stripping the images. Match is narrow on
# purpose — keyed on the field-path apostrophe so
# we don't false-trip on other URL validation
# errors. (issue #23570)
"image_url'. expected",
# ChatGPT-account Codex can also reject corrupt/unsupported
# native image payloads with this wording. Treat it like a
# provider image rejection so the loop strips images and
# retries text-only instead of aborting the session.
"image data you provided does not represent a valid image",
# DeepSeek's OpenAI-compatible API reports text-only
# request-body variants as:
# "unknown variant `image_url`, expected `text`".
"unknown variant `image_url`, expected `text`",
"unknown variant image_url, expected text",
# OpenRouter routes a request to upstream endpoints and,
# when none of the candidate endpoints for the model accept
# image input, returns HTTP 404 "No endpoints found that
# support image input". Without this phrase the agent never
# strips the images, the retry loop re-sends the same
# rejected request until exhaustion, and the gateway leaves
# every subsequent message queued behind the stuck turn —
# the P1 in issue #21160. The 404 passes the 4xx gate in the
# conversation loop.
"no endpoints found that support image input",
# Kimi / Moonshot / other OpenAI-compatible Chinese
# providers reject truncated or corrupt image bytes with
# HTTP 400 "Invalid request: prepare image failed ...
# failed to decode image: invalid or unsupported image
# format". Like the Codex case above, the bad bytes are
# baked into immutable conversation history and re-sent on
# every retry, wedging the session. Strip the images so the
# turn recovers instead of exhausting retries. (issue
# #76884; complements the proactive full-decode validation
# in tools/vision_tools._normalize_to_supported_image)
"failed to decode image",
)
def _looks_like_image_content_rejection(error_body: str) -> bool:
"""Return True when a provider error says image/multimodal input is unsupported."""
body = str(error_body or "").lower()
return any(phrase in body for phrase in _IMAGE_REJECTION_PHRASES)
def _sanitize_structure_non_ascii(payload: Any) -> bool:
"""Strip non-ASCII characters from nested dict/list payloads in-place."""
found = False

View File

@@ -183,6 +183,7 @@ from agent.message_sanitization import ( # noqa: F401
_strip_non_ascii,
_sanitize_messages_non_ascii,
_sanitize_tools_non_ascii,
_looks_like_image_content_rejection,
_strip_images_from_messages,
_sanitize_structure_non_ascii,
coalesce_tool_call_id as _sanitize_coalesce_tool_call_id,

View File

@@ -6,7 +6,7 @@ verify that stripping preserves the role-alternation invariants providers
require, and that the phrase detector fires on the expected error bodies.
"""
from run_agent import _strip_images_from_messages
from run_agent import _looks_like_image_content_rejection, _strip_images_from_messages
class TestStripImagesPreservesAlternation:
@@ -121,31 +121,8 @@ class TestImageRejectionPhraseIsolation:
so they route to the correct recovery handler (e.g. _try_shrink_image_parts).
"""
# Reproduces the phrase list used in run_agent.py's error-handler block.
_REJECTION_PHRASES = (
"only 'text' content type is supported",
"only text content type is supported",
"image_url is not supported",
"image content is not supported",
"multimodal is not supported",
"multimodal content is not supported",
"multimodal input is not supported",
"vision is not supported",
"vision input is not supported",
"does not support images",
"does not support image input",
"does not support multimodal",
"does not support vision",
"model does not support image",
"image_url'. expected",
"no endpoints found that support image input",
"failed to decode image",
"image data you provided does not represent a valid image",
)
def _matches(self, body: str) -> bool:
low = body.lower()
return any(p in low for p in self._REJECTION_PHRASES)
return _looks_like_image_content_rejection(body)
def test_kimi_truncated_image_trips_recovery(self):
# Kimi/Moonshot reject truncated image bytes with this 400; the
@@ -183,6 +160,13 @@ class TestImageRejectionPhraseIsolation:
# match the agent cascaded into compression / context-too-large
# recovery instead of just stripping the images.
"Invalid 'input[56].content[1].image_url'. Expected a valid URL, but got a value with an invalid format.",
# OpenRouter 404 when no upstream endpoint for the model accepts
# image input — issue #21160. The exact wording from the report.
"HTTP 404: No endpoints found that support image input",
# Alibaba/OpenAI-compatible endpoints can reject image-bearing
# messages without naming image_url explicitly. The first failed
# turn should still switch to text-only/aux-vision mode (#57948).
"The provided messages input is invalid. The error info is [Unexpected item type in content].",
"The image data you provided does not represent a valid image. Please check your input and try again.",
]
for body in bodies: