diff --git a/tests/tools/test_vision_native_fast_path.py b/tests/tools/test_vision_native_fast_path.py index 5237ae665f..82b7cb57ce 100644 --- a/tests/tools/test_vision_native_fast_path.py +++ b/tests/tools/test_vision_native_fast_path.py @@ -90,6 +90,30 @@ class TestSupportsMediaInToolResults: assert _supports_media_in_tool_results("", "anything") is False assert _supports_media_in_tool_results(None, "anything") is False # type: ignore[arg-type] + def test_profile_tool_message_veto_overrides_supports_vision(self): + """supports_vision_tool_messages=False is a hard veto even when the + profile declares supports_vision=True (xiaomi/MiMo 400s on list-type + tool-result content, #89981).""" + assert _supports_media_in_tool_results("xiaomi", "mimo-v2.5") is False + + def test_profile_veto_applies_even_when_vision_capable_lookup_agrees(self): + """A capability source marking the model vision-capable must not + re-open the native fast path for a provider that rejects it.""" + from tools.vision_tools import _should_use_native_vision_fast_path + from agent.auxiliary_client import set_runtime_main, clear_runtime_main + from agent import image_routing + + set_runtime_main("xiaomi", "mimo-v2.5") + try: + with patch.object( + image_routing, "decide_image_input_mode", return_value="native" + ), patch.object( + image_routing, "_lookup_supports_vision", return_value=True + ): + assert _should_use_native_vision_fast_path() is False + finally: + clear_runtime_main() + # ─── _build_native_vision_tool_result ──────────────────────────────────────── diff --git a/tools/vision_tools.py b/tools/vision_tools.py index f6c1852182..81a7865264 100644 --- a/tools/vision_tools.py +++ b/tools/vision_tools.py @@ -1077,6 +1077,22 @@ def _resize_image_for_vision(image_path: Path, mime_type: Optional[str] = None, # --------------------------------------------------------------------------- +def _profile_rejects_tool_media(provider: str) -> bool: + """Hard veto: the provider's ``ProviderProfile`` declares + ``supports_vision_tool_messages=False`` — images are accepted in user + messages but list-type tool-result content is rejected with 400 + (xiaomi/MiMo "text is not set"). ``supports_vision`` alone must not + override this, or the multimodal tool-result envelope 400s every turn + and the image never enters context (#89981). + """ + try: + from providers import get_provider_profile + profile = get_provider_profile(str(provider or "").strip().lower()) + return profile is not None and profile.supports_vision_tool_messages is False + except Exception: + return False + + def _supports_media_in_tool_results(provider: str, model: str) -> bool: """Whether the given provider+model combination accepts image content inside a tool-result message. @@ -1100,7 +1116,7 @@ def _supports_media_in_tool_results(provider: str, model: str) -> bool: if not isinstance(provider, str): return False p = provider.strip().lower() - if not p: + if not p or _profile_rejects_tool_media(p): return False # Aggregators that route to multiple vendors — assume support since @@ -1170,6 +1186,11 @@ def _should_use_native_vision_fast_path() -> bool: cfg = load_config() if decide_image_input_mode(provider, model, cfg) != "native": return False + # The profile veto applies ahead of the capability lookup too: a + # model marked vision-capable by models.dev / custom_providers must + # not re-open the multimodal-envelope route the profile rejects. + if _profile_rejects_tool_media(provider): + return False return ( _supports_media_in_tool_results(provider, model) or _lookup_supports_vision(provider, model, cfg) is True