Replace the 'describe everything in thorough detail' auto image-preprocess prompt with a concise 2-4 sentence summary prompt so image-bearing gateway messages stop generating ~2000-char descriptions (35s+ on local models). Prompt-only variant of #10852: the max_tokens=500 cap and the preserve_max_tokens aux-client plumbing from the original PR are intentionally dropped to stay compatible with the max-tokens-knob policy direction (#75253 removes hardcoded vision caps). Fixes #10809
35 lines
1.2 KiB
Python
35 lines
1.2 KiB
Python
"""Gateway vision pre-process prompt should stay concise."""
|
|
|
|
import json
|
|
from unittest.mock import AsyncMock, patch
|
|
|
|
import pytest
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_enrich_message_with_vision_uses_concise_prompt():
|
|
from gateway.run import GatewayRunner
|
|
|
|
runner = GatewayRunner.__new__(GatewayRunner)
|
|
|
|
with patch(
|
|
"tools.vision_tools.vision_analyze_tool",
|
|
new_callable=AsyncMock,
|
|
return_value=json.dumps({"success": True, "analysis": "A cat on a chair."}),
|
|
) as mock_vision:
|
|
result = await runner._enrich_message_with_vision(
|
|
user_text="What is happening here?",
|
|
image_paths=["/tmp/cat.png"],
|
|
)
|
|
|
|
assert "A cat on a chair." in result
|
|
assert "What is happening here?" in result
|
|
assert (
|
|
"Concisely describe this image in 2-4 sentences"
|
|
in mock_vision.await_args.kwargs["user_prompt"]
|
|
)
|
|
assert "Skip decorative details." in mock_vision.await_args.kwargs["user_prompt"]
|
|
# No output cap is forwarded: per the max-tokens-knob policy the aux
|
|
# client decides token handling; conciseness comes from the prompt.
|
|
assert "max_tokens" not in mock_vision.await_args.kwargs
|