diff --git a/tests/tools/test_vision_history_budget.py b/tests/tools/test_vision_history_budget.py index c942ae970f..432735de11 100644 --- a/tests/tools/test_vision_history_budget.py +++ b/tests/tools/test_vision_history_budget.py @@ -154,13 +154,60 @@ class TestNativeTurnDedupe: assert _embedded(_load(same, region=[0, 0, 8, 8])) assert _embedded(_load(other)) - def test_scope_ends_with_the_turn(self, tmp_path): + def test_run_conversation_scopes_the_turn_for_the_tool_loop(self, tmp_path, monkeypatch): + """Production wiring: ``run_conversation`` with a native-parts user message; the model + calls ``vision_analyze`` on the attached path from the tool loop and gets the text + result, not a second embed. After the turn the scope is gone and the image embeds again.""" + from types import SimpleNamespace + from agent.image_routing import build_native_content_parts + from run_agent import AIAgent + from tools import vision_tools + same = _png(tmp_path / "same.png") - parts, _ = build_native_content_parts("look", [same]) - with budget.native_turn_images(parts): - pass - assert _embedded(_load(same)) - # A plain-text turn (no native parts) scopes nothing. - with budget.native_turn_images("look at " + same): - assert _embedded(_load(same)) + parts, skipped = build_native_content_parts("look", [same]) + assert not skipped + tool_results = [] + + def _tool_call_turn(): + call = SimpleNamespace(id="call_1", type="function", function=SimpleNamespace( + name="vision_analyze", arguments=json.dumps({"image_url": same, "question": "q"}))) + msg = SimpleNamespace(content=None, reasoning=None, tool_calls=[call]) + return SimpleNamespace(choices=[SimpleNamespace(message=msg, finish_reason="tool_calls")], usage=None) + + def _final_turn(): + msg = SimpleNamespace(content="done", reasoning=None, tool_calls=[]) + return SimpleNamespace(choices=[SimpleNamespace(message=msg, finish_reason="stop")], usage=None) + + class _Completions: + calls = 0 + + def create(self, **kwargs): + self.calls += 1 + return _tool_call_turn() if self.calls == 1 else _final_turn() + + def _dispatch(name, args, task_id=None, **kwargs): + assert name == "vision_analyze" + result = asyncio.new_event_loop().run_until_complete( + vision_tools._handle_vision_analyze(args, task_id=task_id)) + tool_results.append(result) + return result + + monkeypatch.setattr("agent.process_bootstrap.OpenAI", + lambda **kw: SimpleNamespace(chat=SimpleNamespace(completions=_Completions()))) + monkeypatch.setattr("model_tools.get_tool_definitions", + lambda *a, **kw: [{"function": {"name": "vision_analyze"}}]) + monkeypatch.setattr("model_tools.handle_function_call", _dispatch) + monkeypatch.setattr(vision_tools, "_should_use_native_vision_fast_path", lambda: True) + + agent = AIAgent(model="test-model", api_key="test-key", base_url="http://localhost:8080/v1", + platform="cli", max_iterations=3, quiet_mode=True, skip_memory=True) + agent._disable_streaming = True + result = agent.run_conversation(parts) + + assert result["final_response"].startswith("done") + assert len(tool_results) == 1 + assert not _embedded(tool_results[0]) + assert json.loads(tool_results[0])["already_in_context"] is True + # The scope ended with the turn: a later load (after compression) embeds again. + assert _embedded(asyncio.new_event_loop().run_until_complete(_vision_analyze_native(same, "q"))) diff --git a/website/docs/user-guide/features/vision.md b/website/docs/user-guide/features/vision.md index 630f9055f3..ac920f410a 100644 --- a/website/docs/user-guide/features/vision.md +++ b/website/docs/user-guide/features/vision.md @@ -209,7 +209,7 @@ Which auxiliary model handles the text-description path is configurable under `a ### `vision_analyze` has the same dual behavior -The `vision_analyze` tool itself follows the same routing. When the active main model is vision-capable **and** its provider supports image content inside tool results (currently the Anthropic, OpenAI, Azure-OpenAI, and Gemini 3.x stacks), `vision_analyze` short-circuits the auxiliary describer and returns the raw image pixels as a multimodal tool-result envelope. The main model sees the image natively on its next turn — no aux call, no text-summary information loss, no extra latency. +The `vision_analyze` tool itself follows the same routing. When the active main model is vision-capable **and** its provider supports image content inside tool results (currently the Anthropic, OpenAI, Azure-OpenAI, and Gemini 3.x stacks), `vision_analyze` short-circuits the auxiliary describer and returns the raw image pixels as a multimodal tool-result envelope. The main model sees the image natively on its next turn — no aux call, no text-summary information loss, no extra latency. One exception: an image that is already attached natively to the current user message is not re-embedded — `vision_analyze` on that same path returns a short text result saying the image is already in context (pass a `region` to zoom into part of it, which does embed the crop). For text-only main models (or providers whose tool-result channel doesn't carry images), `vision_analyze` falls back to the legacy path: it asks the configured auxiliary vision model to describe the image and returns the description as plain text. Either way the calling tool signature is the same — the tool decides which path to take at runtime based on the active model.