diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 1d9999339e..2bf36e4ec1 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -1077,8 +1077,13 @@ def _parse_codex_final_response(final: Any) -> Tuple[List[str], List[Any], Any]: item_type = _field(item, "type") if item_type == "message": for part in (_field(item, "content") or []): - if _field(part, "type") in {"output_text", "text"}: + part_type = _field(part, "type") + if part_type in {"output_text", "text"}: text_parts.append(_field(part, "text", "")) + elif part_type == "refusal": + # A refusal part carries the model's explanation; dropping it turns a + # refusal-only turn into an empty response that gets retried. + text_parts.append(_field(part, "refusal", "")) elif item_type == "function_call": tool_calls_raw.append(SimpleNamespace( id=_field(item, "call_id", ""), type="function", diff --git a/agent/codex_runtime.py b/agent/codex_runtime.py index 8cf7e2bb1e..49abfeb10b 100644 --- a/agent/codex_runtime.py +++ b/agent/codex_runtime.py @@ -533,6 +533,7 @@ def _event_field(event: Any, name: str, default: Any = None) -> Any: _CODEX_PROGRESS_DELTA_TYPES = frozenset({ "response.output_text.delta", "response.reasoning_summary_text.delta", "response.text.delta", "response.audio.delta", "response.function_call_arguments.delta", "response.reasoning_text.delta", + "response.refusal.delta", }) diff --git a/tests/agent/test_run_agent_codex_responses.py b/tests/agent/test_run_agent_codex_responses.py index b6ad9be33a..3638e5679f 100644 --- a/tests/agent/test_run_agent_codex_responses.py +++ b/tests/agent/test_run_agent_codex_responses.py @@ -848,7 +848,7 @@ def test_consume_codex_stream_routes_commentary_phase_deltas_to_reasoning(monkey def test_consume_codex_stream_collects_refusal_deltas_as_text(monkeypatch): """A refusal-only Responses stream yields usable text, not RuntimeError. - Port of anomalyco/opencode#43343: the model declines and streams the + The model declines and streams the explanation via ``response.refusal.delta`` with no output_text and (on some compatible backends) no output_item.done — without collecting the refusal the consumer sees zero usable content. @@ -2173,7 +2173,7 @@ def test_dump_api_request_debug_uses_responses_url(monkeypatch, tmp_path): dump_file = agent._dump_api_request_debug(_codex_request_kwargs(), reason="preflight") - payload = json.loads(dump_file.read_text()) + payload = json.loads(dump_file.read_text(encoding="utf-8")) assert payload["request"]["url"] == "http://127.0.0.1:9208/v1/responses" @@ -2197,7 +2197,7 @@ def test_dump_api_request_debug_uses_chat_completions_url(monkeypatch, tmp_path) reason="preflight", ) - payload = json.loads(dump_file.read_text()) + payload = json.loads(dump_file.read_text(encoding="utf-8")) assert payload["request"]["url"] == "http://127.0.0.1:9208/v1/chat/completions" diff --git a/tests/agent/test_streaming.py b/tests/agent/test_streaming.py index a2be5abdbd..116bc9267b 100644 --- a/tests/agent/test_streaming.py +++ b/tests/agent/test_streaming.py @@ -521,70 +521,37 @@ class TestStreamingAccumulator: # ── Test: Streaming Callbacks ──────────────────────────────────────────── + @pytest.mark.parametrize( + "chunks, expect_content, expect_finish, expect_refusal", + [ + pytest.param( + [(None, "I can't"), (None, " help with that."), (None, None)], + "I can't help with that.", "content_filter", "I can't help with that.", + id="refusal-only", + ), + pytest.param( + [("Partial answer.", None), (None, "But I won't do the rest."), (None, None)], + "Partial answer.", "stop", "But I won't do the rest.", + id="refusal-alongside-content", + ), + ], + ) @patch("run_agent.AIAgent._create_request_openai_client") @patch("run_agent.AIAgent._close_request_openai_client") - def test_streamed_refusal_accumulated(self, mock_close, mock_create): - """delta.refusal streams assemble onto message.refusal (opencode#43343). + def test_streamed_refusal_accumulated( + self, mock_close, mock_create, chunks, expect_content, expect_finish, expect_refusal + ): + """delta.refusal streams assemble onto message.refusal. - A refusal-only stream must not raise EmptyStreamError, and the - assembled message must expose ``refusal`` so the transport's - normalize_response promotes it to content + content_filter. + A refusal-only stream must not raise EmptyStreamError; the transport's + normalize_response promotes a sole-payload refusal to content + + content_filter, while a refusal next to real content stays a normal + usable turn with the note in provider_data. """ from run_agent import AIAgent - - def _refusal_chunk(refusal, finish_reason=None): - delta = SimpleNamespace( - content=None, - tool_calls=None, - reasoning_content=None, - reasoning=None, - refusal=refusal, - ) - choice = SimpleNamespace(index=0, delta=delta, finish_reason=finish_reason) - return SimpleNamespace(choices=[choice], model="test-model", usage=None) - - chunks = [ - _refusal_chunk("I can't"), - _refusal_chunk(" help with that."), - _refusal_chunk(None, finish_reason="stop"), - ] - - mock_client = MagicMock() - mock_client.chat.completions.create.return_value = iter(chunks) - mock_create.return_value = mock_client - - agent = AIAgent( - api_key="test-key", - base_url="https://openrouter.ai/api/v1", - model="test/model", - quiet_mode=True, - skip_context_files=True, - skip_memory=True, - ) - agent.api_mode = "chat_completions" - agent._interrupt_requested = False - - response = agent._interruptible_streaming_api_call({}) - - message = response.choices[0].message - assert message.refusal == "I can't help with that." - assert message.content is None - - # The transport promotes a sole-payload refusal to a terminal - # content_filter (same contract as the non-streaming path, #46013). from agent.transports.chat_completions import ChatCompletionsTransport - normalized = ChatCompletionsTransport().normalize_response(response) - assert normalized.content == "I can't help with that." - assert normalized.finish_reason == "content_filter" - - @patch("run_agent.AIAgent._create_request_openai_client") - @patch("run_agent.AIAgent._close_request_openai_client") - def test_streamed_refusal_alongside_content_not_terminal(self, mock_close, mock_create): - """A refusal note next to real content stays a normal usable turn.""" - from run_agent import AIAgent - - def _chunk(content=None, refusal=None, finish_reason=None): + def _chunk(content, refusal, finish_reason=None): delta = SimpleNamespace( content=content, tool_calls=None, @@ -595,14 +562,11 @@ class TestStreamingAccumulator: choice = SimpleNamespace(index=0, delta=delta, finish_reason=finish_reason) return SimpleNamespace(choices=[choice], model="test-model", usage=None) - chunks = [ - _chunk(content="Partial answer."), - _chunk(refusal="But I won't do the rest."), - _chunk(finish_reason="stop"), - ] + *body, last = chunks + stream = [_chunk(*c) for c in body] + [_chunk(*last, finish_reason="stop")] mock_client = MagicMock() - mock_client.chat.completions.create.return_value = iter(chunks) + mock_client.chat.completions.create.return_value = iter(stream) mock_create.return_value = mock_client agent = AIAgent( @@ -617,13 +581,13 @@ class TestStreamingAccumulator: agent._interrupt_requested = False response = agent._interruptible_streaming_api_call({}) - - from agent.transports.chat_completions import ChatCompletionsTransport + assert response.choices[0].message.refusal == expect_refusal normalized = ChatCompletionsTransport().normalize_response(response) - assert normalized.content == "Partial answer." - assert normalized.finish_reason == "stop" - assert normalized.provider_data["refusal"] == "But I won't do the rest." + assert normalized.content == expect_content + assert normalized.finish_reason == expect_finish + if expect_finish == "stop": + assert normalized.provider_data["refusal"] == expect_refusal class TestStreamingCallbacks: diff --git a/website/docs/user-guide/features/fallback-providers.md b/website/docs/user-guide/features/fallback-providers.md index 50d465f655..b59080755f 100644 --- a/website/docs/user-guide/features/fallback-providers.md +++ b/website/docs/user-guide/features/fallback-providers.md @@ -116,7 +116,7 @@ The fallback activates automatically when the primary model fails with: - **Server errors** (HTTP 500, 502, 503) — after exhausting retry attempts - **Auth failures** (HTTP 401, 403) — immediately (no point retrying) - **Not found** (HTTP 404) — immediately -- **Invalid responses** — when the API returns malformed or empty responses repeatedly +- **Invalid responses** — when the API returns malformed or empty responses repeatedly. A streamed refusal (the model declining with an explanation on the refusal channel) is a terminal `content_filter` result, not an empty response, so it is surfaced rather than retried. When triggered, Hermes: