fix(agent): cover the remaining refusal-only surfaces and fold the tests

- codex_runtime._CODEX_PROGRESS_DELTA_TYPES gains response.refusal.delta so the
  stream watchdog sees progress on a refusal-only stream instead of timing it
  out as idle.
- auxiliary_client._parse_codex_final_response reads type=refusal content
  parts; without it an aux refusal-only turn parsed to content=None and hit the
  empty-response path the main loop was just taught to avoid.
- tests: parametrize test_streamed_refusal_accumulated (refusal-only /
  alongside-content) so there is one test per surface; drop upstream product
  references from docstrings (credit stays in the PR body); pass encoding= to
  the read_text calls flagged by the Windows footgun scanner.
- docs: fallback-providers notes that a streamed refusal is a terminal
  content_filter result, not an empty response to retry.
This commit is contained in:
teknium1
2026-09-14 21:05:25 -07:00
committed by Teknium
parent e27f16365d
commit 88f2844d46
5 changed files with 43 additions and 73 deletions

View File

@@ -1077,8 +1077,13 @@ def _parse_codex_final_response(final: Any) -> Tuple[List[str], List[Any], Any]:
item_type = _field(item, "type")
if item_type == "message":
for part in (_field(item, "content") or []):
if _field(part, "type") in {"output_text", "text"}:
part_type = _field(part, "type")
if part_type in {"output_text", "text"}:
text_parts.append(_field(part, "text", ""))
elif part_type == "refusal":
# A refusal part carries the model's explanation; dropping it turns a
# refusal-only turn into an empty response that gets retried.
text_parts.append(_field(part, "refusal", ""))
elif item_type == "function_call":
tool_calls_raw.append(SimpleNamespace(
id=_field(item, "call_id", ""), type="function",

View File

@@ -533,6 +533,7 @@ def _event_field(event: Any, name: str, default: Any = None) -> Any:
_CODEX_PROGRESS_DELTA_TYPES = frozenset({
"response.output_text.delta", "response.reasoning_summary_text.delta", "response.text.delta",
"response.audio.delta", "response.function_call_arguments.delta", "response.reasoning_text.delta",
"response.refusal.delta",
})

View File

@@ -848,7 +848,7 @@ def test_consume_codex_stream_routes_commentary_phase_deltas_to_reasoning(monkey
def test_consume_codex_stream_collects_refusal_deltas_as_text(monkeypatch):
"""A refusal-only Responses stream yields usable text, not RuntimeError.
Port of anomalyco/opencode#43343: the model declines and streams the
The model declines and streams the
explanation via ``response.refusal.delta`` with no output_text and (on
some compatible backends) no output_item.done — without collecting the
refusal the consumer sees zero usable content.
@@ -2173,7 +2173,7 @@ def test_dump_api_request_debug_uses_responses_url(monkeypatch, tmp_path):
dump_file = agent._dump_api_request_debug(_codex_request_kwargs(), reason="preflight")
payload = json.loads(dump_file.read_text())
payload = json.loads(dump_file.read_text(encoding="utf-8"))
assert payload["request"]["url"] == "http://127.0.0.1:9208/v1/responses"
@@ -2197,7 +2197,7 @@ def test_dump_api_request_debug_uses_chat_completions_url(monkeypatch, tmp_path)
reason="preflight",
)
payload = json.loads(dump_file.read_text())
payload = json.loads(dump_file.read_text(encoding="utf-8"))
assert payload["request"]["url"] == "http://127.0.0.1:9208/v1/chat/completions"

View File

@@ -521,70 +521,37 @@ class TestStreamingAccumulator:
# ── Test: Streaming Callbacks ────────────────────────────────────────────
@pytest.mark.parametrize(
"chunks, expect_content, expect_finish, expect_refusal",
[
pytest.param(
[(None, "I can't"), (None, " help with that."), (None, None)],
"I can't help with that.", "content_filter", "I can't help with that.",
id="refusal-only",
),
pytest.param(
[("Partial answer.", None), (None, "But I won't do the rest."), (None, None)],
"Partial answer.", "stop", "But I won't do the rest.",
id="refusal-alongside-content",
),
],
)
@patch("run_agent.AIAgent._create_request_openai_client")
@patch("run_agent.AIAgent._close_request_openai_client")
def test_streamed_refusal_accumulated(self, mock_close, mock_create):
"""delta.refusal streams assemble onto message.refusal (opencode#43343).
def test_streamed_refusal_accumulated(
self, mock_close, mock_create, chunks, expect_content, expect_finish, expect_refusal
):
"""delta.refusal streams assemble onto message.refusal.
A refusal-only stream must not raise EmptyStreamError, and the
assembled message must expose ``refusal`` so the transport's
normalize_response promotes it to content + content_filter.
A refusal-only stream must not raise EmptyStreamError; the transport's
normalize_response promotes a sole-payload refusal to content +
content_filter, while a refusal next to real content stays a normal
usable turn with the note in provider_data.
"""
from run_agent import AIAgent
def _refusal_chunk(refusal, finish_reason=None):
delta = SimpleNamespace(
content=None,
tool_calls=None,
reasoning_content=None,
reasoning=None,
refusal=refusal,
)
choice = SimpleNamespace(index=0, delta=delta, finish_reason=finish_reason)
return SimpleNamespace(choices=[choice], model="test-model", usage=None)
chunks = [
_refusal_chunk("I can't"),
_refusal_chunk(" help with that."),
_refusal_chunk(None, finish_reason="stop"),
]
mock_client = MagicMock()
mock_client.chat.completions.create.return_value = iter(chunks)
mock_create.return_value = mock_client
agent = AIAgent(
api_key="test-key",
base_url="https://openrouter.ai/api/v1",
model="test/model",
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
)
agent.api_mode = "chat_completions"
agent._interrupt_requested = False
response = agent._interruptible_streaming_api_call({})
message = response.choices[0].message
assert message.refusal == "I can't help with that."
assert message.content is None
# The transport promotes a sole-payload refusal to a terminal
# content_filter (same contract as the non-streaming path, #46013).
from agent.transports.chat_completions import ChatCompletionsTransport
normalized = ChatCompletionsTransport().normalize_response(response)
assert normalized.content == "I can't help with that."
assert normalized.finish_reason == "content_filter"
@patch("run_agent.AIAgent._create_request_openai_client")
@patch("run_agent.AIAgent._close_request_openai_client")
def test_streamed_refusal_alongside_content_not_terminal(self, mock_close, mock_create):
"""A refusal note next to real content stays a normal usable turn."""
from run_agent import AIAgent
def _chunk(content=None, refusal=None, finish_reason=None):
def _chunk(content, refusal, finish_reason=None):
delta = SimpleNamespace(
content=content,
tool_calls=None,
@@ -595,14 +562,11 @@ class TestStreamingAccumulator:
choice = SimpleNamespace(index=0, delta=delta, finish_reason=finish_reason)
return SimpleNamespace(choices=[choice], model="test-model", usage=None)
chunks = [
_chunk(content="Partial answer."),
_chunk(refusal="But I won't do the rest."),
_chunk(finish_reason="stop"),
]
*body, last = chunks
stream = [_chunk(*c) for c in body] + [_chunk(*last, finish_reason="stop")]
mock_client = MagicMock()
mock_client.chat.completions.create.return_value = iter(chunks)
mock_client.chat.completions.create.return_value = iter(stream)
mock_create.return_value = mock_client
agent = AIAgent(
@@ -617,13 +581,13 @@ class TestStreamingAccumulator:
agent._interrupt_requested = False
response = agent._interruptible_streaming_api_call({})
from agent.transports.chat_completions import ChatCompletionsTransport
assert response.choices[0].message.refusal == expect_refusal
normalized = ChatCompletionsTransport().normalize_response(response)
assert normalized.content == "Partial answer."
assert normalized.finish_reason == "stop"
assert normalized.provider_data["refusal"] == "But I won't do the rest."
assert normalized.content == expect_content
assert normalized.finish_reason == expect_finish
if expect_finish == "stop":
assert normalized.provider_data["refusal"] == expect_refusal
class TestStreamingCallbacks:

View File

@@ -116,7 +116,7 @@ The fallback activates automatically when the primary model fails with:
- **Server errors** (HTTP 500, 502, 503) — after exhausting retry attempts
- **Auth failures** (HTTP 401, 403) — immediately (no point retrying)
- **Not found** (HTTP 404) — immediately
- **Invalid responses** — when the API returns malformed or empty responses repeatedly
- **Invalid responses** — when the API returns malformed or empty responses repeatedly. A streamed refusal (the model declining with an explanation on the refusal channel) is a terminal `content_filter` result, not an empty response, so it is surfaced rather than retried.
When triggered, Hermes: