fix(agent): cover the remaining refusal-only surfaces and fold the tests
- codex_runtime._CODEX_PROGRESS_DELTA_TYPES gains response.refusal.delta so the stream watchdog sees progress on a refusal-only stream instead of timing it out as idle. - auxiliary_client._parse_codex_final_response reads type=refusal content parts; without it an aux refusal-only turn parsed to content=None and hit the empty-response path the main loop was just taught to avoid. - tests: parametrize test_streamed_refusal_accumulated (refusal-only / alongside-content) so there is one test per surface; drop upstream product references from docstrings (credit stays in the PR body); pass encoding= to the read_text calls flagged by the Windows footgun scanner. - docs: fallback-providers notes that a streamed refusal is a terminal content_filter result, not an empty response to retry.
This commit is contained in:
@@ -1077,8 +1077,13 @@ def _parse_codex_final_response(final: Any) -> Tuple[List[str], List[Any], Any]:
|
||||
item_type = _field(item, "type")
|
||||
if item_type == "message":
|
||||
for part in (_field(item, "content") or []):
|
||||
if _field(part, "type") in {"output_text", "text"}:
|
||||
part_type = _field(part, "type")
|
||||
if part_type in {"output_text", "text"}:
|
||||
text_parts.append(_field(part, "text", ""))
|
||||
elif part_type == "refusal":
|
||||
# A refusal part carries the model's explanation; dropping it turns a
|
||||
# refusal-only turn into an empty response that gets retried.
|
||||
text_parts.append(_field(part, "refusal", ""))
|
||||
elif item_type == "function_call":
|
||||
tool_calls_raw.append(SimpleNamespace(
|
||||
id=_field(item, "call_id", ""), type="function",
|
||||
|
||||
@@ -533,6 +533,7 @@ def _event_field(event: Any, name: str, default: Any = None) -> Any:
|
||||
_CODEX_PROGRESS_DELTA_TYPES = frozenset({
|
||||
"response.output_text.delta", "response.reasoning_summary_text.delta", "response.text.delta",
|
||||
"response.audio.delta", "response.function_call_arguments.delta", "response.reasoning_text.delta",
|
||||
"response.refusal.delta",
|
||||
})
|
||||
|
||||
|
||||
|
||||
@@ -848,7 +848,7 @@ def test_consume_codex_stream_routes_commentary_phase_deltas_to_reasoning(monkey
|
||||
def test_consume_codex_stream_collects_refusal_deltas_as_text(monkeypatch):
|
||||
"""A refusal-only Responses stream yields usable text, not RuntimeError.
|
||||
|
||||
Port of anomalyco/opencode#43343: the model declines and streams the
|
||||
The model declines and streams the
|
||||
explanation via ``response.refusal.delta`` with no output_text and (on
|
||||
some compatible backends) no output_item.done — without collecting the
|
||||
refusal the consumer sees zero usable content.
|
||||
@@ -2173,7 +2173,7 @@ def test_dump_api_request_debug_uses_responses_url(monkeypatch, tmp_path):
|
||||
|
||||
dump_file = agent._dump_api_request_debug(_codex_request_kwargs(), reason="preflight")
|
||||
|
||||
payload = json.loads(dump_file.read_text())
|
||||
payload = json.loads(dump_file.read_text(encoding="utf-8"))
|
||||
assert payload["request"]["url"] == "http://127.0.0.1:9208/v1/responses"
|
||||
|
||||
|
||||
@@ -2197,7 +2197,7 @@ def test_dump_api_request_debug_uses_chat_completions_url(monkeypatch, tmp_path)
|
||||
reason="preflight",
|
||||
)
|
||||
|
||||
payload = json.loads(dump_file.read_text())
|
||||
payload = json.loads(dump_file.read_text(encoding="utf-8"))
|
||||
assert payload["request"]["url"] == "http://127.0.0.1:9208/v1/chat/completions"
|
||||
|
||||
|
||||
|
||||
@@ -521,70 +521,37 @@ class TestStreamingAccumulator:
|
||||
# ── Test: Streaming Callbacks ────────────────────────────────────────────
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"chunks, expect_content, expect_finish, expect_refusal",
|
||||
[
|
||||
pytest.param(
|
||||
[(None, "I can't"), (None, " help with that."), (None, None)],
|
||||
"I can't help with that.", "content_filter", "I can't help with that.",
|
||||
id="refusal-only",
|
||||
),
|
||||
pytest.param(
|
||||
[("Partial answer.", None), (None, "But I won't do the rest."), (None, None)],
|
||||
"Partial answer.", "stop", "But I won't do the rest.",
|
||||
id="refusal-alongside-content",
|
||||
),
|
||||
],
|
||||
)
|
||||
@patch("run_agent.AIAgent._create_request_openai_client")
|
||||
@patch("run_agent.AIAgent._close_request_openai_client")
|
||||
def test_streamed_refusal_accumulated(self, mock_close, mock_create):
|
||||
"""delta.refusal streams assemble onto message.refusal (opencode#43343).
|
||||
def test_streamed_refusal_accumulated(
|
||||
self, mock_close, mock_create, chunks, expect_content, expect_finish, expect_refusal
|
||||
):
|
||||
"""delta.refusal streams assemble onto message.refusal.
|
||||
|
||||
A refusal-only stream must not raise EmptyStreamError, and the
|
||||
assembled message must expose ``refusal`` so the transport's
|
||||
normalize_response promotes it to content + content_filter.
|
||||
A refusal-only stream must not raise EmptyStreamError; the transport's
|
||||
normalize_response promotes a sole-payload refusal to content +
|
||||
content_filter, while a refusal next to real content stays a normal
|
||||
usable turn with the note in provider_data.
|
||||
"""
|
||||
from run_agent import AIAgent
|
||||
|
||||
def _refusal_chunk(refusal, finish_reason=None):
|
||||
delta = SimpleNamespace(
|
||||
content=None,
|
||||
tool_calls=None,
|
||||
reasoning_content=None,
|
||||
reasoning=None,
|
||||
refusal=refusal,
|
||||
)
|
||||
choice = SimpleNamespace(index=0, delta=delta, finish_reason=finish_reason)
|
||||
return SimpleNamespace(choices=[choice], model="test-model", usage=None)
|
||||
|
||||
chunks = [
|
||||
_refusal_chunk("I can't"),
|
||||
_refusal_chunk(" help with that."),
|
||||
_refusal_chunk(None, finish_reason="stop"),
|
||||
]
|
||||
|
||||
mock_client = MagicMock()
|
||||
mock_client.chat.completions.create.return_value = iter(chunks)
|
||||
mock_create.return_value = mock_client
|
||||
|
||||
agent = AIAgent(
|
||||
api_key="test-key",
|
||||
base_url="https://openrouter.ai/api/v1",
|
||||
model="test/model",
|
||||
quiet_mode=True,
|
||||
skip_context_files=True,
|
||||
skip_memory=True,
|
||||
)
|
||||
agent.api_mode = "chat_completions"
|
||||
agent._interrupt_requested = False
|
||||
|
||||
response = agent._interruptible_streaming_api_call({})
|
||||
|
||||
message = response.choices[0].message
|
||||
assert message.refusal == "I can't help with that."
|
||||
assert message.content is None
|
||||
|
||||
# The transport promotes a sole-payload refusal to a terminal
|
||||
# content_filter (same contract as the non-streaming path, #46013).
|
||||
from agent.transports.chat_completions import ChatCompletionsTransport
|
||||
|
||||
normalized = ChatCompletionsTransport().normalize_response(response)
|
||||
assert normalized.content == "I can't help with that."
|
||||
assert normalized.finish_reason == "content_filter"
|
||||
|
||||
@patch("run_agent.AIAgent._create_request_openai_client")
|
||||
@patch("run_agent.AIAgent._close_request_openai_client")
|
||||
def test_streamed_refusal_alongside_content_not_terminal(self, mock_close, mock_create):
|
||||
"""A refusal note next to real content stays a normal usable turn."""
|
||||
from run_agent import AIAgent
|
||||
|
||||
def _chunk(content=None, refusal=None, finish_reason=None):
|
||||
def _chunk(content, refusal, finish_reason=None):
|
||||
delta = SimpleNamespace(
|
||||
content=content,
|
||||
tool_calls=None,
|
||||
@@ -595,14 +562,11 @@ class TestStreamingAccumulator:
|
||||
choice = SimpleNamespace(index=0, delta=delta, finish_reason=finish_reason)
|
||||
return SimpleNamespace(choices=[choice], model="test-model", usage=None)
|
||||
|
||||
chunks = [
|
||||
_chunk(content="Partial answer."),
|
||||
_chunk(refusal="But I won't do the rest."),
|
||||
_chunk(finish_reason="stop"),
|
||||
]
|
||||
*body, last = chunks
|
||||
stream = [_chunk(*c) for c in body] + [_chunk(*last, finish_reason="stop")]
|
||||
|
||||
mock_client = MagicMock()
|
||||
mock_client.chat.completions.create.return_value = iter(chunks)
|
||||
mock_client.chat.completions.create.return_value = iter(stream)
|
||||
mock_create.return_value = mock_client
|
||||
|
||||
agent = AIAgent(
|
||||
@@ -617,13 +581,13 @@ class TestStreamingAccumulator:
|
||||
agent._interrupt_requested = False
|
||||
|
||||
response = agent._interruptible_streaming_api_call({})
|
||||
|
||||
from agent.transports.chat_completions import ChatCompletionsTransport
|
||||
assert response.choices[0].message.refusal == expect_refusal
|
||||
|
||||
normalized = ChatCompletionsTransport().normalize_response(response)
|
||||
assert normalized.content == "Partial answer."
|
||||
assert normalized.finish_reason == "stop"
|
||||
assert normalized.provider_data["refusal"] == "But I won't do the rest."
|
||||
assert normalized.content == expect_content
|
||||
assert normalized.finish_reason == expect_finish
|
||||
if expect_finish == "stop":
|
||||
assert normalized.provider_data["refusal"] == expect_refusal
|
||||
|
||||
|
||||
class TestStreamingCallbacks:
|
||||
|
||||
@@ -116,7 +116,7 @@ The fallback activates automatically when the primary model fails with:
|
||||
- **Server errors** (HTTP 500, 502, 503) — after exhausting retry attempts
|
||||
- **Auth failures** (HTTP 401, 403) — immediately (no point retrying)
|
||||
- **Not found** (HTTP 404) — immediately
|
||||
- **Invalid responses** — when the API returns malformed or empty responses repeatedly
|
||||
- **Invalid responses** — when the API returns malformed or empty responses repeatedly. A streamed refusal (the model declining with an explanation on the refusal channel) is a terminal `content_filter` result, not an empty response, so it is surfaced rather than retried.
|
||||
|
||||
When triggered, Hermes:
|
||||
|
||||
|
||||
Reference in New Issue
Block a user