diff --git a/agent/codex_responses_adapter.py b/agent/codex_responses_adapter.py index f482cd5b24..c884030019 100644 --- a/agent/codex_responses_adapter.py +++ b/agent/codex_responses_adapter.py @@ -571,12 +571,18 @@ def _chat_messages_to_responses_input( msg, is_github_responses=is_github_responses, current_issuer_kind=current_issuer_kind, ) emit(message_items, msg) + fallback = None if not message_items: - # Every reasoning item needs a following item (else missing_following_item), hence the "" fallback. fallback = content_parts or (content_text if content_text.strip() else "" if reasoning_items else None) - if fallback is not None: - emit([{"role": "assistant", "content": fallback}], msg) - emit(_replay_tool_call_items(msg, start_index=len(items), wire_ids=wire_ids), msg) + tool_items = _replay_tool_call_items(msg, start_index=len(items) + (fallback is not None), wire_ids=wire_ids) + # A function_call already follows its reasoning. Inventing an empty assistant + # message between them changes the replayed turn (Muse can emit corrupt finals). + # Keep a follower only for reasoning with no other following item, and make it + # non-empty: strict Responses-compatible providers reject "" with 400. + if fallback is not None and not (fallback == "" and tool_items): + follower = " " if fallback == "" else fallback + emit([{"role": "assistant", "content": follower}], msg) + emit(tool_items, msg) # The server renders nothing placed before a compaction item, so pre-checkpoint history is # dead weight and plaintext asks / merged summaries silently vanish. Keep the newest checkpoint # first, retain pre-checkpoint USER and SUMMARY messages within a token budget, leave the tail. diff --git a/tests/agent/test_responses_empty_tool_replay.py b/tests/agent/test_responses_empty_tool_replay.py new file mode 100644 index 0000000000..452f419221 --- /dev/null +++ b/tests/agent/test_responses_empty_tool_replay.py @@ -0,0 +1,31 @@ +"""Responses replay must not invent a blank assistant turn before a tool call. + +Regression investigation for #103483: synthetic Muse tool-loop A/B reproduction. +Also covers #75202: the lone-reasoning follower must be non-empty for strict +Responses-compatible providers that reject "" with 400. +""" +import pytest +from agent.codex_responses_adapter import _chat_messages_to_responses_input + +@pytest.mark.parametrize('content', ['', None]) +def test_reasoning_tool_round_has_no_invented_assistant_message(content): + messages = [{ + 'role': 'assistant', 'content': content, + 'codex_reasoning_items': [{'type': 'reasoning', 'id': 'rs_test', + 'encrypted_content': 'synthetic-encrypted-fixture', 'summary': []}], + 'tool_calls': [{'id': 'call_test', 'type': 'function', + 'function': {'name': 'record_step', 'arguments': '{"step":1}'}}], + }, {'role': 'tool', 'tool_call_id': 'call_test', 'content': '{"ok":true}'}] + items = _chat_messages_to_responses_input(messages) + assert [item.get('type', item.get('role')) for item in items] == [ + 'reasoning', 'function_call', 'function_call_output'] + assert items[1]['call_id'] == items[2]['call_id'] + + +def test_reasoning_without_tool_keeps_nonempty_following_item(): + items = _chat_messages_to_responses_input([{ + 'role': 'assistant', 'content': '', + 'codex_reasoning_items': [{'type': 'reasoning', 'id': 'rs_test', + 'encrypted_content': 'synthetic-encrypted-fixture', 'summary': []}], + }]) + assert items[-1] == {'role': 'assistant', 'content': ' '}