diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 02f298bcde..8b3244c09e 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -1016,16 +1016,23 @@ _UNMERGEABLE = object() def drop_thinking_only_and_merge_users( - messages: List[Dict[str, Any]], *, drop_codex_reasoning_items: bool = True + messages: List[Dict[str, Any]], *, drop_codex_reasoning_items: bool = True, + drop_nudge_marker: Optional[str] = None, ) -> List[Dict[str, Any]]: """Drop thinking-only assistant turns and merge adjacent user messages left behind, on the per-call ``api_messages`` copy only (``agent.messages`` is never mutated). Drop-and-merge - (not stub text) keeps history honest and preserves role alternation.""" + (not stub text) keeps history honest and preserves role alternation. + + ``drop_nudge_marker`` (#67321): user rows equal to the marker — the synthetic Codex + continuation nudge — are dropped too once the turn has crossed to a non-Codex provider; + doing it in this pass keeps alternation valid when the nudge sat between dropped + reasoning-only interims and a tool result rather than next to the user's message.""" if not messages: return messages kept = [ m for m in messages - if not _ra().AIAgent._is_thinking_only_assistant(m, drop_codex_reasoning_items=drop_codex_reasoning_items) + if not (drop_nudge_marker is not None and m.get("role") == "user" and m.get("content") == drop_nudge_marker) + and not _ra().AIAgent._is_thinking_only_assistant(m, drop_codex_reasoning_items=drop_codex_reasoning_items) ] dropped = len(messages) - len(kept) merged: List[Dict[str, Any]] = [] diff --git a/agent/error_classifier.py b/agent/error_classifier.py index d6b0088d32..b18f6a972d 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -48,7 +48,8 @@ class FailoverReason(enum.Enum): image_corrupt = "image_corrupt" # Provider can't decode image bytes — strip and retry (shrinking won't help) model_not_found = "model_not_found" # 404 or invalid model — fallback to different model provider_policy_blocked = "provider_policy_blocked" # Aggregator account data/privacy policy excluded the only endpoint - content_policy_blocked = "content_policy_blocked" # Provider safety filter rejected this prompt — don't retry unchanged + content_policy_blocked = "content_policy_blocked" # Provider safety filter rejected this prompt — deterministic per-request, don't retry unchanged + incomplete_response = "incomplete_response" # Codex/Responses turn stuck emitting reasoning only (no answer, no tool call) after replay + nudge — hand to a different provider format_error = "format_error" # 400 bad request — abort or strip + retry role_alternation = "role_alternation" # Strict chat template rejected adjacent same-role messages — merge them for this destination and retry invalid_encrypted_content = "invalid_encrypted_content" # Responses replay blob rejected — strip replay state and retry diff --git a/agent/turn_context.py b/agent/turn_context.py index 62d367d909..835cf7b1da 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -507,6 +507,9 @@ def _bind_turn_identity( _PER_TURN_RESET_STATE: Tuple[Tuple[str, Any], ...] = ( ("_invalid_tool_retries", 0), ("_invalid_json_retries", 0), ("_empty_content_retries", 0), ("_incomplete_scratchpad_retries", 0), ("_codex_incomplete_retries", 0), + # Consecutive Codex reasoning-only (no answer, no tool call) responses, kept apart from + # the aggregate incomplete count so a visible partial resets it (#67321). + ("_codex_reasoning_only_streak", 0), ("_thinking_prefill_retries", 0), ("_post_tool_empty_retried", False), ("_last_content_with_tools", None), ("_last_content_tools_all_housekeeping", False), ("_mute_post_response", False), ("_unicode_sanitization_passes", 0), diff --git a/agent/turn_request_assembly.py b/agent/turn_request_assembly.py index 3595532097..79a56c6278 100644 --- a/agent/turn_request_assembly.py +++ b/agent/turn_request_assembly.py @@ -112,8 +112,8 @@ def assemble_api_request( are injected only after whitespace normalization, the orphan sweep, thinking-only drop / user merge and surrogate stripping, so the same row's bytes never vary across turns.""" from agent.conversation_loop import ( - _apply_context_engine_selection, _canonicalize_api_tool_calls, _clone_message_for_send, - _midturn_request_pressure_tokens, _pressure_with_real_floor, + _CODEX_INCOMPLETE_NUDGE, _apply_context_engine_selection, _canonicalize_api_tool_calls, + _clone_message_for_send, _midturn_request_pressure_tokens, _pressure_with_real_floor, ) from agent.model_metadata import estimate_messages_tokens_rough @@ -167,8 +167,13 @@ def assemble_api_request( # Drop thinking-only assistant turns + merge adjacent users, API copy only: # Anthropic-style backends 400 on a trailing `thinking` block; history keeps it. + # Off the Codex wire (e.g. after a reasoning-only stall fell over to a Chat Completions + # provider, #67321) the synthetic continuation nudge is Codex-only control text: drop it + # alongside the opaque replay state. + _cross_protocol = agent.api_mode != "codex_responses" api_messages = agent._drop_thinking_only_and_merge_users( - api_messages, drop_codex_reasoning_items=agent.api_mode != "codex_responses" + api_messages, drop_codex_reasoning_items=_cross_protocol, + drop_nudge_marker=_CODEX_INCOMPLETE_NUDGE if _cross_protocol else None, ) # Normalize whitespace and tool-call JSON for bit-perfect prefixes across turns diff --git a/agent/turn_response_intake.py b/agent/turn_response_intake.py index b24e6bdf84..67374daa78 100644 --- a/agent/turn_response_intake.py +++ b/agent/turn_response_intake.py @@ -14,7 +14,9 @@ from typing import Any, Dict, Optional from agent.provider_projection import splice_provider_projection from agent.trajectory import has_incomplete_scratchpad -from agent.turn_truncation import continue_codex_incomplete, normalize_response_for_agent, partial_result +from agent.turn_truncation import ( + CODEX_FALLBACK_ACTIVATED, continue_codex_incomplete, normalize_response_for_agent, partial_result, +) logger = logging.getLogger("agent.conversation_loop") @@ -25,12 +27,14 @@ _REASONING_TAG_RE = re.compile(r'') class ResponseIntakeVerdict: """``action``: ``"fallthrough"`` (process ``assistant_message``), ``"continue"`` (retry the iteration: incomplete scratchpad / Codex continuation) or ``"return"`` (``result`` is the - turn's result dict). ``assistant_message``/``finish_reason`` are the normalized outputs.""" + turn's result dict). ``assistant_message``/``finish_reason`` are the normalized outputs; + ``active_system_prompt`` is rebound after a Codex reasoning-only fallover (#67321).""" action: str assistant_message: Any finish_reason: Any result: Optional[Dict[str, Any]] = None + active_system_prompt: Any = None def _coerce_content_text(raw: Any) -> str: @@ -115,7 +119,7 @@ def _relay_thinking(agent: Any, content: str) -> None: def normalize_model_response( agent: Any, *, response: Any, messages: Any, api_messages: Any, conversation_history: Any, api_call_count: Any, api_duration: Any, api_start_time: Any, api_request_id: Any, - effective_task_id: Any, turn_id: Any, + effective_task_id: Any, turn_id: Any, active_system_prompt: Any = None, ) -> ResponseIntakeVerdict: """Normalize ``response`` into ``assistant_message`` (str content, never dict/list) and run the post-response hooks and continuation guards, in the original order.""" @@ -125,7 +129,7 @@ def normalize_model_response( def _verdict(action: str, result: Optional[Dict[str, Any]] = None) -> ResponseIntakeVerdict: return ResponseIntakeVerdict( action=action, assistant_message=assistant_message, finish_reason=finish_reason, - result=result, + result=result, active_system_prompt=active_system_prompt, ) if assistant_message.content is not None and not isinstance(assistant_message.content, str): @@ -175,9 +179,16 @@ def normalize_model_response( conversation_history=conversation_history, api_call_count=api_call_count, response=response, ) + if _codex_result is CODEX_FALLBACK_ACTIVATED: + # The failover rewrote the Model:/Provider: identity on the cached system prompt; + # rebind it so the next iteration's request is rebuilt with the new identity. + from agent.conversation_loop import _sync_failover_system_message + active_system_prompt = _sync_failover_system_message(agent, api_messages, active_system_prompt) + return _verdict("continue") if _codex_result is not None: return _verdict("return", _codex_result) return _verdict("continue") if hasattr(agent, "_codex_incomplete_retries"): agent._codex_incomplete_retries = 0 + agent._codex_reasoning_only_streak = 0 return _verdict("fallthrough") diff --git a/agent/turn_truncation.py b/agent/turn_truncation.py index dc5bd57e0c..0be2085034 100644 --- a/agent/turn_truncation.py +++ b/agent/turn_truncation.py @@ -447,11 +447,15 @@ _CODEX_REPLAY_KEYS = ( "codex_reasoning_items", "codex_message_items", ) +# Third return value of ``continue_codex_incomplete``: the reasoning-only stall was handed to a +# fallback provider — the caller re-syncs the system prompt identity and continues the turn. +CODEX_FALLBACK_ACTIVATED = "codex_fallback_activated" + def continue_codex_incomplete( agent: Any, assistant_message: Any, finish_reason: str, *, messages: List[Dict[str, Any]], conversation_history: Any, api_call_count: int, response: Any = None, -) -> Optional[Dict[str, Any]]: +) -> Optional[Any]: """Codex Responses ``status=incomplete`` continuation (max 3 per turn). Appends the interim assistant message (deduped on visible content only — opaque @@ -459,7 +463,17 @@ def continue_codex_incomplete( overwritten, because the earlier response holds the only native-compaction checkpoint) and, when a bare retry would be byte-identical, a user-role nudge — only after an assistant row, to preserve role alternation. Returns ``None`` to continue - the turn loop, or the terminal ``partial`` result once retries are exhausted. + the turn loop, ``CODEX_FALLBACK_ACTIVATED`` when a reasoning-only stall was handed to + the next fallback provider, or the terminal ``partial`` result once retries are exhausted. + + Reasoning-only stall ladder (#67321): a response with neither visible text nor a tool + call advances ``_codex_reasoning_only_streak`` (a visible partial resets it; the aggregate + ``_codex_incomplete_retries`` stays the cap for partials). Encrypted reasoning replays + byte-for-byte, so after replay (1) and nudge (2) the third consecutive reasoning-only + response goes to the configured fallback with the semantic ``incomplete_response`` reason + instead of ending on the sentinel; when that response consumed the last iteration the + fallback gets exactly one grace call (``_budget_grace_call`` is consumed by the next + iteration, and the streak restarts from 0, so a second grace call is unreachable). When ``response`` hit ``max_output_tokens`` with no visible text (reasoning ate the whole budget), the next attempt goes out with reasoning off and a doubled output @@ -477,6 +491,9 @@ def continue_codex_incomplete( interim_has_reasoning = isinstance(_reasoning, str) and bool(_reasoning.strip()) interim_has_codex_reasoning = bool(interim_msg.get("codex_reasoning_items")) interim_has_codex_message_items = bool(interim_msg.get("codex_message_items")) + reasoning_only = not interim_has_content and not getattr(assistant_message, "tool_calls", None) + agent._codex_reasoning_only_streak = agent._codex_reasoning_only_streak + 1 if reasoning_only else 0 + streak = agent._codex_reasoning_only_streak if interim_has_content or interim_has_reasoning or interim_has_codex_reasoning or interim_has_codex_message_items: last_msg = messages[-1] if messages else None @@ -510,7 +527,26 @@ def continue_codex_incomplete( append_message(messages, interim_msg) agent._emit_interim_assistant_message(interim_msg) - if n < 3: + if reasoning_only and streak >= 3: + if agent._try_activate_fallback(reason=FailoverReason.incomplete_response): + # The trigger may have consumed the turn budget; without a grace call the loop + # exits before the fallback is ever asked. + if api_call_count >= agent.max_iterations or agent.iteration_budget.remaining <= 0: + agent._budget_grace_call = True + agent._codex_incomplete_retries = 0 + agent._codex_reasoning_only_streak = 0 + if not agent.quiet_mode: + agent._vprint( + f"{agent.log_prefix}↻ Codex reasoning-only stall after {streak} attempts — " + f"switching to fallback {agent.model} ({agent.provider})", diagnostic=True, + ) + agent._emit_diagnostic_wait("↻ model stuck on internal reasoning — switching to fallback provider") + agent._session_messages = messages + return CODEX_FALLBACK_ACTIVATED + # No fallback left: fall through to the terminal sentinel. + elif n < 3 or reasoning_only: + # A reasoning-only streak below 3 continues even once partials used up the aggregate + # cap, so the mixed partial-then-stall variant reaches the ladder above. # If the interim has nothing the Responses converter will replay, a bare retry is # byte-identical; a replayable interim holding only a ``compaction`` checkpoint # ALSO re-sends identically. One bare retry, then always nudge. @@ -551,6 +587,7 @@ def continue_codex_incomplete( return None agent._codex_incomplete_retries = 0 + agent._codex_reasoning_only_streak = 0 agent._persist_session(messages, conversation_history) return partial_result( messages, api_call_count, "Codex response remained incomplete after 3 continuation attempts" diff --git a/tests/agent/test_codex_incomplete_budget_escalation.py b/tests/agent/test_codex_incomplete_budget_escalation.py index 4fe2d0f339..3a76b51e85 100644 --- a/tests/agent/test_codex_incomplete_budget_escalation.py +++ b/tests/agent/test_codex_incomplete_budget_escalation.py @@ -18,6 +18,7 @@ def _agent(max_tokens: int | None = 2000): agent.quiet_mode = True agent.log_prefix = "" agent._codex_incomplete_retries = 0 + agent._codex_reasoning_only_streak = 0 agent._ephemeral_reasoning_off = False agent._ephemeral_max_output_tokens = None agent._build_assistant_message.side_effect = lambda msg, fr: { diff --git a/tests/agent/test_codex_reasoning_only_streak.py b/tests/agent/test_codex_reasoning_only_streak.py new file mode 100644 index 0000000000..4d8b21bcf9 --- /dev/null +++ b/tests/agent/test_codex_reasoning_only_streak.py @@ -0,0 +1,108 @@ +"""Codex Responses reasoning-only stall recovery (#67321). + +Encrypted reasoning items replay byte-for-byte, so a bare continuation of a +reasoning-only ``status=incomplete`` response repeats the stall. After three +consecutive reasoning-only responses the turn must reach the configured +fallback provider (with one bounded grace call when the trigger consumed the +iteration budget) instead of ending on the internal incomplete sentinel; a +visible partial resets the local streak; a cross-protocol fallback drops the +Codex-only nudge from the wire. +""" + +from __future__ import annotations + +import run_agent +from agent.agent_runtime_helpers import drop_thinking_only_and_merge_users +from agent.conversation_loop import _CODEX_INCOMPLETE_NUDGE +from agent.error_classifier import FailoverReason +from tests.agent.test_run_agent_codex_responses import ( + _build_agent, + _codex_incomplete_message_response, + _codex_message_response, + _codex_reasoning_only_response, +) + + +def _spy_fallback(agent, monkeypatch): + """Record fallback activations; keep ``api_mode`` on codex_responses so the + stub fallback answer still parses through the Codex path.""" + calls = [] + + def _fake(reason=None): + calls.append(reason) + return True + + monkeypatch.setattr(agent, "_try_activate_fallback", _fake) + return calls + + +def _drive(agent, monkeypatch, responses): + api_calls = {"n": 0} + + def _fake_api_call(api_kwargs): + api_calls["n"] += 1 + return responses.pop(0) + + monkeypatch.setattr(agent, "_interruptible_api_call", _fake_api_call) + return api_calls + + +def test_reasoning_only_streak_reaches_fallback_with_one_grace_call(monkeypatch): + agent = _build_agent(monkeypatch) + agent.max_iterations = 3 + agent.iteration_budget = run_agent.IterationBudget(3) + calls = _spy_fallback(agent, monkeypatch) + api_calls = _drive(agent, monkeypatch, [ + _codex_reasoning_only_response(encrypted_content="enc_a"), + _codex_reasoning_only_response(encrypted_content="enc_b"), + _codex_reasoning_only_response(encrypted_content="enc_c"), + _codex_message_response("Fallback answered."), + ]) + + result = agent.run_conversation("keep thinking") + + assert result["completed"] is True + assert result["final_response"] == "Fallback answered." + assert calls == [FailoverReason.incomplete_response] + # Three budgeted calls + exactly one grace call; the grace flag is consumed. + assert api_calls["n"] == 4 + assert agent._budget_grace_call is False + + +def test_visible_partial_resets_reasoning_only_streak(monkeypatch): + agent = _build_agent(monkeypatch) + agent.max_iterations = 6 + agent.iteration_budget = run_agent.IterationBudget(6) + calls = _spy_fallback(agent, monkeypatch) + _drive(agent, monkeypatch, [ + _codex_incomplete_message_response("Partial visible progress."), + _codex_reasoning_only_response(encrypted_content="enc_a"), + _codex_reasoning_only_response(encrypted_content="enc_b"), + _codex_reasoning_only_response(encrypted_content="enc_c"), + _codex_message_response("Recovered."), + ]) + + result = agent.run_conversation("partial then stall") + + assert result["completed"] is True + assert result["final_response"] == "Recovered." + assert calls == [FailoverReason.incomplete_response] + + +def test_cross_protocol_wire_drops_codex_nudge_and_keeps_alternation(): + messages = [ + {"role": "user", "content": "do it"}, + {"role": "assistant", "content": "", "tool_calls": [{"id": "c1", "type": "function", + "function": {"name": "terminal", "arguments": "{}"}}]}, + {"role": "tool", "tool_call_id": "c1", "content": "ok"}, + {"role": "assistant", "content": "", "finish_reason": "incomplete", + "codex_reasoning_items": [{"type": "reasoning", "id": "rs_1", "encrypted_content": "x"}]}, + {"role": "user", "content": _CODEX_INCOMPLETE_NUDGE}, + ] + + wire = drop_thinking_only_and_merge_users( + messages, drop_codex_reasoning_items=True, drop_nudge_marker=_CODEX_INCOMPLETE_NUDGE, + ) + + assert [m["role"] for m in wire] == ["user", "assistant", "tool"] + assert not any(m.get("codex_reasoning_items") for m in wire) diff --git a/tests/agent/test_error_classifier.py b/tests/agent/test_error_classifier.py index 8e8dab5aa1..961d2575cc 100644 --- a/tests/agent/test_error_classifier.py +++ b/tests/agent/test_error_classifier.py @@ -70,6 +70,7 @@ class TestFailoverReason: "reasoning_mandatory", "provider_policy_blocked", "content_policy_blocked", + "incomplete_response", "thinking_signature", "long_context_tier", "oauth_long_context_beta_forbidden", "llama_cpp_grammar_pattern",