fix: hand Codex reasoning-only stalls to the fallback provider instead of the incomplete sentinel
Three consecutive Codex Responses answers that carry only (encrypted) reasoning — no visible text, no tool call — used to exhaust the 3-continuation budget and end the turn on "Codex response remained incomplete after 3 continuation attempts", never touching configured fallback_providers (#67321). Encrypted reasoning items replay byte-for-byte, so a bare retry deterministically repeats the stall. - Track a per-turn `_codex_reasoning_only_streak` apart from the aggregate `_codex_incomplete_retries`: a visible partial resets the streak, so the mixed partial-then-stall variant still reaches its own recovery threshold while the turn-wide iteration budget stays the hard bound. - At streak 3, `continue_codex_incomplete` activates the next fallback with the semantic `FailoverReason.incomplete_response`, grants exactly one grace call when the trigger consumed the last iteration, and returns `CODEX_FALLBACK_ACTIVATED`; the intake re-syncs the Model:/Provider: identity on the system prompt. - Off the Codex wire the synthetic continuation nudge is stripped alongside the opaque replay state (`drop_nudge_marker`) so the Chat Completions payload keeps valid role ordering and no Codex-only control text. - No fallback configured: unchanged terminal sentinel, still bounded at 3 calls. Ported from PR #67336 by @PRATHAMESH75 onto the decomposed agent/turn_*.py siblings.
This commit is contained in:
@@ -1016,16 +1016,23 @@ _UNMERGEABLE = object()
|
||||
|
||||
|
||||
def drop_thinking_only_and_merge_users(
|
||||
messages: List[Dict[str, Any]], *, drop_codex_reasoning_items: bool = True
|
||||
messages: List[Dict[str, Any]], *, drop_codex_reasoning_items: bool = True,
|
||||
drop_nudge_marker: Optional[str] = None,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Drop thinking-only assistant turns and merge adjacent user messages left behind, on the
|
||||
per-call ``api_messages`` copy only (``agent.messages`` is never mutated). Drop-and-merge
|
||||
(not stub text) keeps history honest and preserves role alternation."""
|
||||
(not stub text) keeps history honest and preserves role alternation.
|
||||
|
||||
``drop_nudge_marker`` (#67321): user rows equal to the marker — the synthetic Codex
|
||||
continuation nudge — are dropped too once the turn has crossed to a non-Codex provider;
|
||||
doing it in this pass keeps alternation valid when the nudge sat between dropped
|
||||
reasoning-only interims and a tool result rather than next to the user's message."""
|
||||
if not messages:
|
||||
return messages
|
||||
kept = [
|
||||
m for m in messages
|
||||
if not _ra().AIAgent._is_thinking_only_assistant(m, drop_codex_reasoning_items=drop_codex_reasoning_items)
|
||||
if not (drop_nudge_marker is not None and m.get("role") == "user" and m.get("content") == drop_nudge_marker)
|
||||
and not _ra().AIAgent._is_thinking_only_assistant(m, drop_codex_reasoning_items=drop_codex_reasoning_items)
|
||||
]
|
||||
dropped = len(messages) - len(kept)
|
||||
merged: List[Dict[str, Any]] = []
|
||||
|
||||
@@ -48,7 +48,8 @@ class FailoverReason(enum.Enum):
|
||||
image_corrupt = "image_corrupt" # Provider can't decode image bytes — strip and retry (shrinking won't help)
|
||||
model_not_found = "model_not_found" # 404 or invalid model — fallback to different model
|
||||
provider_policy_blocked = "provider_policy_blocked" # Aggregator account data/privacy policy excluded the only endpoint
|
||||
content_policy_blocked = "content_policy_blocked" # Provider safety filter rejected this prompt — don't retry unchanged
|
||||
content_policy_blocked = "content_policy_blocked" # Provider safety filter rejected this prompt — deterministic per-request, don't retry unchanged
|
||||
incomplete_response = "incomplete_response" # Codex/Responses turn stuck emitting reasoning only (no answer, no tool call) after replay + nudge — hand to a different provider
|
||||
format_error = "format_error" # 400 bad request — abort or strip + retry
|
||||
role_alternation = "role_alternation" # Strict chat template rejected adjacent same-role messages — merge them for this destination and retry
|
||||
invalid_encrypted_content = "invalid_encrypted_content" # Responses replay blob rejected — strip replay state and retry
|
||||
|
||||
@@ -507,6 +507,9 @@ def _bind_turn_identity(
|
||||
_PER_TURN_RESET_STATE: Tuple[Tuple[str, Any], ...] = (
|
||||
("_invalid_tool_retries", 0), ("_invalid_json_retries", 0), ("_empty_content_retries", 0),
|
||||
("_incomplete_scratchpad_retries", 0), ("_codex_incomplete_retries", 0),
|
||||
# Consecutive Codex reasoning-only (no answer, no tool call) responses, kept apart from
|
||||
# the aggregate incomplete count so a visible partial resets it (#67321).
|
||||
("_codex_reasoning_only_streak", 0),
|
||||
("_thinking_prefill_retries", 0), ("_post_tool_empty_retried", False),
|
||||
("_last_content_with_tools", None), ("_last_content_tools_all_housekeeping", False),
|
||||
("_mute_post_response", False), ("_unicode_sanitization_passes", 0),
|
||||
|
||||
@@ -112,8 +112,8 @@ def assemble_api_request(
|
||||
are injected only after whitespace normalization, the orphan sweep, thinking-only drop /
|
||||
user merge and surrogate stripping, so the same row's bytes never vary across turns."""
|
||||
from agent.conversation_loop import (
|
||||
_apply_context_engine_selection, _canonicalize_api_tool_calls, _clone_message_for_send,
|
||||
_midturn_request_pressure_tokens, _pressure_with_real_floor,
|
||||
_CODEX_INCOMPLETE_NUDGE, _apply_context_engine_selection, _canonicalize_api_tool_calls,
|
||||
_clone_message_for_send, _midturn_request_pressure_tokens, _pressure_with_real_floor,
|
||||
)
|
||||
from agent.model_metadata import estimate_messages_tokens_rough
|
||||
|
||||
@@ -167,8 +167,13 @@ def assemble_api_request(
|
||||
|
||||
# Drop thinking-only assistant turns + merge adjacent users, API copy only:
|
||||
# Anthropic-style backends 400 on a trailing `thinking` block; history keeps it.
|
||||
# Off the Codex wire (e.g. after a reasoning-only stall fell over to a Chat Completions
|
||||
# provider, #67321) the synthetic continuation nudge is Codex-only control text: drop it
|
||||
# alongside the opaque replay state.
|
||||
_cross_protocol = agent.api_mode != "codex_responses"
|
||||
api_messages = agent._drop_thinking_only_and_merge_users(
|
||||
api_messages, drop_codex_reasoning_items=agent.api_mode != "codex_responses"
|
||||
api_messages, drop_codex_reasoning_items=_cross_protocol,
|
||||
drop_nudge_marker=_CODEX_INCOMPLETE_NUDGE if _cross_protocol else None,
|
||||
)
|
||||
|
||||
# Normalize whitespace and tool-call JSON for bit-perfect prefixes across turns
|
||||
|
||||
@@ -14,7 +14,9 @@ from typing import Any, Dict, Optional
|
||||
|
||||
from agent.provider_projection import splice_provider_projection
|
||||
from agent.trajectory import has_incomplete_scratchpad
|
||||
from agent.turn_truncation import continue_codex_incomplete, normalize_response_for_agent, partial_result
|
||||
from agent.turn_truncation import (
|
||||
CODEX_FALLBACK_ACTIVATED, continue_codex_incomplete, normalize_response_for_agent, partial_result,
|
||||
)
|
||||
|
||||
logger = logging.getLogger("agent.conversation_loop")
|
||||
|
||||
@@ -25,12 +27,14 @@ _REASONING_TAG_RE = re.compile(r'</?(?:REASONING_SCRATCHPAD|think|reasoning)>')
|
||||
class ResponseIntakeVerdict:
|
||||
"""``action``: ``"fallthrough"`` (process ``assistant_message``), ``"continue"`` (retry the
|
||||
iteration: incomplete scratchpad / Codex continuation) or ``"return"`` (``result`` is the
|
||||
turn's result dict). ``assistant_message``/``finish_reason`` are the normalized outputs."""
|
||||
turn's result dict). ``assistant_message``/``finish_reason`` are the normalized outputs;
|
||||
``active_system_prompt`` is rebound after a Codex reasoning-only fallover (#67321)."""
|
||||
|
||||
action: str
|
||||
assistant_message: Any
|
||||
finish_reason: Any
|
||||
result: Optional[Dict[str, Any]] = None
|
||||
active_system_prompt: Any = None
|
||||
|
||||
|
||||
def _coerce_content_text(raw: Any) -> str:
|
||||
@@ -115,7 +119,7 @@ def _relay_thinking(agent: Any, content: str) -> None:
|
||||
def normalize_model_response(
|
||||
agent: Any, *, response: Any, messages: Any, api_messages: Any, conversation_history: Any,
|
||||
api_call_count: Any, api_duration: Any, api_start_time: Any, api_request_id: Any,
|
||||
effective_task_id: Any, turn_id: Any,
|
||||
effective_task_id: Any, turn_id: Any, active_system_prompt: Any = None,
|
||||
) -> ResponseIntakeVerdict:
|
||||
"""Normalize ``response`` into ``assistant_message`` (str content, never dict/list) and run
|
||||
the post-response hooks and continuation guards, in the original order."""
|
||||
@@ -125,7 +129,7 @@ def normalize_model_response(
|
||||
def _verdict(action: str, result: Optional[Dict[str, Any]] = None) -> ResponseIntakeVerdict:
|
||||
return ResponseIntakeVerdict(
|
||||
action=action, assistant_message=assistant_message, finish_reason=finish_reason,
|
||||
result=result,
|
||||
result=result, active_system_prompt=active_system_prompt,
|
||||
)
|
||||
|
||||
if assistant_message.content is not None and not isinstance(assistant_message.content, str):
|
||||
@@ -175,9 +179,16 @@ def normalize_model_response(
|
||||
conversation_history=conversation_history, api_call_count=api_call_count,
|
||||
response=response,
|
||||
)
|
||||
if _codex_result is CODEX_FALLBACK_ACTIVATED:
|
||||
# The failover rewrote the Model:/Provider: identity on the cached system prompt;
|
||||
# rebind it so the next iteration's request is rebuilt with the new identity.
|
||||
from agent.conversation_loop import _sync_failover_system_message
|
||||
active_system_prompt = _sync_failover_system_message(agent, api_messages, active_system_prompt)
|
||||
return _verdict("continue")
|
||||
if _codex_result is not None:
|
||||
return _verdict("return", _codex_result)
|
||||
return _verdict("continue")
|
||||
if hasattr(agent, "_codex_incomplete_retries"):
|
||||
agent._codex_incomplete_retries = 0
|
||||
agent._codex_reasoning_only_streak = 0
|
||||
return _verdict("fallthrough")
|
||||
|
||||
@@ -447,11 +447,15 @@ _CODEX_REPLAY_KEYS = (
|
||||
"codex_reasoning_items", "codex_message_items",
|
||||
)
|
||||
|
||||
# Third return value of ``continue_codex_incomplete``: the reasoning-only stall was handed to a
|
||||
# fallback provider — the caller re-syncs the system prompt identity and continues the turn.
|
||||
CODEX_FALLBACK_ACTIVATED = "codex_fallback_activated"
|
||||
|
||||
|
||||
def continue_codex_incomplete(
|
||||
agent: Any, assistant_message: Any, finish_reason: str, *, messages: List[Dict[str, Any]],
|
||||
conversation_history: Any, api_call_count: int, response: Any = None,
|
||||
) -> Optional[Dict[str, Any]]:
|
||||
) -> Optional[Any]:
|
||||
"""Codex Responses ``status=incomplete`` continuation (max 3 per turn).
|
||||
|
||||
Appends the interim assistant message (deduped on visible content only — opaque
|
||||
@@ -459,7 +463,17 @@ def continue_codex_incomplete(
|
||||
overwritten, because the earlier response holds the only native-compaction
|
||||
checkpoint) and, when a bare retry would be byte-identical, a user-role nudge — only
|
||||
after an assistant row, to preserve role alternation. Returns ``None`` to continue
|
||||
the turn loop, or the terminal ``partial`` result once retries are exhausted.
|
||||
the turn loop, ``CODEX_FALLBACK_ACTIVATED`` when a reasoning-only stall was handed to
|
||||
the next fallback provider, or the terminal ``partial`` result once retries are exhausted.
|
||||
|
||||
Reasoning-only stall ladder (#67321): a response with neither visible text nor a tool
|
||||
call advances ``_codex_reasoning_only_streak`` (a visible partial resets it; the aggregate
|
||||
``_codex_incomplete_retries`` stays the cap for partials). Encrypted reasoning replays
|
||||
byte-for-byte, so after replay (1) and nudge (2) the third consecutive reasoning-only
|
||||
response goes to the configured fallback with the semantic ``incomplete_response`` reason
|
||||
instead of ending on the sentinel; when that response consumed the last iteration the
|
||||
fallback gets exactly one grace call (``_budget_grace_call`` is consumed by the next
|
||||
iteration, and the streak restarts from 0, so a second grace call is unreachable).
|
||||
|
||||
When ``response`` hit ``max_output_tokens`` with no visible text (reasoning ate the
|
||||
whole budget), the next attempt goes out with reasoning off and a doubled output
|
||||
@@ -477,6 +491,9 @@ def continue_codex_incomplete(
|
||||
interim_has_reasoning = isinstance(_reasoning, str) and bool(_reasoning.strip())
|
||||
interim_has_codex_reasoning = bool(interim_msg.get("codex_reasoning_items"))
|
||||
interim_has_codex_message_items = bool(interim_msg.get("codex_message_items"))
|
||||
reasoning_only = not interim_has_content and not getattr(assistant_message, "tool_calls", None)
|
||||
agent._codex_reasoning_only_streak = agent._codex_reasoning_only_streak + 1 if reasoning_only else 0
|
||||
streak = agent._codex_reasoning_only_streak
|
||||
|
||||
if interim_has_content or interim_has_reasoning or interim_has_codex_reasoning or interim_has_codex_message_items:
|
||||
last_msg = messages[-1] if messages else None
|
||||
@@ -510,7 +527,26 @@ def continue_codex_incomplete(
|
||||
append_message(messages, interim_msg)
|
||||
agent._emit_interim_assistant_message(interim_msg)
|
||||
|
||||
if n < 3:
|
||||
if reasoning_only and streak >= 3:
|
||||
if agent._try_activate_fallback(reason=FailoverReason.incomplete_response):
|
||||
# The trigger may have consumed the turn budget; without a grace call the loop
|
||||
# exits before the fallback is ever asked.
|
||||
if api_call_count >= agent.max_iterations or agent.iteration_budget.remaining <= 0:
|
||||
agent._budget_grace_call = True
|
||||
agent._codex_incomplete_retries = 0
|
||||
agent._codex_reasoning_only_streak = 0
|
||||
if not agent.quiet_mode:
|
||||
agent._vprint(
|
||||
f"{agent.log_prefix}↻ Codex reasoning-only stall after {streak} attempts — "
|
||||
f"switching to fallback {agent.model} ({agent.provider})", diagnostic=True,
|
||||
)
|
||||
agent._emit_diagnostic_wait("↻ model stuck on internal reasoning — switching to fallback provider")
|
||||
agent._session_messages = messages
|
||||
return CODEX_FALLBACK_ACTIVATED
|
||||
# No fallback left: fall through to the terminal sentinel.
|
||||
elif n < 3 or reasoning_only:
|
||||
# A reasoning-only streak below 3 continues even once partials used up the aggregate
|
||||
# cap, so the mixed partial-then-stall variant reaches the ladder above.
|
||||
# If the interim has nothing the Responses converter will replay, a bare retry is
|
||||
# byte-identical; a replayable interim holding only a ``compaction`` checkpoint
|
||||
# ALSO re-sends identically. One bare retry, then always nudge.
|
||||
@@ -551,6 +587,7 @@ def continue_codex_incomplete(
|
||||
return None
|
||||
|
||||
agent._codex_incomplete_retries = 0
|
||||
agent._codex_reasoning_only_streak = 0
|
||||
agent._persist_session(messages, conversation_history)
|
||||
return partial_result(
|
||||
messages, api_call_count, "Codex response remained incomplete after 3 continuation attempts"
|
||||
|
||||
@@ -18,6 +18,7 @@ def _agent(max_tokens: int | None = 2000):
|
||||
agent.quiet_mode = True
|
||||
agent.log_prefix = ""
|
||||
agent._codex_incomplete_retries = 0
|
||||
agent._codex_reasoning_only_streak = 0
|
||||
agent._ephemeral_reasoning_off = False
|
||||
agent._ephemeral_max_output_tokens = None
|
||||
agent._build_assistant_message.side_effect = lambda msg, fr: {
|
||||
|
||||
108
tests/agent/test_codex_reasoning_only_streak.py
Normal file
108
tests/agent/test_codex_reasoning_only_streak.py
Normal file
@@ -0,0 +1,108 @@
|
||||
"""Codex Responses reasoning-only stall recovery (#67321).
|
||||
|
||||
Encrypted reasoning items replay byte-for-byte, so a bare continuation of a
|
||||
reasoning-only ``status=incomplete`` response repeats the stall. After three
|
||||
consecutive reasoning-only responses the turn must reach the configured
|
||||
fallback provider (with one bounded grace call when the trigger consumed the
|
||||
iteration budget) instead of ending on the internal incomplete sentinel; a
|
||||
visible partial resets the local streak; a cross-protocol fallback drops the
|
||||
Codex-only nudge from the wire.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import run_agent
|
||||
from agent.agent_runtime_helpers import drop_thinking_only_and_merge_users
|
||||
from agent.conversation_loop import _CODEX_INCOMPLETE_NUDGE
|
||||
from agent.error_classifier import FailoverReason
|
||||
from tests.agent.test_run_agent_codex_responses import (
|
||||
_build_agent,
|
||||
_codex_incomplete_message_response,
|
||||
_codex_message_response,
|
||||
_codex_reasoning_only_response,
|
||||
)
|
||||
|
||||
|
||||
def _spy_fallback(agent, monkeypatch):
|
||||
"""Record fallback activations; keep ``api_mode`` on codex_responses so the
|
||||
stub fallback answer still parses through the Codex path."""
|
||||
calls = []
|
||||
|
||||
def _fake(reason=None):
|
||||
calls.append(reason)
|
||||
return True
|
||||
|
||||
monkeypatch.setattr(agent, "_try_activate_fallback", _fake)
|
||||
return calls
|
||||
|
||||
|
||||
def _drive(agent, monkeypatch, responses):
|
||||
api_calls = {"n": 0}
|
||||
|
||||
def _fake_api_call(api_kwargs):
|
||||
api_calls["n"] += 1
|
||||
return responses.pop(0)
|
||||
|
||||
monkeypatch.setattr(agent, "_interruptible_api_call", _fake_api_call)
|
||||
return api_calls
|
||||
|
||||
|
||||
def test_reasoning_only_streak_reaches_fallback_with_one_grace_call(monkeypatch):
|
||||
agent = _build_agent(monkeypatch)
|
||||
agent.max_iterations = 3
|
||||
agent.iteration_budget = run_agent.IterationBudget(3)
|
||||
calls = _spy_fallback(agent, monkeypatch)
|
||||
api_calls = _drive(agent, monkeypatch, [
|
||||
_codex_reasoning_only_response(encrypted_content="enc_a"),
|
||||
_codex_reasoning_only_response(encrypted_content="enc_b"),
|
||||
_codex_reasoning_only_response(encrypted_content="enc_c"),
|
||||
_codex_message_response("Fallback answered."),
|
||||
])
|
||||
|
||||
result = agent.run_conversation("keep thinking")
|
||||
|
||||
assert result["completed"] is True
|
||||
assert result["final_response"] == "Fallback answered."
|
||||
assert calls == [FailoverReason.incomplete_response]
|
||||
# Three budgeted calls + exactly one grace call; the grace flag is consumed.
|
||||
assert api_calls["n"] == 4
|
||||
assert agent._budget_grace_call is False
|
||||
|
||||
|
||||
def test_visible_partial_resets_reasoning_only_streak(monkeypatch):
|
||||
agent = _build_agent(monkeypatch)
|
||||
agent.max_iterations = 6
|
||||
agent.iteration_budget = run_agent.IterationBudget(6)
|
||||
calls = _spy_fallback(agent, monkeypatch)
|
||||
_drive(agent, monkeypatch, [
|
||||
_codex_incomplete_message_response("Partial visible progress."),
|
||||
_codex_reasoning_only_response(encrypted_content="enc_a"),
|
||||
_codex_reasoning_only_response(encrypted_content="enc_b"),
|
||||
_codex_reasoning_only_response(encrypted_content="enc_c"),
|
||||
_codex_message_response("Recovered."),
|
||||
])
|
||||
|
||||
result = agent.run_conversation("partial then stall")
|
||||
|
||||
assert result["completed"] is True
|
||||
assert result["final_response"] == "Recovered."
|
||||
assert calls == [FailoverReason.incomplete_response]
|
||||
|
||||
|
||||
def test_cross_protocol_wire_drops_codex_nudge_and_keeps_alternation():
|
||||
messages = [
|
||||
{"role": "user", "content": "do it"},
|
||||
{"role": "assistant", "content": "", "tool_calls": [{"id": "c1", "type": "function",
|
||||
"function": {"name": "terminal", "arguments": "{}"}}]},
|
||||
{"role": "tool", "tool_call_id": "c1", "content": "ok"},
|
||||
{"role": "assistant", "content": "", "finish_reason": "incomplete",
|
||||
"codex_reasoning_items": [{"type": "reasoning", "id": "rs_1", "encrypted_content": "x"}]},
|
||||
{"role": "user", "content": _CODEX_INCOMPLETE_NUDGE},
|
||||
]
|
||||
|
||||
wire = drop_thinking_only_and_merge_users(
|
||||
messages, drop_codex_reasoning_items=True, drop_nudge_marker=_CODEX_INCOMPLETE_NUDGE,
|
||||
)
|
||||
|
||||
assert [m["role"] for m in wire] == ["user", "assistant", "tool"]
|
||||
assert not any(m.get("codex_reasoning_items") for m in wire)
|
||||
@@ -70,6 +70,7 @@ class TestFailoverReason:
|
||||
"reasoning_mandatory",
|
||||
"provider_policy_blocked",
|
||||
"content_policy_blocked",
|
||||
"incomplete_response",
|
||||
"thinking_signature", "long_context_tier",
|
||||
"oauth_long_context_beta_forbidden",
|
||||
"llama_cpp_grammar_pattern",
|
||||
|
||||
Reference in New Issue
Block a user