fix: hand Codex reasoning-only stalls to the fallback provider instead of the incomplete sentinel

Three consecutive Codex Responses answers that carry only (encrypted) reasoning —
no visible text, no tool call — used to exhaust the 3-continuation budget and end
the turn on "Codex response remained incomplete after 3 continuation attempts",
never touching configured fallback_providers (#67321). Encrypted reasoning items
replay byte-for-byte, so a bare retry deterministically repeats the stall.

- Track a per-turn `_codex_reasoning_only_streak` apart from the aggregate
  `_codex_incomplete_retries`: a visible partial resets the streak, so the mixed
  partial-then-stall variant still reaches its own recovery threshold while the
  turn-wide iteration budget stays the hard bound.
- At streak 3, `continue_codex_incomplete` activates the next fallback with the
  semantic `FailoverReason.incomplete_response`, grants exactly one grace call when
  the trigger consumed the last iteration, and returns `CODEX_FALLBACK_ACTIVATED`;
  the intake re-syncs the Model:/Provider: identity on the system prompt.
- Off the Codex wire the synthetic continuation nudge is stripped alongside the
  opaque replay state (`drop_nudge_marker`) so the Chat Completions payload keeps
  valid role ordering and no Codex-only control text.
- No fallback configured: unchanged terminal sentinel, still bounded at 3 calls.

Ported from PR #67336 by @PRATHAMESH75 onto the decomposed agent/turn_*.py siblings.
This commit is contained in:
PRATHAMESH75
2026-09-19 00:04:42 -07:00
committed by teknium1
parent afaa53e5fe
commit f7567a62af
9 changed files with 188 additions and 14 deletions

View File

@@ -1016,16 +1016,23 @@ _UNMERGEABLE = object()
def drop_thinking_only_and_merge_users(
messages: List[Dict[str, Any]], *, drop_codex_reasoning_items: bool = True
messages: List[Dict[str, Any]], *, drop_codex_reasoning_items: bool = True,
drop_nudge_marker: Optional[str] = None,
) -> List[Dict[str, Any]]:
"""Drop thinking-only assistant turns and merge adjacent user messages left behind, on the
per-call ``api_messages`` copy only (``agent.messages`` is never mutated). Drop-and-merge
(not stub text) keeps history honest and preserves role alternation."""
(not stub text) keeps history honest and preserves role alternation.
``drop_nudge_marker`` (#67321): user rows equal to the marker — the synthetic Codex
continuation nudge — are dropped too once the turn has crossed to a non-Codex provider;
doing it in this pass keeps alternation valid when the nudge sat between dropped
reasoning-only interims and a tool result rather than next to the user's message."""
if not messages:
return messages
kept = [
m for m in messages
if not _ra().AIAgent._is_thinking_only_assistant(m, drop_codex_reasoning_items=drop_codex_reasoning_items)
if not (drop_nudge_marker is not None and m.get("role") == "user" and m.get("content") == drop_nudge_marker)
and not _ra().AIAgent._is_thinking_only_assistant(m, drop_codex_reasoning_items=drop_codex_reasoning_items)
]
dropped = len(messages) - len(kept)
merged: List[Dict[str, Any]] = []

View File

@@ -48,7 +48,8 @@ class FailoverReason(enum.Enum):
image_corrupt = "image_corrupt" # Provider can't decode image bytes — strip and retry (shrinking won't help)
model_not_found = "model_not_found" # 404 or invalid model — fallback to different model
provider_policy_blocked = "provider_policy_blocked" # Aggregator account data/privacy policy excluded the only endpoint
content_policy_blocked = "content_policy_blocked" # Provider safety filter rejected this prompt — don't retry unchanged
content_policy_blocked = "content_policy_blocked" # Provider safety filter rejected this prompt — deterministic per-request, don't retry unchanged
incomplete_response = "incomplete_response" # Codex/Responses turn stuck emitting reasoning only (no answer, no tool call) after replay + nudge — hand to a different provider
format_error = "format_error" # 400 bad request — abort or strip + retry
role_alternation = "role_alternation" # Strict chat template rejected adjacent same-role messages — merge them for this destination and retry
invalid_encrypted_content = "invalid_encrypted_content" # Responses replay blob rejected — strip replay state and retry

View File

@@ -507,6 +507,9 @@ def _bind_turn_identity(
_PER_TURN_RESET_STATE: Tuple[Tuple[str, Any], ...] = (
("_invalid_tool_retries", 0), ("_invalid_json_retries", 0), ("_empty_content_retries", 0),
("_incomplete_scratchpad_retries", 0), ("_codex_incomplete_retries", 0),
# Consecutive Codex reasoning-only (no answer, no tool call) responses, kept apart from
# the aggregate incomplete count so a visible partial resets it (#67321).
("_codex_reasoning_only_streak", 0),
("_thinking_prefill_retries", 0), ("_post_tool_empty_retried", False),
("_last_content_with_tools", None), ("_last_content_tools_all_housekeeping", False),
("_mute_post_response", False), ("_unicode_sanitization_passes", 0),

View File

@@ -112,8 +112,8 @@ def assemble_api_request(
are injected only after whitespace normalization, the orphan sweep, thinking-only drop /
user merge and surrogate stripping, so the same row's bytes never vary across turns."""
from agent.conversation_loop import (
_apply_context_engine_selection, _canonicalize_api_tool_calls, _clone_message_for_send,
_midturn_request_pressure_tokens, _pressure_with_real_floor,
_CODEX_INCOMPLETE_NUDGE, _apply_context_engine_selection, _canonicalize_api_tool_calls,
_clone_message_for_send, _midturn_request_pressure_tokens, _pressure_with_real_floor,
)
from agent.model_metadata import estimate_messages_tokens_rough
@@ -167,8 +167,13 @@ def assemble_api_request(
# Drop thinking-only assistant turns + merge adjacent users, API copy only:
# Anthropic-style backends 400 on a trailing `thinking` block; history keeps it.
# Off the Codex wire (e.g. after a reasoning-only stall fell over to a Chat Completions
# provider, #67321) the synthetic continuation nudge is Codex-only control text: drop it
# alongside the opaque replay state.
_cross_protocol = agent.api_mode != "codex_responses"
api_messages = agent._drop_thinking_only_and_merge_users(
api_messages, drop_codex_reasoning_items=agent.api_mode != "codex_responses"
api_messages, drop_codex_reasoning_items=_cross_protocol,
drop_nudge_marker=_CODEX_INCOMPLETE_NUDGE if _cross_protocol else None,
)
# Normalize whitespace and tool-call JSON for bit-perfect prefixes across turns

View File

@@ -14,7 +14,9 @@ from typing import Any, Dict, Optional
from agent.provider_projection import splice_provider_projection
from agent.trajectory import has_incomplete_scratchpad
from agent.turn_truncation import continue_codex_incomplete, normalize_response_for_agent, partial_result
from agent.turn_truncation import (
CODEX_FALLBACK_ACTIVATED, continue_codex_incomplete, normalize_response_for_agent, partial_result,
)
logger = logging.getLogger("agent.conversation_loop")
@@ -25,12 +27,14 @@ _REASONING_TAG_RE = re.compile(r'</?(?:REASONING_SCRATCHPAD|think|reasoning)>')
class ResponseIntakeVerdict:
"""``action``: ``"fallthrough"`` (process ``assistant_message``), ``"continue"`` (retry the
iteration: incomplete scratchpad / Codex continuation) or ``"return"`` (``result`` is the
turn's result dict). ``assistant_message``/``finish_reason`` are the normalized outputs."""
turn's result dict). ``assistant_message``/``finish_reason`` are the normalized outputs;
``active_system_prompt`` is rebound after a Codex reasoning-only fallover (#67321)."""
action: str
assistant_message: Any
finish_reason: Any
result: Optional[Dict[str, Any]] = None
active_system_prompt: Any = None
def _coerce_content_text(raw: Any) -> str:
@@ -115,7 +119,7 @@ def _relay_thinking(agent: Any, content: str) -> None:
def normalize_model_response(
agent: Any, *, response: Any, messages: Any, api_messages: Any, conversation_history: Any,
api_call_count: Any, api_duration: Any, api_start_time: Any, api_request_id: Any,
effective_task_id: Any, turn_id: Any,
effective_task_id: Any, turn_id: Any, active_system_prompt: Any = None,
) -> ResponseIntakeVerdict:
"""Normalize ``response`` into ``assistant_message`` (str content, never dict/list) and run
the post-response hooks and continuation guards, in the original order."""
@@ -125,7 +129,7 @@ def normalize_model_response(
def _verdict(action: str, result: Optional[Dict[str, Any]] = None) -> ResponseIntakeVerdict:
return ResponseIntakeVerdict(
action=action, assistant_message=assistant_message, finish_reason=finish_reason,
result=result,
result=result, active_system_prompt=active_system_prompt,
)
if assistant_message.content is not None and not isinstance(assistant_message.content, str):
@@ -175,9 +179,16 @@ def normalize_model_response(
conversation_history=conversation_history, api_call_count=api_call_count,
response=response,
)
if _codex_result is CODEX_FALLBACK_ACTIVATED:
# The failover rewrote the Model:/Provider: identity on the cached system prompt;
# rebind it so the next iteration's request is rebuilt with the new identity.
from agent.conversation_loop import _sync_failover_system_message
active_system_prompt = _sync_failover_system_message(agent, api_messages, active_system_prompt)
return _verdict("continue")
if _codex_result is not None:
return _verdict("return", _codex_result)
return _verdict("continue")
if hasattr(agent, "_codex_incomplete_retries"):
agent._codex_incomplete_retries = 0
agent._codex_reasoning_only_streak = 0
return _verdict("fallthrough")

View File

@@ -447,11 +447,15 @@ _CODEX_REPLAY_KEYS = (
"codex_reasoning_items", "codex_message_items",
)
# Third return value of ``continue_codex_incomplete``: the reasoning-only stall was handed to a
# fallback provider — the caller re-syncs the system prompt identity and continues the turn.
CODEX_FALLBACK_ACTIVATED = "codex_fallback_activated"
def continue_codex_incomplete(
agent: Any, assistant_message: Any, finish_reason: str, *, messages: List[Dict[str, Any]],
conversation_history: Any, api_call_count: int, response: Any = None,
) -> Optional[Dict[str, Any]]:
) -> Optional[Any]:
"""Codex Responses ``status=incomplete`` continuation (max 3 per turn).
Appends the interim assistant message (deduped on visible content only — opaque
@@ -459,7 +463,17 @@ def continue_codex_incomplete(
overwritten, because the earlier response holds the only native-compaction
checkpoint) and, when a bare retry would be byte-identical, a user-role nudge — only
after an assistant row, to preserve role alternation. Returns ``None`` to continue
the turn loop, or the terminal ``partial`` result once retries are exhausted.
the turn loop, ``CODEX_FALLBACK_ACTIVATED`` when a reasoning-only stall was handed to
the next fallback provider, or the terminal ``partial`` result once retries are exhausted.
Reasoning-only stall ladder (#67321): a response with neither visible text nor a tool
call advances ``_codex_reasoning_only_streak`` (a visible partial resets it; the aggregate
``_codex_incomplete_retries`` stays the cap for partials). Encrypted reasoning replays
byte-for-byte, so after replay (1) and nudge (2) the third consecutive reasoning-only
response goes to the configured fallback with the semantic ``incomplete_response`` reason
instead of ending on the sentinel; when that response consumed the last iteration the
fallback gets exactly one grace call (``_budget_grace_call`` is consumed by the next
iteration, and the streak restarts from 0, so a second grace call is unreachable).
When ``response`` hit ``max_output_tokens`` with no visible text (reasoning ate the
whole budget), the next attempt goes out with reasoning off and a doubled output
@@ -477,6 +491,9 @@ def continue_codex_incomplete(
interim_has_reasoning = isinstance(_reasoning, str) and bool(_reasoning.strip())
interim_has_codex_reasoning = bool(interim_msg.get("codex_reasoning_items"))
interim_has_codex_message_items = bool(interim_msg.get("codex_message_items"))
reasoning_only = not interim_has_content and not getattr(assistant_message, "tool_calls", None)
agent._codex_reasoning_only_streak = agent._codex_reasoning_only_streak + 1 if reasoning_only else 0
streak = agent._codex_reasoning_only_streak
if interim_has_content or interim_has_reasoning or interim_has_codex_reasoning or interim_has_codex_message_items:
last_msg = messages[-1] if messages else None
@@ -510,7 +527,26 @@ def continue_codex_incomplete(
append_message(messages, interim_msg)
agent._emit_interim_assistant_message(interim_msg)
if n < 3:
if reasoning_only and streak >= 3:
if agent._try_activate_fallback(reason=FailoverReason.incomplete_response):
# The trigger may have consumed the turn budget; without a grace call the loop
# exits before the fallback is ever asked.
if api_call_count >= agent.max_iterations or agent.iteration_budget.remaining <= 0:
agent._budget_grace_call = True
agent._codex_incomplete_retries = 0
agent._codex_reasoning_only_streak = 0
if not agent.quiet_mode:
agent._vprint(
f"{agent.log_prefix}↻ Codex reasoning-only stall after {streak} attempts — "
f"switching to fallback {agent.model} ({agent.provider})", diagnostic=True,
)
agent._emit_diagnostic_wait("↻ model stuck on internal reasoning — switching to fallback provider")
agent._session_messages = messages
return CODEX_FALLBACK_ACTIVATED
# No fallback left: fall through to the terminal sentinel.
elif n < 3 or reasoning_only:
# A reasoning-only streak below 3 continues even once partials used up the aggregate
# cap, so the mixed partial-then-stall variant reaches the ladder above.
# If the interim has nothing the Responses converter will replay, a bare retry is
# byte-identical; a replayable interim holding only a ``compaction`` checkpoint
# ALSO re-sends identically. One bare retry, then always nudge.
@@ -551,6 +587,7 @@ def continue_codex_incomplete(
return None
agent._codex_incomplete_retries = 0
agent._codex_reasoning_only_streak = 0
agent._persist_session(messages, conversation_history)
return partial_result(
messages, api_call_count, "Codex response remained incomplete after 3 continuation attempts"

View File

@@ -18,6 +18,7 @@ def _agent(max_tokens: int | None = 2000):
agent.quiet_mode = True
agent.log_prefix = ""
agent._codex_incomplete_retries = 0
agent._codex_reasoning_only_streak = 0
agent._ephemeral_reasoning_off = False
agent._ephemeral_max_output_tokens = None
agent._build_assistant_message.side_effect = lambda msg, fr: {

View File

@@ -0,0 +1,108 @@
"""Codex Responses reasoning-only stall recovery (#67321).
Encrypted reasoning items replay byte-for-byte, so a bare continuation of a
reasoning-only ``status=incomplete`` response repeats the stall. After three
consecutive reasoning-only responses the turn must reach the configured
fallback provider (with one bounded grace call when the trigger consumed the
iteration budget) instead of ending on the internal incomplete sentinel; a
visible partial resets the local streak; a cross-protocol fallback drops the
Codex-only nudge from the wire.
"""
from __future__ import annotations
import run_agent
from agent.agent_runtime_helpers import drop_thinking_only_and_merge_users
from agent.conversation_loop import _CODEX_INCOMPLETE_NUDGE
from agent.error_classifier import FailoverReason
from tests.agent.test_run_agent_codex_responses import (
_build_agent,
_codex_incomplete_message_response,
_codex_message_response,
_codex_reasoning_only_response,
)
def _spy_fallback(agent, monkeypatch):
"""Record fallback activations; keep ``api_mode`` on codex_responses so the
stub fallback answer still parses through the Codex path."""
calls = []
def _fake(reason=None):
calls.append(reason)
return True
monkeypatch.setattr(agent, "_try_activate_fallback", _fake)
return calls
def _drive(agent, monkeypatch, responses):
api_calls = {"n": 0}
def _fake_api_call(api_kwargs):
api_calls["n"] += 1
return responses.pop(0)
monkeypatch.setattr(agent, "_interruptible_api_call", _fake_api_call)
return api_calls
def test_reasoning_only_streak_reaches_fallback_with_one_grace_call(monkeypatch):
agent = _build_agent(monkeypatch)
agent.max_iterations = 3
agent.iteration_budget = run_agent.IterationBudget(3)
calls = _spy_fallback(agent, monkeypatch)
api_calls = _drive(agent, monkeypatch, [
_codex_reasoning_only_response(encrypted_content="enc_a"),
_codex_reasoning_only_response(encrypted_content="enc_b"),
_codex_reasoning_only_response(encrypted_content="enc_c"),
_codex_message_response("Fallback answered."),
])
result = agent.run_conversation("keep thinking")
assert result["completed"] is True
assert result["final_response"] == "Fallback answered."
assert calls == [FailoverReason.incomplete_response]
# Three budgeted calls + exactly one grace call; the grace flag is consumed.
assert api_calls["n"] == 4
assert agent._budget_grace_call is False
def test_visible_partial_resets_reasoning_only_streak(monkeypatch):
agent = _build_agent(monkeypatch)
agent.max_iterations = 6
agent.iteration_budget = run_agent.IterationBudget(6)
calls = _spy_fallback(agent, monkeypatch)
_drive(agent, monkeypatch, [
_codex_incomplete_message_response("Partial visible progress."),
_codex_reasoning_only_response(encrypted_content="enc_a"),
_codex_reasoning_only_response(encrypted_content="enc_b"),
_codex_reasoning_only_response(encrypted_content="enc_c"),
_codex_message_response("Recovered."),
])
result = agent.run_conversation("partial then stall")
assert result["completed"] is True
assert result["final_response"] == "Recovered."
assert calls == [FailoverReason.incomplete_response]
def test_cross_protocol_wire_drops_codex_nudge_and_keeps_alternation():
messages = [
{"role": "user", "content": "do it"},
{"role": "assistant", "content": "", "tool_calls": [{"id": "c1", "type": "function",
"function": {"name": "terminal", "arguments": "{}"}}]},
{"role": "tool", "tool_call_id": "c1", "content": "ok"},
{"role": "assistant", "content": "", "finish_reason": "incomplete",
"codex_reasoning_items": [{"type": "reasoning", "id": "rs_1", "encrypted_content": "x"}]},
{"role": "user", "content": _CODEX_INCOMPLETE_NUDGE},
]
wire = drop_thinking_only_and_merge_users(
messages, drop_codex_reasoning_items=True, drop_nudge_marker=_CODEX_INCOMPLETE_NUDGE,
)
assert [m["role"] for m in wire] == ["user", "assistant", "tool"]
assert not any(m.get("codex_reasoning_items") for m in wire)

View File

@@ -70,6 +70,7 @@ class TestFailoverReason:
"reasoning_mandatory",
"provider_policy_blocked",
"content_policy_blocked",
"incomplete_response",
"thinking_signature", "long_context_tier",
"oauth_long_context_beta_forbidden",
"llama_cpp_grammar_pattern",