fix(agent): retry a truncated tool call with a boosted budget on the Responses wire

A ``status=incomplete`` (max_output_tokens) Responses reply whose function_call item was
cut mid-arguments used to end the turn on the first API call: ``_codex_finish_reason``
mapped it to ``incomplete``, the normalizer produced ``tool_calls``, and the tool-argument
validator refused the half-written JSON as "reply was cut off". Chat modes retry that same
shape up to 4x with a 2x/4x/8x/16x ``max_tokens`` boost; ``codex_responses`` was outside
``_CONTINUABLE_MODES`` so it never did.

``_derive_finish_reason`` now routes an incomplete Responses reply that carries a tool call
to ``finish_reason == "length"``, and ``codex_responses`` joins ``_CONTINUABLE_MODES`` so
``recover_from_truncation`` re-issues the same call with the boosted output cap (reasoning
kept, no interim row, no nudge). Text-only incompletes still return ``incomplete`` and stay
on the Codex continuation path, so the length text branch never double-continues them.

Fixes #91770
Co-authored-by: StanleyStetson <24758295+StanleyStetson@users.noreply.github.com>
This commit is contained in:
teknium1
2026-09-19 00:47:21 -07:00
committed by Teknium
parent 52705fba6c
commit b0952eac0c
3 changed files with 65 additions and 2 deletions

View File

@@ -65,7 +65,13 @@ def _codex_finish_reason(response: Any) -> str:
def _derive_finish_reason(agent: Any, response: Any, messages: Any) -> str:
if agent.api_mode == "codex_responses":
return _codex_finish_reason(response)
finish_reason = _codex_finish_reason(response)
# A function_call cut off by max_output_tokens is not a text turn to continue: the
# Codex incomplete path would replay the partial and re-hit the same cap. Route it
# to the length path so the same call is retried with a boosted budget (#91770).
if finish_reason == "incomplete" and agent._get_transport().normalize_response(response).tool_calls:
return "length"
return finish_reason
transport = agent._get_transport()
if agent.api_mode == "anthropic_messages":
return transport.response_finish_reason(response)

View File

@@ -26,7 +26,10 @@ from hermes_constants import PARTIAL_STREAM_STUB_ID
logger = logging.getLogger("agent.conversation_loop")
_CONTINUABLE_MODES = {"chat_completions", "bedrock_converse", "anthropic_messages"}
# codex_responses only reaches ``finish_reason == "length"`` for a tool call cut off by
# max_output_tokens (turn_response_check.py::_derive_finish_reason); text truncation stays on
# the Codex incomplete continuation, so the text branch below never double-continues it.
_CONTINUABLE_MODES = {"chat_completions", "bedrock_converse", "anthropic_messages", "codex_responses"}
_THINK_TAG_RE = re.compile(r'<(?:think|thinking|reasoning|REASONING_SCRATCHPAD)[^>]*>', re.IGNORECASE)
_TRUNCATED_FINAL = site_copy("truncated")
_FIRST_TRUNCATED_FINAL = _TRUNCATED_FINAL

View File

@@ -2842,3 +2842,57 @@ def test_run_codex_stream_retired_request_stops_firing_callbacks(monkeypatch):
assert streamed == ["keep"]
assert "DROPPED" not in streamed
def _codex_truncated_tool_call_response():
"""``status=incomplete`` (max_output_tokens) whose function_call item was cut mid-arguments
and settled as ``completed`` — the self-hosted /v1/responses shape from #91770."""
return SimpleNamespace(
output=[
SimpleNamespace(
type="function_call", id="fc_1", call_id="call_1", name="terminal",
arguments='{"command": "echo hel', status="completed",
)
],
usage=SimpleNamespace(input_tokens=50, output_tokens=8, total_tokens=58),
status="incomplete",
incomplete_details=SimpleNamespace(reason="max_output_tokens"),
model="gpt-5.4",
)
def test_codex_truncated_tool_call_is_retried_with_boosted_output_budget(monkeypatch):
"""A tool call cut off by max_output_tokens on the Responses wire gets the same
budget-boost retry as chat modes instead of a refused partial turn (#91770)."""
agent = _build_copilot_agent(monkeypatch)
agent.max_tokens = 1000
responses = [_codex_truncated_tool_call_response(), _codex_message_response("Done.")]
seen_caps: list = []
def _fake_call(api_kwargs):
seen_caps.append(api_kwargs.get("max_output_tokens"))
return responses.pop(0)
monkeypatch.setattr(agent, "_interruptible_api_call", _fake_call)
result = agent.run_conversation("run it")
assert result["completed"] is True
assert result["final_response"] == "Done."
assert seen_caps == [1000, 2000]
# The retry re-issues the same call: no interim assistant row, no continuation nudge.
assert [m["role"] for m in result["messages"] if m["role"] != "system"] == ["user", "assistant"]
def test_codex_text_only_max_output_incomplete_keeps_codex_continuation(monkeypatch):
"""Text truncation is not rerouted: it stays on the Codex incomplete continuation and
never takes the length path's nudge (no double continuation, #91770)."""
agent = _build_copilot_agent(monkeypatch)
responses = [_codex_max_output_incomplete_response("Partial"), _codex_message_response("rest.")]
monkeypatch.setattr(agent, "_interruptible_api_call", lambda api_kwargs: responses.pop(0))
result = agent.run_conversation("write")
assert result["completed"] is True
assert not any(m.get("_length_continuation_nudge") for m in result["messages"])
assert any(m.get("finish_reason") == "incomplete" for m in result["messages"] if m["role"] == "assistant")