fix(agent): retry a truncated tool call with a boosted budget on the Responses wire
A ``status=incomplete`` (max_output_tokens) Responses reply whose function_call item was cut mid-arguments used to end the turn on the first API call: ``_codex_finish_reason`` mapped it to ``incomplete``, the normalizer produced ``tool_calls``, and the tool-argument validator refused the half-written JSON as "reply was cut off". Chat modes retry that same shape up to 4x with a 2x/4x/8x/16x ``max_tokens`` boost; ``codex_responses`` was outside ``_CONTINUABLE_MODES`` so it never did. ``_derive_finish_reason`` now routes an incomplete Responses reply that carries a tool call to ``finish_reason == "length"``, and ``codex_responses`` joins ``_CONTINUABLE_MODES`` so ``recover_from_truncation`` re-issues the same call with the boosted output cap (reasoning kept, no interim row, no nudge). Text-only incompletes still return ``incomplete`` and stay on the Codex continuation path, so the length text branch never double-continues them. Fixes #91770 Co-authored-by: StanleyStetson <24758295+StanleyStetson@users.noreply.github.com>
This commit is contained in:
@@ -65,7 +65,13 @@ def _codex_finish_reason(response: Any) -> str:
|
||||
|
||||
def _derive_finish_reason(agent: Any, response: Any, messages: Any) -> str:
|
||||
if agent.api_mode == "codex_responses":
|
||||
return _codex_finish_reason(response)
|
||||
finish_reason = _codex_finish_reason(response)
|
||||
# A function_call cut off by max_output_tokens is not a text turn to continue: the
|
||||
# Codex incomplete path would replay the partial and re-hit the same cap. Route it
|
||||
# to the length path so the same call is retried with a boosted budget (#91770).
|
||||
if finish_reason == "incomplete" and agent._get_transport().normalize_response(response).tool_calls:
|
||||
return "length"
|
||||
return finish_reason
|
||||
transport = agent._get_transport()
|
||||
if agent.api_mode == "anthropic_messages":
|
||||
return transport.response_finish_reason(response)
|
||||
|
||||
@@ -26,7 +26,10 @@ from hermes_constants import PARTIAL_STREAM_STUB_ID
|
||||
|
||||
logger = logging.getLogger("agent.conversation_loop")
|
||||
|
||||
_CONTINUABLE_MODES = {"chat_completions", "bedrock_converse", "anthropic_messages"}
|
||||
# codex_responses only reaches ``finish_reason == "length"`` for a tool call cut off by
|
||||
# max_output_tokens (turn_response_check.py::_derive_finish_reason); text truncation stays on
|
||||
# the Codex incomplete continuation, so the text branch below never double-continues it.
|
||||
_CONTINUABLE_MODES = {"chat_completions", "bedrock_converse", "anthropic_messages", "codex_responses"}
|
||||
_THINK_TAG_RE = re.compile(r'<(?:think|thinking|reasoning|REASONING_SCRATCHPAD)[^>]*>', re.IGNORECASE)
|
||||
_TRUNCATED_FINAL = site_copy("truncated")
|
||||
_FIRST_TRUNCATED_FINAL = _TRUNCATED_FINAL
|
||||
|
||||
@@ -2842,3 +2842,57 @@ def test_run_codex_stream_retired_request_stops_firing_callbacks(monkeypatch):
|
||||
|
||||
assert streamed == ["keep"]
|
||||
assert "DROPPED" not in streamed
|
||||
|
||||
|
||||
def _codex_truncated_tool_call_response():
|
||||
"""``status=incomplete`` (max_output_tokens) whose function_call item was cut mid-arguments
|
||||
and settled as ``completed`` — the self-hosted /v1/responses shape from #91770."""
|
||||
return SimpleNamespace(
|
||||
output=[
|
||||
SimpleNamespace(
|
||||
type="function_call", id="fc_1", call_id="call_1", name="terminal",
|
||||
arguments='{"command": "echo hel', status="completed",
|
||||
)
|
||||
],
|
||||
usage=SimpleNamespace(input_tokens=50, output_tokens=8, total_tokens=58),
|
||||
status="incomplete",
|
||||
incomplete_details=SimpleNamespace(reason="max_output_tokens"),
|
||||
model="gpt-5.4",
|
||||
)
|
||||
|
||||
|
||||
def test_codex_truncated_tool_call_is_retried_with_boosted_output_budget(monkeypatch):
|
||||
"""A tool call cut off by max_output_tokens on the Responses wire gets the same
|
||||
budget-boost retry as chat modes instead of a refused partial turn (#91770)."""
|
||||
agent = _build_copilot_agent(monkeypatch)
|
||||
agent.max_tokens = 1000
|
||||
responses = [_codex_truncated_tool_call_response(), _codex_message_response("Done.")]
|
||||
seen_caps: list = []
|
||||
|
||||
def _fake_call(api_kwargs):
|
||||
seen_caps.append(api_kwargs.get("max_output_tokens"))
|
||||
return responses.pop(0)
|
||||
|
||||
monkeypatch.setattr(agent, "_interruptible_api_call", _fake_call)
|
||||
|
||||
result = agent.run_conversation("run it")
|
||||
|
||||
assert result["completed"] is True
|
||||
assert result["final_response"] == "Done."
|
||||
assert seen_caps == [1000, 2000]
|
||||
# The retry re-issues the same call: no interim assistant row, no continuation nudge.
|
||||
assert [m["role"] for m in result["messages"] if m["role"] != "system"] == ["user", "assistant"]
|
||||
|
||||
|
||||
def test_codex_text_only_max_output_incomplete_keeps_codex_continuation(monkeypatch):
|
||||
"""Text truncation is not rerouted: it stays on the Codex incomplete continuation and
|
||||
never takes the length path's nudge (no double continuation, #91770)."""
|
||||
agent = _build_copilot_agent(monkeypatch)
|
||||
responses = [_codex_max_output_incomplete_response("Partial"), _codex_message_response("rest.")]
|
||||
monkeypatch.setattr(agent, "_interruptible_api_call", lambda api_kwargs: responses.pop(0))
|
||||
|
||||
result = agent.run_conversation("write")
|
||||
|
||||
assert result["completed"] is True
|
||||
assert not any(m.get("_length_continuation_nudge") for m in result["messages"])
|
||||
assert any(m.get("finish_reason") == "incomplete" for m in result["messages"] if m["role"] == "assistant")
|
||||
|
||||
Reference in New Issue
Block a user