Files
hermes-agent/tests/gateway/test_api_server_turn_boundary.py
teknium1 d19963782b fix(api-server): anchor the Responses current turn on this turn's user row, not a history prefix match
`_response_messages_turn_start_index` located the current turn by matching
`result["messages"]` against `conversation_history + [user]` row by row. The
loop repairs host-fed history before its first call (consecutive assistant or
user rows merge, orphan tool results drop) and compaction rewrites it, so the
transcript legitimately stops sharing a prefix with the client history; the
match then returned 0 and the WHOLE transcript was treated as the current
turn: earlier turns' function_call / function_call_output items were replayed
as this turn's `output` / `run.completed` turn_messages, and the stored
`previous_response_id` chain grew by another copy of the history every turn.

Anchor on the loop's canonical current-turn user index instead
(`agent.turn_context.reanchor_current_turn_user_idx`: last user row that
says this turn's text, else the last user-originated row), and keep the
semantic prefix match only for suffix-only results without a user row
(mocked/legacy paths). The boundary logic moves to the topical sibling
`api_server_turn_boundary.py`; the routes mixin delegates.

The dedupe test that asserted "divergent transcript => append history +
transcript" encoded the duplication itself; it now asserts the fallback for
the case it actually exists for (a suffix-only result).

Fixes #89891
Co-authored-by: heyf <tonyheyifan@gmail.com>
2026-09-19 12:11:23 -07:00

67 lines
3.6 KiB
Python

"""The Responses current-turn boundary is anchored on this turn's user row, not on prefix
equality with the client history (#89891).
The loop repairs host-fed history before its first call (consecutive assistant rows merge,
stray tool results drop) and compaction rewrites it, so ``result["messages"]`` legitimately
stops sharing a prefix with ``conversation_history``. A prefix match then reported 0 and the
whole transcript became "the current turn": earlier ``function_call`` items were replayed as
this turn's output and the stored history doubled on every chained turn.
"""
from agent.agent_runtime_helpers import repair_message_sequence
from agent.message_metadata import append_message
from gateway.platforms.api_server import APIServerAdapter
class _Agent:
api_mode = "chat_completions"
session_id = "s"
def _tool_turn(messages, call_id, answer):
append_message(messages, {"role": "assistant", "content": "", "tool_calls": [
{"id": call_id, "type": "function", "function": {"name": "terminal", "arguments": "{}"}}]})
append_message(messages, {"role": "tool", "tool_call_id": call_id, "content": "ok"})
append_message(messages, {"role": "assistant", "content": answer, "finish_reason": "stop"})
def test_repaired_earlier_rows_keep_only_this_turn_as_output():
# A stateless client replays two consecutive assistant items; the loop merges them, so the
# transcript is one row shorter than the client history from index 1 on.
history = [{"role": "user", "content": "hi"}, {"role": "assistant", "content": "hello"},
{"role": "assistant", "content": "again"}]
prompt = "Q1: run the tool"
messages = list(history)
append_message(messages, {"role": "user", "content": prompt})
repair_message_sequence(_Agent(), messages)
_tool_turn(messages, "call-current", "done")
result = {"messages": messages}
start = APIServerAdapter._response_messages_turn_start_index(history, prompt, result)
items = APIServerAdapter._extract_output_items(result, start_index=start)
assert messages[start - 1]["content"] == prompt
assert [i["type"] for i in items] == ["function_call", "function_call_output", "message"]
# Stored as the agent's transcript, not client history + transcript again.
stored = APIServerAdapter._build_response_conversation_history(history, prompt, result, "done")
assert [m["role"] for m in stored] == ["user", "assistant", "user", "assistant", "tool", "assistant"]
def test_compacted_transcript_keeps_only_this_turn_as_output():
# Compaction replaced the oldest rows with a summary carrier and kept the recent
# tool-bearing turn verbatim: nothing before the current user row is this turn's output.
history = [{"role": "user", "content": "old ask"}, {"role": "assistant", "content": "old answer"},
{"role": "user", "content": "recent ask"}]
messages = [{"role": "user", "content": "[Compressed summary of earlier turns]"},
{"role": "assistant", "content": "Understood."},
{"role": "user", "content": "recent ask"}]
_tool_turn(messages, "call-old", "recent answer")
history = history + [dict(m) for m in messages[3:]]
append_message(messages, {"role": "user", "content": "Q2"})
_tool_turn(messages, "call-current", "answer 2")
result = {"messages": messages, "_compressed": True}
start = APIServerAdapter._response_messages_turn_start_index(history, "Q2", result)
items = APIServerAdapter._extract_output_items(result, start_index=start)
assert [i["type"] for i in items] == ["function_call", "function_call_output", "message"]
assert items[0]["call_id"] == "call-current"