From 170d616ca6959170edaa721de8c7fff132f431c3 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 16 Sep 2026 21:57:00 -0700 Subject: [PATCH] fix(anthropic): merged user turns keep each turn as its own text block _concat_content joined two string user contents into one "a\nb" string when _merge_consecutive_roles collapsed adjacent user turns for the Anthropic Messages wire. For a MoA aggregator on that wire, iteration 1 of a turn ends [user(task), user(guidance)] and was sent as user("task\n"), while iteration 2 replays user("task") alone, so the prompt-cache prefix diverged at the first user block and the #112358 collapse persisted there (the first-pass fix in #113175 only covered the OpenAI-compatible wire). Merged turns are now always a block list with each side's blocks intact (a string becomes one text block), matching what the list+list and list+str shapes already did. The task block is byte-identical to the standalone turn later iterations replay, a cache_control marker on it stays put, and the guidance follows as its own text block. Assistant merges are unaffected (assistant content is already a block list). Bedrock Converse and native Gemini already merge at block/part granularity, so the docs' Anthropic/Converse/Gemini fold caveat is replaced with the accurate statement and the byte-identical-extension claim no longer needs the OpenAI-compatible scope. Part of #112358 --- agent/anthropic_message_convert.py | 13 +++++----- agent/moa_loop.py | 6 ++--- tests/agent/test_anthropic_adapter.py | 5 ++-- .../test_moa_aggregator_prefix_stability.py | 25 +++++++++++++++++++ .../user-guide/features/mixture-of-agents.md | 2 +- 5 files changed, 39 insertions(+), 12 deletions(-) diff --git a/agent/anthropic_message_convert.py b/agent/anthropic_message_convert.py index b5931676f5..1397c0cf86 100644 --- a/agent/anthropic_message_convert.py +++ b/agent/anthropic_message_convert.py @@ -508,12 +508,13 @@ def _strip_orphaned_tool_blocks(result: List[Dict[str, Any]]) -> None: m["content"] = new_content if new_content else [_text_block("(tool result removed)")] -def _concat_content(prev: Any, curr: Any) -> Any: - """Merge two message contents: str+str joined by newline, list+list concatenated, mixed shapes - promoted to block lists.""" - if isinstance(prev, str) and isinstance(curr, str): - return prev + "\n" + curr - as_blocks = lambda c: [_text_block(c)] if isinstance(c, str) else c # noqa: E731 +def _concat_content(prev: Any, curr: Any) -> List[Any]: + """Merge two message contents into one block list, each side's blocks kept intact (a string + becomes its own text block). Strings are never joined: the first turn's bytes must equal what a + later request replays as a standalone turn, or the prompt-cache prefix diverges at that block + (MoA appends per-turn guidance after ``user(task)`` on iteration 1 and replays ``user(task)`` + alone on iteration 2 — #112358).""" + as_blocks = lambda c: [_text_block(c)] if isinstance(c, str) else list(c) # noqa: E731 return as_blocks(prev) + as_blocks(curr) diff --git a/agent/moa_loop.py b/agent/moa_loop.py index 0742b88871..21b747fac8 100644 --- a/agent/moa_loop.py +++ b/agent/moa_loop.py @@ -897,9 +897,9 @@ def _attach_reference_guidance(agg_messages: list[dict[str, Any]], guidance: str from the ``user(task)`` every later iteration replays, so the provider prefix cache collapsed to the system prompt on iteration 2 of every turn (#112358). Converters that require strict alternation (Anthropic Messages, Converse, native Gemini) merge - adjacent same-role turns, so there the task turn still varies on iteration 1; on the - OpenAI-compatible wire the request ends ``user(task), user(guidance)``, which a - chat template that enforces strict user/assistant alternation rejects. + the two user turns as SEPARATE content blocks, so the task block stays byte-stable + there too; on the OpenAI-compatible wire the request ends ``user(task), user(guidance)``, + which a chat template that enforces strict user/assistant alternation rejects. """ agg_messages.append({"role": "user", "content": guidance}) diff --git a/tests/agent/test_anthropic_adapter.py b/tests/agent/test_anthropic_adapter.py index 2bddea3bb1..5af08b5f36 100644 --- a/tests/agent/test_anthropic_adapter.py +++ b/tests/agent/test_anthropic_adapter.py @@ -1203,8 +1203,9 @@ class TestRoleAlternation: _, result = convert_messages_to_anthropic(messages) assert len(result) == 1 assert result[0]["role"] == "user" - assert "Hello" in result[0]["content"] - assert "World" in result[0]["content"] + # Each turn stays its own text block (never joined into one string), so the first turn's + # bytes match what a later request replays standalone and the cache prefix survives. + assert result[0]["content"] == [{"type": "text", "text": "Hello"}, {"type": "text", "text": "World"}] def test_preserves_proper_alternation(self): messages = [ diff --git a/tests/agent/test_moa_aggregator_prefix_stability.py b/tests/agent/test_moa_aggregator_prefix_stability.py index bb336ff67b..71d1616a04 100644 --- a/tests/agent/test_moa_aggregator_prefix_stability.py +++ b/tests/agent/test_moa_aggregator_prefix_stability.py @@ -8,6 +8,7 @@ the provider had just cached and the prompt cache collapsed to the system prompt from types import SimpleNamespace from agent import moa_loop +from agent.anthropic_message_convert import convert_messages_to_anthropic def test_attach_reference_guidance_never_mutates_the_trailing_user_turn(): @@ -47,3 +48,27 @@ def test_prepared_aggregator_requests_share_a_byte_identical_prefix_across_itera first, second = (c["messages"] for c in calls) assert second[: len(first) - 1] == first[:-1] assert second[-1] == first[-1] == {"role": "user", "content": guidance} + + +def test_anthropic_wire_keeps_the_task_block_byte_stable_with_guidance_as_its_own_block(): + """Anthropic Messages merges adjacent user turns; the merge must keep ``user(task)`` as an + unchanged text block and add the guidance AFTER it, never fold both into one string + (``"task\\n"``) — otherwise the prefix diverges at the task on iteration 1 of every turn.""" + guidance = "[Mixture of Agents reference context]\nadvice" + history = [{"role": "system", "content": "sys"}, {"role": "user", "content": "task"}] + iteration_1 = [*history] + moa_loop._attach_reference_guidance(iteration_1, guidance) + iteration_2 = [ + *history, + {"role": "assistant", "content": "", "tool_calls": [ + {"id": "1", "type": "function", "function": {"name": "lookup", "arguments": "{}"}}]}, + {"role": "tool", "tool_call_id": "1", "content": "result"}, + ] + moa_loop._attach_reference_guidance(iteration_2, guidance) + + _, first = convert_messages_to_anthropic(iteration_1) + _, second = convert_messages_to_anthropic(iteration_2) + + assert first[0]["content"] == [{"type": "text", "text": "task"}, {"type": "text", "text": guidance}] + assert second[0]["content"] == "task" + assert first[0]["content"][0]["text"] == second[0]["content"] diff --git a/website/docs/user-guide/features/mixture-of-agents.md b/website/docs/user-guide/features/mixture-of-agents.md index 25b73724a1..72ae9e465b 100644 --- a/website/docs/user-guide/features/mixture-of-agents.md +++ b/website/docs/user-guide/features/mixture-of-agents.md @@ -243,7 +243,7 @@ MoA is built so the **main conversation's prompt cache is never broken**. Select Both internal call types cache normally: - **Reference models** receive a trimmed, deterministic view of the conversation (system prompt and tool transcript stripped — see the loop above). Because that view is a stable function of the stable history, a reference model's prompt prefix repeats across iterations and caches normally. References are short advisory calls with no tools. -- **The aggregator** is the acting model. The reference outputs are appended as their *own* trailing user message of private guidance — never merged into your message. Because that block sits at the tail — below the entire stable prefix (system prompt + your message + prior tool history) — it does not invalidate any cached prefix: on OpenAI-compatible aggregators every request in a tool loop is a byte-identical extension of the previous one minus its guidance block, so the aggregator gets a cache hit on everything above the injection and only the freshly appended tail is new. That is exactly how every normal turn behaves, where each new user message is also uncached tail tokens. (Aggregators on the Anthropic Messages, Bedrock Converse or native Gemini wire fold the guidance into your message on the first iteration of a turn — those converters merge adjacent user turns — so there the cache hit still starts above your message.) +- **The aggregator** is the acting model. The reference outputs are appended as their *own* trailing user message of private guidance — never merged into your message. Because that block sits at the tail — below the entire stable prefix (system prompt + your message + prior tool history) — it does not invalidate any cached prefix: every request in a tool loop is a byte-identical extension of the previous one minus its guidance block, so the aggregator gets a cache hit on everything above the injection and only the freshly appended tail is new. That is exactly how every normal turn behaves, where each new user message is also uncached tail tokens. Aggregators on the Anthropic Messages, Bedrock Converse or native Gemini wire merge the two adjacent user turns into one message, but as separate content blocks: your message's block is byte-identical to the one later iterations replay, and the guidance block follows it, so the cached prefix still runs through your message. So MoA does not sacrifice prompt caching on either call type. Its only real cost is the extra reference calls (once per user turn with the default `fanout`) — you pay for multiple model perspectives, not for broken caches. The long-lived conversation prefix shared with the rest of Hermes is fully intact.