From 4e7e103ba6ddffc4e92c3f965c40c04ba46287af Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 5 Aug 2026 17:14:16 -0700 Subject: [PATCH] fix(gemini): interpose placeholder model turn between tool result and user text MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Port from google-gemini/gemini-cli#28700: when an interrupted/failed turn leaves history ending on an unanswered tool result and the user sends a new message, fusing the two into one Gemini user content makes the model read the trailing text as a continuation of the tool result — it 'finishes your sentence' instead of answering. Builds on #68863 (@rille111), which split the mixed functionResponse/text merge but emitted two consecutive user contents — a shape Gemini's alternation contract rejects with HTTP 400 on other request paths (#55125). This follow-up interposes gemini-cli's INTERRUPTED_RESPONSE_PLACEHOLDER model turn between the split contents so the request stays alternation-valid while the user's message remains a turn of its own. --- agent/gemini_native_adapter.py | 39 +++++++++++++++++++---- tests/agent/test_gemini_native_adapter.py | 13 ++++++-- 2 files changed, 43 insertions(+), 9 deletions(-) diff --git a/agent/gemini_native_adapter.py b/agent/gemini_native_adapter.py index f1547759b5..0fb31b7aa3 100644 --- a/agent/gemini_native_adapter.py +++ b/agent/gemini_native_adapter.py @@ -289,6 +289,16 @@ def _tool_call_extra_signature(tool_call: Dict[str, Any]) -> Optional[str]: return None +# Stands in for a model turn that never arrived (stream failure / interrupt / +# quota fallback) when history leaves a human user text turn directly after a +# tool-result turn. Interposed between the two user contents so the request +# stays alternation-valid while the user's message remains a turn of its own. +# Mirrors gemini-cli's INTERRUPTED_RESPONSE_PLACEHOLDER (gemini-cli#28700). +_INTERRUPTED_RESPONSE_PLACEHOLDER = ( + "[The previous response was interrupted before it completed.]" +) + + def _translate_tool_call_to_gemini(tool_call: Dict[str, Any]) -> Dict[str, Any]: fn = tool_call.get("function") or {} args_raw = fn.get("arguments", "") @@ -394,14 +404,23 @@ def _build_gemini_contents(messages: List[Dict[str, Any]]) -> tuple[List[Dict[st # Compatibility contract for native Gemini generateContent: # 1) Same-role adjacent contents still merge in general (strict user/model - # alternation for ordinary text turns and parallel tool-result grouping). - # 2) Exception: do NOT merge a human user text turn into a preceding user + # alternation for ordinary text turns and parallel tool-result grouping; + # consecutive same-role contents are rejected with HTTP 400 "Please + # ensure that multiturn requests alternate between user and model"). + # 2) Exception: do NOT fuse a human user text turn into a preceding user # content that only carries functionResponse parts (or vice versa). - # Gemini 3 accepts that fold with HTTP 200 but then returns an empty - # model response; keeping the boundary emits two consecutive user - # contents, which current Gemini APIs accept for this specific case. - # 3) Parallel tool results (functionResponse + functionResponse) still - # merge into one user content — only mixed functionResponse/text is split. + # Gemini 3 accepts that fold with HTTP 200 but then reads the trailing + # text as a continuation of the tool result — it returns an empty + # candidate or "finishes the user's sentence" instead of answering + # (same defect gemini-cli fixed in google-gemini/gemini-cli#28700). + # 3) Because rule 1's HTTP 400 makes two consecutive user contents unsafe + # to emit (#55125 — the reason this merge exists), the split pair is + # kept API-valid by interposing a placeholder model turn between the + # functionResponse content and the human text content, mirroring + # gemini-cli's INTERRUPTED_RESPONSE_PLACEHOLDER repair. + # 4) Parallel tool results (functionResponse + functionResponse) still + # merge into one user content — only mixed functionResponse/text is + # kept apart. merged_contents: List[Dict[str, Any]] = [] for content in contents: same_role = bool( @@ -418,6 +437,12 @@ def _build_gemini_contents(messages: List[Dict[str, Any]]) -> tuple[List[Dict[st ) if previous_has_function_response != current_has_function_response: same_role = False + merged_contents.append( + { + "role": "model", + "parts": [{"text": _INTERRUPTED_RESPONSE_PLACEHOLDER}], + } + ) if same_role: merged_contents[-1]["parts"].extend(content["parts"]) diff --git a/tests/agent/test_gemini_native_adapter.py b/tests/agent/test_gemini_native_adapter.py index 323dae4d40..ab98fed371 100644 --- a/tests/agent/test_gemini_native_adapter.py +++ b/tests/agent/test_gemini_native_adapter.py @@ -31,11 +31,18 @@ class DummyResponse: def test_followup_user_turn_is_not_merged_into_function_response_turn(): """Human follow-up after tool results must stay its own user content. + The split pair is kept alternation-valid by interposing a placeholder + model turn between the functionResponse content and the human text + content (mirrors gemini-cli#28700's INTERRUPTED_RESPONSE_PLACEHOLDER). + Scope: only the functionResponse↔human-text boundary. Ordinary same-role merges (parallel tool results, back-to-back plain user texts) remain required for Gemini alternation and are covered by sibling tests. """ - from agent.gemini_native_adapter import _build_gemini_contents + from agent.gemini_native_adapter import ( + _INTERRUPTED_RESPONSE_PLACEHOLDER, + _build_gemini_contents, + ) messages = [ {"role": "user", "content": "Load the skill"}, @@ -63,9 +70,11 @@ def test_followup_user_turn_is_not_merged_into_function_response_turn(): "user", "model", "user", + "model", "user", ] - assert "functionResponse" in contents[-2]["parts"][0] + assert "functionResponse" in contents[2]["parts"][0] + assert contents[3]["parts"] == [{"text": _INTERRUPTED_RESPONSE_PLACEHOLDER}] assert contents[-1]["parts"] == [{"text": "Continue"}]