fix(gemini): interpose placeholder model turn between tool result and user text
Port from google-gemini/gemini-cli#28700: when an interrupted/failed turn leaves history ending on an unanswered tool result and the user sends a new message, fusing the two into one Gemini user content makes the model read the trailing text as a continuation of the tool result — it 'finishes your sentence' instead of answering. Builds on #68863 (@rille111), which split the mixed functionResponse/text merge but emitted two consecutive user contents — a shape Gemini's alternation contract rejects with HTTP 400 on other request paths (#55125). This follow-up interposes gemini-cli's INTERRUPTED_RESPONSE_PLACEHOLDER model turn between the split contents so the request stays alternation-valid while the user's message remains a turn of its own.
This commit is contained in:
@@ -289,6 +289,16 @@ def _tool_call_extra_signature(tool_call: Dict[str, Any]) -> Optional[str]:
|
||||
return None
|
||||
|
||||
|
||||
# Stands in for a model turn that never arrived (stream failure / interrupt /
|
||||
# quota fallback) when history leaves a human user text turn directly after a
|
||||
# tool-result turn. Interposed between the two user contents so the request
|
||||
# stays alternation-valid while the user's message remains a turn of its own.
|
||||
# Mirrors gemini-cli's INTERRUPTED_RESPONSE_PLACEHOLDER (gemini-cli#28700).
|
||||
_INTERRUPTED_RESPONSE_PLACEHOLDER = (
|
||||
"[The previous response was interrupted before it completed.]"
|
||||
)
|
||||
|
||||
|
||||
def _translate_tool_call_to_gemini(tool_call: Dict[str, Any]) -> Dict[str, Any]:
|
||||
fn = tool_call.get("function") or {}
|
||||
args_raw = fn.get("arguments", "")
|
||||
@@ -394,14 +404,23 @@ def _build_gemini_contents(messages: List[Dict[str, Any]]) -> tuple[List[Dict[st
|
||||
|
||||
# Compatibility contract for native Gemini generateContent:
|
||||
# 1) Same-role adjacent contents still merge in general (strict user/model
|
||||
# alternation for ordinary text turns and parallel tool-result grouping).
|
||||
# 2) Exception: do NOT merge a human user text turn into a preceding user
|
||||
# alternation for ordinary text turns and parallel tool-result grouping;
|
||||
# consecutive same-role contents are rejected with HTTP 400 "Please
|
||||
# ensure that multiturn requests alternate between user and model").
|
||||
# 2) Exception: do NOT fuse a human user text turn into a preceding user
|
||||
# content that only carries functionResponse parts (or vice versa).
|
||||
# Gemini 3 accepts that fold with HTTP 200 but then returns an empty
|
||||
# model response; keeping the boundary emits two consecutive user
|
||||
# contents, which current Gemini APIs accept for this specific case.
|
||||
# 3) Parallel tool results (functionResponse + functionResponse) still
|
||||
# merge into one user content — only mixed functionResponse/text is split.
|
||||
# Gemini 3 accepts that fold with HTTP 200 but then reads the trailing
|
||||
# text as a continuation of the tool result — it returns an empty
|
||||
# candidate or "finishes the user's sentence" instead of answering
|
||||
# (same defect gemini-cli fixed in google-gemini/gemini-cli#28700).
|
||||
# 3) Because rule 1's HTTP 400 makes two consecutive user contents unsafe
|
||||
# to emit (#55125 — the reason this merge exists), the split pair is
|
||||
# kept API-valid by interposing a placeholder model turn between the
|
||||
# functionResponse content and the human text content, mirroring
|
||||
# gemini-cli's INTERRUPTED_RESPONSE_PLACEHOLDER repair.
|
||||
# 4) Parallel tool results (functionResponse + functionResponse) still
|
||||
# merge into one user content — only mixed functionResponse/text is
|
||||
# kept apart.
|
||||
merged_contents: List[Dict[str, Any]] = []
|
||||
for content in contents:
|
||||
same_role = bool(
|
||||
@@ -418,6 +437,12 @@ def _build_gemini_contents(messages: List[Dict[str, Any]]) -> tuple[List[Dict[st
|
||||
)
|
||||
if previous_has_function_response != current_has_function_response:
|
||||
same_role = False
|
||||
merged_contents.append(
|
||||
{
|
||||
"role": "model",
|
||||
"parts": [{"text": _INTERRUPTED_RESPONSE_PLACEHOLDER}],
|
||||
}
|
||||
)
|
||||
|
||||
if same_role:
|
||||
merged_contents[-1]["parts"].extend(content["parts"])
|
||||
|
||||
@@ -31,11 +31,18 @@ class DummyResponse:
|
||||
def test_followup_user_turn_is_not_merged_into_function_response_turn():
|
||||
"""Human follow-up after tool results must stay its own user content.
|
||||
|
||||
The split pair is kept alternation-valid by interposing a placeholder
|
||||
model turn between the functionResponse content and the human text
|
||||
content (mirrors gemini-cli#28700's INTERRUPTED_RESPONSE_PLACEHOLDER).
|
||||
|
||||
Scope: only the functionResponse↔human-text boundary. Ordinary same-role
|
||||
merges (parallel tool results, back-to-back plain user texts) remain
|
||||
required for Gemini alternation and are covered by sibling tests.
|
||||
"""
|
||||
from agent.gemini_native_adapter import _build_gemini_contents
|
||||
from agent.gemini_native_adapter import (
|
||||
_INTERRUPTED_RESPONSE_PLACEHOLDER,
|
||||
_build_gemini_contents,
|
||||
)
|
||||
|
||||
messages = [
|
||||
{"role": "user", "content": "Load the skill"},
|
||||
@@ -63,9 +70,11 @@ def test_followup_user_turn_is_not_merged_into_function_response_turn():
|
||||
"user",
|
||||
"model",
|
||||
"user",
|
||||
"model",
|
||||
"user",
|
||||
]
|
||||
assert "functionResponse" in contents[-2]["parts"][0]
|
||||
assert "functionResponse" in contents[2]["parts"][0]
|
||||
assert contents[3]["parts"] == [{"text": _INTERRUPTED_RESPONSE_PLACEHOLDER}]
|
||||
assert contents[-1]["parts"] == [{"text": "Continue"}]
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user