diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 48c7517827..06f340ae47 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -3148,6 +3148,11 @@ class _StreamingCall(StreamingWaitMonitor): base_timeout, read_timeout, conn_cap = self._stream_timeouts() content_parts: list = [] reasoning_parts: list = [] + # Live-display accumulator for detail-derived reasoning text: de-gluing must + # compare against what the display actually received, not ``reasoning_parts`` + # (a provider that mirrors the same text in both fields would otherwise read + # as already-glued on the first chunk and get a spurious break). + detail_display_parts: list[str] = [] # OpenAI structured refusal (``delta.refusal``): the explanation streams here and # ``delta.content`` stays empty, so an un-accumulated refusal looks like an empty # stream and burns the empty-response retries (the non-streaming fix is #46013). @@ -3247,7 +3252,17 @@ class _StreamingCall(StreamingWaitMonitor): # Details may carry the full text while ordinary reasoning is only # a sparse fragment or a mirror. Deliver one representation per # chunk, without rewriting either persisted/replayed field. - display_reasoning = "".join(detail_text_parts) or reasoning_text + # Summary-part boundaries need the same repair the plain path applies: + # de-glue against the detail display's own accumulator (not + # ``reasoning_parts`` — with mirrored fields that would insert a + # spurious break on the first chunk), so the live box and the + # persisted ``reasoning_content`` agree. + detail_text = "".join(detail_text_parts) + if detail_text: + detail_text = separate_glued_reasoning_blocks( + detail_display_parts[-1] if detail_display_parts else "", detail_text) + detail_display_parts.append(detail_text) + display_reasoning = detail_text or reasoning_text if display_reasoning: self._emit_reasoning(display_reasoning) # Not routed to the live display: the transport promotes a sole-payload diff --git a/tests/agent/test_streamed_reasoning_details.py b/tests/agent/test_streamed_reasoning_details.py index b12fc8a47d..714e93517b 100644 --- a/tests/agent/test_streamed_reasoning_details.py +++ b/tests/agent/test_streamed_reasoning_details.py @@ -78,6 +78,52 @@ def test_streamed_details_land_on_final_message_and_persist(_mock_close, mock_cr assert persisted["reasoning_details"] == msg.reasoning_details +@patch("run_agent.AIAgent._create_request_openai_client") +@patch("run_agent.AIAgent._close_request_openai_client") +def test_live_details_deglue_summary_part_boundaries(_mock_close, mock_create): + """The live display gets the same boundary repair as the persisted reasoning. + + A reasoning-summary model streams one delta per completed summary part, each + a bare bold heading. Without the ``separate_glued_reasoning_blocks`` repair + on the detail path, the live box glues parts head-to-tail while + ``reasoning_content`` stays de-glued — display and history disagree. + """ + agent = _agent() + client = MagicMock() + mock_create.return_value = client + + def summary_chunk(summary, **kw): + chunk = _make_chunk(**kw) + chunk.choices[0].delta.reasoning = summary # provider mirrors both fields + chunk.choices[0].delta.model_extra = { + "reasoning_details": [{"type": "reasoning.summary", "summary": summary}]} + return chunk + + client.chat.completions.create.return_value = iter([ + summary_chunk("**One**"), + summary_chunk("**Two**"), + _make_chunk(content="Answer", finish_reason="stop", model="test-model"), + ]) + delivered = [] + agent.reasoning_callback = delivered.append + response = agent._interruptible_streaming_api_call({}) + + assert "".join(delivered) == "**One**\n\n**Two**" + assert response.choices[0].message.reasoning_content == "**One**\n\n**Two**" + + # Details-only variant: the delta carries no plain ``reasoning`` field at all. + client.chat.completions.create.return_value = iter([ + _make_chunk(reasoning_details=[{"type": "reasoning.summary", "summary": "**One**"}]), + _make_chunk(reasoning_details=[{"type": "reasoning.summary", "summary": "**Two**"}]), + _make_chunk(content="Answer", finish_reason="stop", model="test-model"), + ]) + delivered = [] + agent.reasoning_callback = delivered.append + response = agent._interruptible_streaming_api_call({}) + + assert "".join(delivered) == "**One**\n\n**Two**" + + @patch("run_agent.AIAgent._create_request_openai_client") @patch("run_agent.AIAgent._close_request_openai_client") def test_no_details_leaves_attribute_absent(_mock_close, mock_create):