diff --git a/gateway/platforms/api_server_openai_routes.py b/gateway/platforms/api_server_openai_routes.py index 991a5d74c6..db2edf0ac2 100644 --- a/gateway/platforms/api_server_openai_routes.py +++ b/gateway/platforms/api_server_openai_routes.py @@ -88,6 +88,37 @@ def _message_item(text: Any) -> Dict[str, Any]: "content": [{"type": "output_text", "text": text}]} +def _reasoning_item(text: str) -> Dict[str, Any]: + """Completed Responses ``reasoning`` output item (same shape the SSE writer closes with).""" + return {"id": f"rs_{uuid.uuid4().hex[:24]}", "type": "reasoning", "status": "completed", + "summary": [{"type": "summary_text", "text": text}]} + + +def _is_reasoning_input_item(item: Any) -> bool: + """Echoed-back ``reasoning`` output item: Responses SDK clients replay a prior response's + ``output`` list as the next ``input``. It carries no message content, so it must be + skipped rather than parsed into an empty ``user`` turn (#99552).""" + return isinstance(item, dict) and item.get("type") == "reasoning" + + +def _turn_reasoning_text( + conversation_history: List[Dict[str, Any]], user_message: Any, result: Dict[str, Any]) -> str: + """Reasoning the model produced on this turn, joined for a non-streaming + ``message.reasoning_content``. Read from the assistant messages the agent already + persisted (``build_assistant_message`` stores the structured reasoning under + ``reasoning``) rather than re-accumulating callback deltas, so it is exactly what the + stream would have carried and cannot double-count the post-response fallback.""" + messages = result.get("messages") if isinstance(result, dict) else None + if not isinstance(messages, list): + return "" + start = OpenAICompatRoutesMixin._response_messages_turn_start_index( + conversation_history, user_message, result) + parts = [m["reasoning"] for m in messages[start:] + if isinstance(m, dict) and m.get("role") == "assistant" + and isinstance(m.get("reasoning"), str) and m["reasoning"].strip()] + return "\n\n".join(parts) + + def _trim_tool_items(items: List[Dict[str, Any]]) -> List[Dict[str, Any]]: """Trim large tool payloads in place so response.completed stays under ~100KB (clients already received the full details via the incremental events).""" @@ -256,7 +287,7 @@ class _ResponsesStream: await self.write_event("response.reasoning_summary_part.done", { "type": "response.reasoning_summary_part.done", **base, "part": {"type": "summary_text", "text": text}}) - self.emitted_items.append({"type": "reasoning", "summary": item["summary"]}) + self.emitted_items.append(item) await self.write_event("response.output_item.done", { "type": "response.output_item.done", "output_index": rs["output_index"], "item": item}) @@ -645,6 +676,10 @@ class OpenAICompatRoutesMixin: "choices": [{"index": 0, "message": {"role": "assistant", "content": "" if presentation_muted else final_response}, "finish_reason": finish_reason}], "usage": _chat_usage_payload(usage)} + # Non-streaming twin of ``delta.reasoning_content`` (#99552). + reasoning_text = _turn_reasoning_text(history, user_message, result) + if reasoning_text and not presentation_muted: + response_data["choices"][0]["message"]["reasoning_content"] = reasoning_text if is_partial or is_failed or not completed: response_data["hermes"] = _hermes_extras( completed, is_partial, is_failed, "" if presentation_muted else err_msg, finish_reason) @@ -867,6 +902,8 @@ class OpenAICompatRoutesMixin: for idx, item in enumerate(raw_input): if isinstance(item, str): input_messages.append({"role": "user", "content": item}) + elif _is_reasoning_input_item(item): + continue elif isinstance(item, dict): try: content = _normalize_multimodal_content(item.get("content", "")) @@ -883,6 +920,8 @@ class OpenAICompatRoutesMixin: if not isinstance(raw_history, list): return _error_response("'conversation_history' must be an array of message objects", 400) for i, entry in enumerate(raw_history): + if _is_reasoning_input_item(entry): + continue if not isinstance(entry, dict) or "role" not in entry or "content" not in entry: return _error_response(f"conversation_history[{i}] must have 'role' and 'content' fields", 400) try: @@ -1096,6 +1135,11 @@ class OpenAICompatRoutesMixin: messages = messages[start_index:] for msg in messages: role = msg.get("role") + reasoning = msg.get("reasoning") if role == "assistant" else None + if isinstance(reasoning, str) and reasoning.strip(): + # Precedes this message's function_call items, like the SSE writer closes a + # thinking burst before the next tool item opens (#99552). + items.append(_reasoning_item(reasoning)) if role == "assistant" and msg.get("tool_calls"): for tc in msg["tool_calls"]: func = tc.get("function", {}) diff --git a/tests/gateway/test_api_server_reasoning_nonstream.py b/tests/gateway/test_api_server_reasoning_nonstream.py new file mode 100644 index 0000000000..33244a98e4 --- /dev/null +++ b/tests/gateway/test_api_server_reasoning_nonstream.py @@ -0,0 +1,93 @@ +"""Reasoning on the non-streaming OpenAI-compatible routes and the Responses input parser (#99552). + +Non-streaming ``/v1/chat/completions`` carries ``message.reasoning_content`` and non-streaming +``/v1/responses`` a ``reasoning`` output item, read from the assistant messages the agent +persisted; a client replaying a prior response's output list (its ``reasoning`` item included) +as the next ``input`` / ``conversation_history`` must not get a 400 or an empty user turn. +""" + +from unittest.mock import patch + +import pytest +from aiohttp import web +from aiohttp.test_utils import TestClient, TestServer + +from gateway.config import PlatformConfig +from gateway.platforms.api_server import APIServerAdapter + +REASONING = "Let me think about this carefully." + + +def _result(user_text: str) -> dict: + """Transcript-shaped agent result whose assistant message carries structured reasoning.""" + return {"final_response": "42", "completed": True, + "messages": [{"role": "user", "content": user_text}, + {"role": "assistant", "content": "42", "reasoning": REASONING}]} + + +def _app() -> tuple: + adapter = APIServerAdapter(PlatformConfig(enabled=True, extra={})) + app = web.Application() + app["api_server_adapter"] = adapter + app.router.add_post("/v1/chat/completions", adapter._handle_chat_completions) + app.router.add_post("/v1/responses", adapter._handle_responses) + app.router.add_get("/v1/responses/{response_id}", adapter._handle_get_response) + return TestClient(TestServer(app)), adapter + + +@pytest.mark.asyncio +async def test_non_streaming_routes_carry_reasoning_once(): + """Non-stream chat: ``message.reasoning_content`` equals the persisted reasoning exactly + (no delta+fallback doubling); non-stream responses: one completed ``reasoning`` item + before the message, also present on ``GET /v1/responses/{id}`` replay.""" + client, adapter = _app() + async with client: + async def _fake_run_agent(**kw): + return _result(kw["user_message"]), {"input_tokens": 1, "output_tokens": 1, "total_tokens": 2} + + with patch.object(adapter, "_run_agent", side_effect=_fake_run_agent): + r = await client.post("/v1/chat/completions", json={ + "model": "hermes-agent", "messages": [{"role": "user", "content": "q"}]}) + assert r.status == 200 + message = (await r.json())["choices"][0]["message"] + assert message["reasoning_content"] == REASONING + assert REASONING not in message["content"] + + r = await client.post("/v1/responses", json={"model": "hermes-agent", "input": "q", "store": True}) + assert r.status == 200 + data = await r.json() + assert [o["type"] for o in data["output"]] == ["reasoning", "message"] + assert data["output"][0]["status"] == "completed" + assert data["output"][0]["summary"] == [{"type": "summary_text", "text": REASONING}] + replay = await (await client.get(f"/v1/responses/{data['id']}")).json() + assert [o["type"] for o in replay["output"]] == ["reasoning", "message"] + + +@pytest.mark.asyncio +async def test_responses_input_ignores_echoed_reasoning_items(): + """A ``{type: reasoning}`` item replayed in ``input`` or ``conversation_history`` is skipped: + no 400, no empty ``user`` message in the history the agent receives.""" + client, adapter = _app() + async with client: + captured = {} + + async def _fake_run_agent(**kw): + captured["history"] = kw["conversation_history"] + captured["user"] = kw["user_message"] + return _result(kw["user_message"]), {} + + reasoning_item = {"type": "reasoning", "id": "rs_1", + "summary": [{"type": "summary_text", "text": "thought"}]} + with patch.object(adapter, "_run_agent", side_effect=_fake_run_agent): + r = await client.post("/v1/responses", json={ + "model": "hermes-agent", "store": False, + "conversation_history": [{"role": "user", "content": "h0"}, reasoning_item, + {"role": "assistant", "content": "a0"}], + "input": [{"role": "user", "content": "first"}, reasoning_item, + {"type": "message", "role": "assistant", + "content": [{"type": "output_text", "text": "hi"}]}, + {"role": "user", "content": "second"}]}) + assert r.status == 200, await r.text() + assert captured["user"] == "second" + assert [(m["role"], m["content"]) for m in captured["history"]] == [ + ("user", "h0"), ("assistant", "a0"), ("user", "first"), ("assistant", "hi")] diff --git a/website/docs/user-guide/features/api-server.md b/website/docs/user-guide/features/api-server.md index e0d49ebc82..925f055b39 100644 --- a/website/docs/user-guide/features/api-server.md +++ b/website/docs/user-guide/features/api-server.md @@ -114,10 +114,12 @@ All SSE streams (Chat Completions, Responses, `/api/sessions/{id}/chat/stream`, - **Chat Completions**: Hermes emits `event: hermes.tool.progress` for tool-start visibility without polluting persisted assistant text. - **Responses**: Hermes emits spec-native `function_call` and `function_call_output` output items during the SSE stream, so clients can render structured tool UI in real time. -**Model reasoning in streams** (emitted only when the model actually produces reasoning and the resolved `reasoning` config allows it; the input-side opt-out is `model_options.reasoning.enabled: false`): +**Model reasoning** (emitted only when the model actually produces reasoning and the resolved `reasoning` config allows it; the input-side opt-out is `model_options.reasoning.enabled: false`): - **Chat Completions**: reasoning deltas arrive as `choices[0].delta.reasoning_content` chunks (the DeepSeek-style field Open WebUI, opencode and the Vercel AI SDK render as a thinking block); answer text stays in `delta.content`. -- **Responses**: each thinking burst is a spec-native `reasoning` output item — `response.output_item.added` (`item.type: "reasoning"`), `response.reasoning_summary_part.added`, `response.reasoning_summary_text.delta` … `response.reasoning_summary_text.done`, `response.reasoning_summary_part.done`, `response.output_item.done` — closed before the next message or `function_call` item opens, and echoed in the `response.completed` output as `{"type": "reasoning", "summary": [{"type": "summary_text", "text": "…"}]}`. `sequence_number` stays monotonic across reasoning, text and tool events. -- Support is advertised as `features.reasoning_streaming: true` on `GET /v1/capabilities`. Non-streaming responses do not carry reasoning. +- **Responses**: each thinking burst is a spec-native `reasoning` output item — `response.output_item.added` (`item.type: "reasoning"`), `response.reasoning_summary_part.added`, `response.reasoning_summary_text.delta` … `response.reasoning_summary_text.done`, `response.reasoning_summary_part.done`, `response.output_item.done` — closed before the next message or `function_call` item opens, and echoed in the `response.completed` output as `{"id": "rs_…", "type": "reasoning", "status": "completed", "summary": [{"type": "summary_text", "text": "…"}]}`. `sequence_number` stays monotonic across reasoning, text and tool events. +- **Non-streaming**: `/v1/chat/completions` returns the turn's reasoning on `choices[0].message.reasoning_content`; `/v1/responses` returns the same `reasoning` output item(s) ahead of the message (and of that step's `function_call` items), also on `GET /v1/responses/{id}` replay. +- Echoing a prior response's `output` list back as the next `input` (what Responses SDK clients do) is fine: `reasoning` items are ignored on input rather than parsed as empty user turns. +- Support is advertised as `features.reasoning_streaming: true` on `GET /v1/capabilities`. ### POST /v1/responses