Files
hermes-agent/tests/gateway/test_api_server_reasoning_nonstream.py
teknium1 0a846a2e2d feat(api-server): reasoning on non-streaming replies; echoed reasoning input items are ignored
Follow-up to the streaming reasoning commit for #99552 covering the atoms it left open:

- Non-streaming `/v1/chat/completions` returns the turn's reasoning as
  `choices[0].message.reasoning_content`; non-streaming `/v1/responses` emits a
  completed `reasoning` output item ahead of the message (and ahead of that step's
  `function_call` items), so `GET /v1/responses/{id}` replays it too. The text is read
  from the assistant messages the agent already persisted (`build_assistant_message`
  stores it under `reasoning`) instead of re-accumulating `reasoning_callback` deltas:
  in non-stream mode the agent fires the callback per provider delta AND once more with
  the full text as the post-response fallback, so accumulating would double it.
- The Responses input parser skips `{type: "reasoning"}` items in `input` and
  `conversation_history`. Responses SDK clients replay a prior response's `output`
  list as the next `input`; before, the item became an empty `user` message (in
  `input`) or a 400 (in `conversation_history`).
- The streaming `response.completed` envelope carries the full reasoning item
  (`id`, `status: completed`) rather than a `{type, summary}` stub, matching the
  `output_item.done` item and the non-streaming shape.

Live (real route handlers, real AIAgent, fake OpenAI SSE model replaying
`delta.reasoning_content`): before — non-stream chat had no `reasoning_content`,
non-stream responses output was `[message]`, a chained turn echoing the output list
stored an empty user turn and the model saw a repaired placeholder; after — chat
`reasoning_content` equals the streamed text exactly, responses/GET output is
`[reasoning, message]`, the chained turn's model input is `[user, assistant, user]`.
2026-09-19 09:48:35 -07:00

94 lines
4.6 KiB
Python

"""Reasoning on the non-streaming OpenAI-compatible routes and the Responses input parser (#99552).
Non-streaming ``/v1/chat/completions`` carries ``message.reasoning_content`` and non-streaming
``/v1/responses`` a ``reasoning`` output item, read from the assistant messages the agent
persisted; a client replaying a prior response's output list (its ``reasoning`` item included)
as the next ``input`` / ``conversation_history`` must not get a 400 or an empty user turn.
"""
from unittest.mock import patch
import pytest
from aiohttp import web
from aiohttp.test_utils import TestClient, TestServer
from gateway.config import PlatformConfig
from gateway.platforms.api_server import APIServerAdapter
REASONING = "Let me think about this carefully."
def _result(user_text: str) -> dict:
"""Transcript-shaped agent result whose assistant message carries structured reasoning."""
return {"final_response": "42", "completed": True,
"messages": [{"role": "user", "content": user_text},
{"role": "assistant", "content": "42", "reasoning": REASONING}]}
def _app() -> tuple:
adapter = APIServerAdapter(PlatformConfig(enabled=True, extra={}))
app = web.Application()
app["api_server_adapter"] = adapter
app.router.add_post("/v1/chat/completions", adapter._handle_chat_completions)
app.router.add_post("/v1/responses", adapter._handle_responses)
app.router.add_get("/v1/responses/{response_id}", adapter._handle_get_response)
return TestClient(TestServer(app)), adapter
@pytest.mark.asyncio
async def test_non_streaming_routes_carry_reasoning_once():
"""Non-stream chat: ``message.reasoning_content`` equals the persisted reasoning exactly
(no delta+fallback doubling); non-stream responses: one completed ``reasoning`` item
before the message, also present on ``GET /v1/responses/{id}`` replay."""
client, adapter = _app()
async with client:
async def _fake_run_agent(**kw):
return _result(kw["user_message"]), {"input_tokens": 1, "output_tokens": 1, "total_tokens": 2}
with patch.object(adapter, "_run_agent", side_effect=_fake_run_agent):
r = await client.post("/v1/chat/completions", json={
"model": "hermes-agent", "messages": [{"role": "user", "content": "q"}]})
assert r.status == 200
message = (await r.json())["choices"][0]["message"]
assert message["reasoning_content"] == REASONING
assert REASONING not in message["content"]
r = await client.post("/v1/responses", json={"model": "hermes-agent", "input": "q", "store": True})
assert r.status == 200
data = await r.json()
assert [o["type"] for o in data["output"]] == ["reasoning", "message"]
assert data["output"][0]["status"] == "completed"
assert data["output"][0]["summary"] == [{"type": "summary_text", "text": REASONING}]
replay = await (await client.get(f"/v1/responses/{data['id']}")).json()
assert [o["type"] for o in replay["output"]] == ["reasoning", "message"]
@pytest.mark.asyncio
async def test_responses_input_ignores_echoed_reasoning_items():
"""A ``{type: reasoning}`` item replayed in ``input`` or ``conversation_history`` is skipped:
no 400, no empty ``user`` message in the history the agent receives."""
client, adapter = _app()
async with client:
captured = {}
async def _fake_run_agent(**kw):
captured["history"] = kw["conversation_history"]
captured["user"] = kw["user_message"]
return _result(kw["user_message"]), {}
reasoning_item = {"type": "reasoning", "id": "rs_1",
"summary": [{"type": "summary_text", "text": "thought"}]}
with patch.object(adapter, "_run_agent", side_effect=_fake_run_agent):
r = await client.post("/v1/responses", json={
"model": "hermes-agent", "store": False,
"conversation_history": [{"role": "user", "content": "h0"}, reasoning_item,
{"role": "assistant", "content": "a0"}],
"input": [{"role": "user", "content": "first"}, reasoning_item,
{"type": "message", "role": "assistant",
"content": [{"type": "output_text", "text": "hi"}]},
{"role": "user", "content": "second"}]})
assert r.status == 200, await r.text()
assert captured["user"] == "second"
assert [(m["role"], m["content"]) for m in captured["history"]] == [
("user", "h0"), ("assistant", "a0"), ("user", "first"), ("assistant", "hi")]