Files
hermes-agent/tests/agent/test_served_model.py
teknium1 bafb778edc feat(gateway): opt-in served_model footer field shows the model that really answered
What: a new opt-in gateway runtime-footer field `served_model` rendered as
`alias → served`. It is populated from the `x-litellm-model-id` response header
(fallback `x-litellm-model-api-base`) that routing proxies send on every chat
completion, captured through an httpx response hook installed on the agent's
OpenAI client (agent/served_model.py, wired in create_openai_client), and from
Hermes' own provider fallback (primary runtime model → active model) when no
header is present. The turn result carries `requested_model` / `served_model`;
gateway/run_turn.py passes them to the footer. Off unless listed in
`display.runtime_footer.fields`; the default field set renders exactly as before.

Why: behind a routing proxy (or during a silent Hermes fallback) every reply
shows the configured alias, so operators cannot see which deployment actually
served a request (#54864). The SDK's parsed objects drop response headers, so
the capture has to sit on the transport.
2026-09-19 10:38:30 -07:00

46 lines
2.1 KiB
Python

"""agent.served_model — per-response served-model capture behind routing proxies (#54864)."""
from __future__ import annotations
import httpx
import openai
from agent.served_model import install_served_model_capture, result_model_fields
class _Agent:
model = "hermes-router"
_fallback_activated = False
_primary_runtime: dict = {}
def test_httpx_hook_captures_litellm_header_and_clears_when_absent():
served = {"value": "gpt-4o-2024-11-20"}
def handler(request: httpx.Request) -> httpx.Response:
headers = {"x-litellm-model-id": served["value"]} if served["value"] else {}
return httpx.Response(200, json={
"id": "c", "object": "chat.completion", "created": 0, "model": "hermes-router",
"choices": [{"index": 0, "message": {"role": "assistant", "content": "ok"}, "finish_reason": "stop"}],
}, headers=headers)
client = openai.OpenAI(api_key="x", base_url="http://proxy.test/v1",
http_client=httpx.Client(transport=httpx.MockTransport(handler)), max_retries=0)
agent = _Agent()
install_served_model_capture(agent, client)
install_served_model_capture(agent, client) # idempotent: one hook per client
assert len(client._client.event_hooks["response"]) == 1
client.chat.completions.create(model="hermes-router", messages=[{"role": "user", "content": "hi"}])
assert result_model_fields(agent) == {"requested_model": "hermes-router", "served_model": "gpt-4o-2024-11-20"}
served["value"] = "" # next response has no routing header: the stale value must not survive
client.chat.completions.create(model="hermes-router", messages=[{"role": "user", "content": "hi"}])
assert result_model_fields(agent) == {"requested_model": "hermes-router", "served_model": None}
# Hermes' own fallback route surfaces the same way when no proxy header is present.
agent._fallback_activated = True
agent._primary_runtime = {"model": "gpt-5.6-sol"}
agent.model = "qwen/qwen3.8-max"
assert result_model_fields(agent) == {"requested_model": "gpt-5.6-sol", "served_model": "qwen/qwen3.8-max"}