From 6b5a8fad389119230a3142df45dadbb267936d71 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Sat, 19 Sep 2026 01:26:12 -0700 Subject: [PATCH] fix(api_server): emit the run record's runtime in the canonical _sanitize_runtime_metadata shape /v1/runs published a thinner {provider, model} twin under the same wire key that /v1/chat/completions, /v1/responses and the session chat stream fill via _sanitize_runtime_metadata (route_source, requested, cleaned ids). Route the served pair through the same classmethod so one concept has one schema; requested comes from the run's requested_model/requested_provider overrides, route_source from the model_routes/raw/global vocabulary the other endpoints use. Doc example updated. --- gateway/platforms/api_server_runs.py | 8 +++++++- tests/gateway/test_api_server_runs.py | 10 ++++++++-- website/docs/user-guide/features/api-server.md | 4 ++-- 3 files changed, 17 insertions(+), 5 deletions(-) diff --git a/gateway/platforms/api_server_runs.py b/gateway/platforms/api_server_runs.py index 94691c8891..6930d6d33d 100644 --- a/gateway/platforms/api_server_runs.py +++ b/gateway/platforms/api_server_runs.py @@ -790,7 +790,13 @@ async def _execute_run(self, run: _RunLaunch, *, _api_server) -> None: # Non-retryable client errors (401/400) return failed=True rather than raising. _finish("failed", fields, error=_redact_api_error_text(result.get("error") or "agent run failed")) else: - # ``runtime`` rides on both the pollable status and the run.completed event via _finish. + # ``runtime`` rides on both the pollable status and the run.completed event via _finish, in the + # canonical shape every other api_server surface emits (route_source/requested, cleaned ids). + requested = {k: run.agent_kwargs.get(f"requested_{k}") for k in ("provider", "model")} + served_runtime = self._sanitize_runtime_metadata( + runtime=served_runtime, requested_runtime=requested if any(requested.values()) else None, + route_source=("model_routes" if run.agent_kwargs.get("route") + else "raw_request" if any(requested.values()) else "global")) _finish(status, fields, output=result.get("final_response", ""), usage=usage, runtime=served_runtime) except asyncio.CancelledError: _finish("cancelled") diff --git a/tests/gateway/test_api_server_runs.py b/tests/gateway/test_api_server_runs.py index adf12472e2..7146cf755f 100644 --- a/tests/gateway/test_api_server_runs.py +++ b/tests/gateway/test_api_server_runs.py @@ -443,7 +443,11 @@ class TestRunStatus: assert status["status"] == "completed" # Top-level model still echoes the request; the served pair is disclosed alongside. assert status["model"] == "deepseek-v4-pro" - assert status["runtime"] == {"provider": "openai-codex", "model": "gpt-5.6-luna"} + # Canonical api_server runtime shape (same as /v1/chat/completions), not a thinner twin. + assert status["runtime"] == { + "provider": "openai-codex", "model": "gpt-5.6-luna", "route_source": "raw_request", + "requested": {"provider": "", "model": "deepseek-v4-pro"}, + } assert status["usage"] == { "input_tokens": 100, "output_tokens": 5, "total_tokens": 105, "cache_read_tokens": 84, "cache_write_tokens": 11, @@ -545,7 +549,9 @@ class TestRunEvents: completed = payload break assert completed is not None, "run.completed event missing from stream" - assert completed["runtime"] == {"provider": "openai-codex", "model": "gpt-5.6-luna"} + assert completed["runtime"]["provider"] == "openai-codex" + assert completed["runtime"]["model"] == "gpt-5.6-luna" + assert completed["runtime"]["requested"]["model"] == "deepseek-v4-pro" assert completed["usage"]["cache_read_tokens"] == 650000 assert completed["usage"]["cache_write_tokens"] == 42 diff --git a/website/docs/user-guide/features/api-server.md b/website/docs/user-guide/features/api-server.md index 4b77943ade..84dd349435 100644 --- a/website/docs/user-guide/features/api-server.md +++ b/website/docs/user-guide/features/api-server.md @@ -474,11 +474,11 @@ Poll the current run state. This is useful for dashboards that need status witho "model": "hermes-agent", "output": "Done.", "usage": {"input_tokens": 50, "output_tokens": 200, "total_tokens": 250, "cache_read_tokens": 40, "cache_write_tokens": 0}, - "runtime": {"provider": "openai", "model": "gpt-5"} + "runtime": {"provider": "openai", "model": "gpt-5", "route_source": "global"} } ``` -`model` echoes what the request asked for. On a completed run, `runtime` is the provider/model pair that actually served the turn — after a [fallback provider](fallback-providers.md) switch it names the fallback pair, so a cost-attribution poller books the run to the right provider. `usage.cache_read_tokens` / `usage.cache_write_tokens` are the session's prompt-cache reads and writes, so cached input is not priced as full-price input. The `run.completed` event on the events stream carries the same `usage` and `runtime` fields. +`model` echoes what the request asked for. On a completed run, `runtime` is the provider/model pair that actually served the turn — after a [fallback provider](fallback-providers.md) switch it names the fallback pair, so a cost-attribution poller books the run to the right provider. `usage.cache_read_tokens` / `usage.cache_write_tokens` are the session's prompt-cache reads and writes, so cached input is not priced as full-price input. `runtime` has the same shape as on `/v1/chat/completions` and `/v1/responses`: `route_source` says how the runtime was chosen (`global`, `raw_request`, `model_routes`), and a request that named a `model`/`provider` also gets `requested: {provider, model}` so the asked-for and served pairs can be compared. The `run.completed` event on the events stream carries the same `usage` and `runtime` fields. Statuses are retained briefly after terminal states (`completed`, `failed`, `cancelled`, or `interrupted`) for polling and UI reconciliation. When the gateway shuts down while a run is active, the run is persisted as `interrupted` (error `Gateway shutdown interrupted the run.`, terminal event `run.interrupted`) before the agent is asked to stop, so a durable run never survives a restart as `running`; a late result from the interrupted turn cannot overwrite it.