fix(api_server): emit the run record's runtime in the canonical _sanitize_runtime_metadata shape
/v1/runs published a thinner {provider, model} twin under the same wire key that
/v1/chat/completions, /v1/responses and the session chat stream fill via
_sanitize_runtime_metadata (route_source, requested, cleaned ids). Route the served pair through
the same classmethod so one concept has one schema; requested comes from the run's
requested_model/requested_provider overrides, route_source from the model_routes/raw/global
vocabulary the other endpoints use. Doc example updated.
This commit is contained in:
@@ -790,7 +790,13 @@ async def _execute_run(self, run: _RunLaunch, *, _api_server) -> None:
|
||||
# Non-retryable client errors (401/400) return failed=True rather than raising.
|
||||
_finish("failed", fields, error=_redact_api_error_text(result.get("error") or "agent run failed"))
|
||||
else:
|
||||
# ``runtime`` rides on both the pollable status and the run.completed event via _finish.
|
||||
# ``runtime`` rides on both the pollable status and the run.completed event via _finish, in the
|
||||
# canonical shape every other api_server surface emits (route_source/requested, cleaned ids).
|
||||
requested = {k: run.agent_kwargs.get(f"requested_{k}") for k in ("provider", "model")}
|
||||
served_runtime = self._sanitize_runtime_metadata(
|
||||
runtime=served_runtime, requested_runtime=requested if any(requested.values()) else None,
|
||||
route_source=("model_routes" if run.agent_kwargs.get("route")
|
||||
else "raw_request" if any(requested.values()) else "global"))
|
||||
_finish(status, fields, output=result.get("final_response", ""), usage=usage, runtime=served_runtime)
|
||||
except asyncio.CancelledError:
|
||||
_finish("cancelled")
|
||||
|
||||
@@ -443,7 +443,11 @@ class TestRunStatus:
|
||||
assert status["status"] == "completed"
|
||||
# Top-level model still echoes the request; the served pair is disclosed alongside.
|
||||
assert status["model"] == "deepseek-v4-pro"
|
||||
assert status["runtime"] == {"provider": "openai-codex", "model": "gpt-5.6-luna"}
|
||||
# Canonical api_server runtime shape (same as /v1/chat/completions), not a thinner twin.
|
||||
assert status["runtime"] == {
|
||||
"provider": "openai-codex", "model": "gpt-5.6-luna", "route_source": "raw_request",
|
||||
"requested": {"provider": "", "model": "deepseek-v4-pro"},
|
||||
}
|
||||
assert status["usage"] == {
|
||||
"input_tokens": 100, "output_tokens": 5, "total_tokens": 105,
|
||||
"cache_read_tokens": 84, "cache_write_tokens": 11,
|
||||
@@ -545,7 +549,9 @@ class TestRunEvents:
|
||||
completed = payload
|
||||
break
|
||||
assert completed is not None, "run.completed event missing from stream"
|
||||
assert completed["runtime"] == {"provider": "openai-codex", "model": "gpt-5.6-luna"}
|
||||
assert completed["runtime"]["provider"] == "openai-codex"
|
||||
assert completed["runtime"]["model"] == "gpt-5.6-luna"
|
||||
assert completed["runtime"]["requested"]["model"] == "deepseek-v4-pro"
|
||||
assert completed["usage"]["cache_read_tokens"] == 650000
|
||||
assert completed["usage"]["cache_write_tokens"] == 42
|
||||
|
||||
|
||||
@@ -474,11 +474,11 @@ Poll the current run state. This is useful for dashboards that need status witho
|
||||
"model": "hermes-agent",
|
||||
"output": "Done.",
|
||||
"usage": {"input_tokens": 50, "output_tokens": 200, "total_tokens": 250, "cache_read_tokens": 40, "cache_write_tokens": 0},
|
||||
"runtime": {"provider": "openai", "model": "gpt-5"}
|
||||
"runtime": {"provider": "openai", "model": "gpt-5", "route_source": "global"}
|
||||
}
|
||||
```
|
||||
|
||||
`model` echoes what the request asked for. On a completed run, `runtime` is the provider/model pair that actually served the turn — after a [fallback provider](fallback-providers.md) switch it names the fallback pair, so a cost-attribution poller books the run to the right provider. `usage.cache_read_tokens` / `usage.cache_write_tokens` are the session's prompt-cache reads and writes, so cached input is not priced as full-price input. The `run.completed` event on the events stream carries the same `usage` and `runtime` fields.
|
||||
`model` echoes what the request asked for. On a completed run, `runtime` is the provider/model pair that actually served the turn — after a [fallback provider](fallback-providers.md) switch it names the fallback pair, so a cost-attribution poller books the run to the right provider. `usage.cache_read_tokens` / `usage.cache_write_tokens` are the session's prompt-cache reads and writes, so cached input is not priced as full-price input. `runtime` has the same shape as on `/v1/chat/completions` and `/v1/responses`: `route_source` says how the runtime was chosen (`global`, `raw_request`, `model_routes`), and a request that named a `model`/`provider` also gets `requested: {provider, model}` so the asked-for and served pairs can be compared. The `run.completed` event on the events stream carries the same `usage` and `runtime` fields.
|
||||
|
||||
Statuses are retained briefly after terminal states (`completed`, `failed`, `cancelled`, or `interrupted`) for polling and UI reconciliation. When the gateway shuts down while a run is active, the run is persisted as `interrupted` (error `Gateway shutdown interrupted the run.`, terminal event `run.interrupted`) before the agent is asked to stop, so a durable run never survives a restart as `running`; a late result from the interrupted turn cannot overwrite it.
|
||||
|
||||
|
||||
Reference in New Issue
Block a user