657 lines
30 KiB
Python
657 lines
30 KiB
Python
"""OpenAI Responses API (Codex) transport.
|
|
|
|
Owns format conversion/normalization on top of agent/codex_responses_adapter.py —
|
|
NOT client lifecycle, streaming, or the _run_codex_stream() call path.
|
|
"""
|
|
|
|
import hashlib
|
|
import json
|
|
import logging
|
|
import re
|
|
from typing import Any, Callable, Dict, List, Optional, Tuple
|
|
|
|
from agent.reasoning_effort import (
|
|
ACTUAL_RELAY_EFFORTS,
|
|
XAI_GROK46_EFFORTS,
|
|
XAI_LEGACY_EFFORTS,
|
|
clamp_effort,
|
|
codex_supported_efforts,
|
|
)
|
|
from agent.transports.base import ProviderTransport
|
|
from agent.transports.types import NormalizedResponse, ToolCall
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Cron fires use ``cron_<job_id>_<YYYYMMDD_HHMMSS>``; the per-fire timestamp is
|
|
# stripped so repeat fires of one job share a cache scope.
|
|
_CRON_SESSION_ID_RE = re.compile(r"^(cron_.+)_\d{8}_\d{6}$")
|
|
|
|
|
|
def _cache_scope_from_session_id(session_id: Optional[str]) -> str:
|
|
"""Normalize a physical session_id into a stable logical cache scope."""
|
|
sid = str(session_id or "")
|
|
match = _CRON_SESSION_ID_RE.match(sid)
|
|
return match.group(1) if match else sid
|
|
|
|
|
|
def _bounded_prompt_cache_key(value: Any) -> Optional[str]:
|
|
"""Return a provider-safe (<=64 char) cache key without changing session identity."""
|
|
key = "" if value is None else str(value).strip()
|
|
if not key:
|
|
return None
|
|
if len(key) <= 64:
|
|
return key
|
|
return "pck_" + hashlib.sha256(key.encode("utf-8", errors="replace")).hexdigest()[:24]
|
|
|
|
|
|
def _bound_prompt_cache_key_field(container: Any) -> None:
|
|
"""Bound (or drop, when empty) an in-place ``prompt_cache_key`` entry."""
|
|
if isinstance(container, dict) and "prompt_cache_key" in container:
|
|
bounded = _bounded_prompt_cache_key(container["prompt_cache_key"])
|
|
if bounded:
|
|
container["prompt_cache_key"] = bounded
|
|
else:
|
|
container.pop("prompt_cache_key", None)
|
|
|
|
|
|
def _str_headers(existing: Any) -> Dict[str, str]:
|
|
"""Copy an ``extra_headers`` dict with str-coerced keys/values."""
|
|
return {str(k): str(v) for k, v in existing.items() if k and v is not None} if isinstance(existing, dict) else {}
|
|
|
|
|
|
# Client-side ``web_search`` on xAI Responses collides with Grok's native
|
|
# server-side tool (incomplete hang / HTTP 400); it goes on the wire under this
|
|
# alias and is mapped back in normalize_response.
|
|
_XAI_CLIENT_WEB_SEARCH_ALIAS = "hermes_web_search"
|
|
|
|
# OpenCode /v1/responses rejects client tools using these names (HTTP 400
|
|
# "custom function name 'X' is reserved", #85589); xAI rejects ``tool_search``
|
|
# (reserved for Grok's native Tool Search, #95003). Aliased as hermes_<name>.
|
|
_OPENCODE_RESERVED_TOOL_NAMES = ("web_search", "search_files")
|
|
_XAI_RESERVED_TOOL_NAMES = ("tool_search",)
|
|
_RESERVED_TOOL_ALIAS_PREFIX = "hermes_"
|
|
|
|
# Reverse map used ONLY when normalize_response runs on a transport that never
|
|
# built a request; real requests carry request-local ``_last_wire_aliases`` so a
|
|
# genuine tool named ``hermes_tool_search`` is never silently rewritten.
|
|
_LEGACY_ALIAS_FALLBACK = {
|
|
f"{_RESERVED_TOOL_ALIAS_PREFIX}{name}": name
|
|
for name in (*_OPENCODE_RESERVED_TOOL_NAMES, *_XAI_RESERVED_TOOL_NAMES)
|
|
}
|
|
_LEGACY_ALIAS_FALLBACK[_XAI_CLIENT_WEB_SEARCH_ALIAS] = "web_search"
|
|
|
|
|
|
def _is_opencode_responses_backend(params: Dict[str, Any]) -> bool:
|
|
"""True for opencode-zen/go providers, ``opencode-*`` families, or opencode.ai hosts."""
|
|
try:
|
|
from hermes_cli.models import opencode_provider_family
|
|
|
|
if opencode_provider_family(params.get("provider")) is not None:
|
|
return True
|
|
except Exception:
|
|
pass
|
|
try:
|
|
from utils import base_url_hostname
|
|
|
|
return base_url_hostname(str(params.get("base_url") or "")).lower() == "opencode.ai"
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
def _alias_reserved_tools(
|
|
response_tools: List[Dict[str, Any]],
|
|
reserved_names: Tuple[str, ...],
|
|
name_of: Callable[[dict], Any] = lambda t: t.get("name"),
|
|
rename: Callable[[dict, str], dict] = lambda t, alias: {**t, "name": alias},
|
|
) -> Tuple[List[Dict[str, Any]], Dict[str, str]]:
|
|
"""Alias provider-reserved function names on the wire.
|
|
|
|
Returns ``(rewritten_tools, {alias: original_name})``. An alias that is
|
|
already taken by a real tool gets a ``_2``/``_3`` suffix rather than
|
|
duplicating a wire name. ``name_of``/``rename`` adapt the tool shape
|
|
(Responses ``{name}`` by default; chat_completions passes ``function.name``).
|
|
"""
|
|
rewritten: List[Dict[str, Any]] = []
|
|
alias_map: Dict[str, str] = {}
|
|
taken = {name_of(tool) for tool in response_tools if isinstance(tool, dict) and name_of(tool)}
|
|
for tool in response_tools:
|
|
name = name_of(tool) if isinstance(tool, dict) else None
|
|
if name in reserved_names:
|
|
base = alias = f"{_RESERVED_TOOL_ALIAS_PREFIX}{name}"
|
|
suffix = 2
|
|
while alias in taken:
|
|
alias = f"{base}_{suffix}"
|
|
suffix += 1
|
|
taken.add(alias)
|
|
alias_map[alias] = name
|
|
rewritten.append(rename(tool, alias))
|
|
else:
|
|
rewritten.append(tool)
|
|
return rewritten, alias_map
|
|
|
|
|
|
def _xai_prefers_native_web_search() -> bool:
|
|
"""True when xAI Responses should use Grok's native ``web_search`` built-in.
|
|
|
|
Resolves via the web-search registry (config ``web.search_backend`` /
|
|
``web.backend``), falling back to the legacy ``_get_search_backend`` probe.
|
|
Fails closed to native (True) on any error — preserves the #48108 fix.
|
|
"""
|
|
try:
|
|
from agent.web_search_registry import get_active_search_provider
|
|
|
|
provider = get_active_search_provider()
|
|
if provider is not None:
|
|
return getattr(provider, "name", None) == "xai"
|
|
|
|
from tools.web_tools import _get_search_backend
|
|
|
|
return (_get_search_backend() or "").strip().lower() == "xai"
|
|
except Exception:
|
|
return True
|
|
|
|
|
|
def _alias_wire_tools(
|
|
response_tools: Any, params: Dict[str, Any], is_xai_responses: bool
|
|
) -> Tuple[Any, Dict[str, str]]:
|
|
"""Apply provider-reserved tool-name aliasing; returns ``(tools, {alias: original})``.
|
|
|
|
The alias map is wire provenance for THIS request — normalize_response
|
|
reverses only these. xAI: a client function named ``web_search`` collides
|
|
with Grok's native search (#48108); native mode swaps it 1:1 for the
|
|
built-in, client mode keeps Hermes dispatch under an alias.
|
|
"""
|
|
wire_aliases: Dict[str, str] = {}
|
|
def is_client_web_search(t: Any) -> bool:
|
|
return isinstance(t, dict) and t.get("name") == "web_search"
|
|
|
|
if is_xai_responses and response_tools and any(is_client_web_search(t) for t in response_tools):
|
|
if _xai_prefers_native_web_search():
|
|
response_tools = [t for t in response_tools if not is_client_web_search(t)] + [{"type": "web_search"}]
|
|
else:
|
|
response_tools = [
|
|
{**t, "name": _XAI_CLIENT_WEB_SEARCH_ALIAS} if is_client_web_search(t) else t for t in response_tools
|
|
]
|
|
wire_aliases[_XAI_CLIENT_WEB_SEARCH_ALIAS] = "web_search"
|
|
if response_tools and _is_opencode_responses_backend(params):
|
|
response_tools, _oc_aliases = _alias_reserved_tools(response_tools, _OPENCODE_RESERVED_TOOL_NAMES)
|
|
wire_aliases.update(_oc_aliases)
|
|
if is_xai_responses and response_tools:
|
|
response_tools, _xai_aliases = _alias_reserved_tools(response_tools, _XAI_RESERVED_TOOL_NAMES)
|
|
wire_aliases.update(_xai_aliases)
|
|
return response_tools, wire_aliases
|
|
|
|
|
|
def _resolve_reasoning(model: str, params: Dict[str, Any]) -> Tuple[Any, bool]:
|
|
"""``(effort, enabled)`` for the request, effort clamped to the endpoint's vocabulary.
|
|
|
|
Wire vocabularies live in agent.reasoning_effort; clamp_effort picks the
|
|
nearest weaker supported level and never escalates. A profile-declared
|
|
``()`` means "no reasoning parameters accepted" (such backends 400 on any
|
|
reasoning field) and disables reasoning outright.
|
|
"""
|
|
reasoning_effort, reasoning_enabled = "medium", True
|
|
reasoning_config = params.get("reasoning_config")
|
|
if reasoning_config and isinstance(reasoning_config, dict):
|
|
if reasoning_config.get("enabled") is False:
|
|
reasoning_enabled = False
|
|
elif reasoning_config.get("effort"):
|
|
reasoning_effort = reasoning_config["effort"]
|
|
|
|
if params.get("is_xai_responses", False):
|
|
from agent.model_metadata import is_grok_46_family
|
|
|
|
# Grok 4.6 accepts xhigh; older Grok tops out at high.
|
|
supported = XAI_GROK46_EFFORTS if is_grok_46_family(model) else XAI_LEGACY_EFFORTS
|
|
elif (params.get("provider") or "").strip().lower() == "actual":
|
|
supported = ACTUAL_RELAY_EFFORTS
|
|
else:
|
|
declared = _profile_declared_efforts(params.get("provider"), model, params.get("base_url"))
|
|
if declared is not None and not declared:
|
|
reasoning_enabled = False
|
|
supported = declared or codex_supported_efforts(model)
|
|
return clamp_effort(reasoning_effort, supported), reasoning_enabled
|
|
|
|
|
|
def _merge_extra_headers(kwargs: Dict[str, Any], **headers: str) -> None:
|
|
"""Merge str-coerced ``headers`` into ``kwargs['extra_headers']`` (SDK kwarg -> HTTP headers)."""
|
|
merged = _str_headers(kwargs.get("extra_headers"))
|
|
merged.update(headers)
|
|
kwargs["extra_headers"] = merged
|
|
|
|
|
|
_EXTENDED_PROMPT_CACHE_MODELS = (
|
|
"gpt-5.5-pro", "gpt-5.5", "gpt-5.4", "gpt-5.2",
|
|
"gpt-5.1-codex-max", "gpt-5.1-codex-mini", "gpt-5.1-chat-latest", "gpt-5.1-codex", "gpt-5.1",
|
|
"gpt-5-codex", "gpt-5", "gpt-4.1",
|
|
)
|
|
_EXTENDED_PROMPT_CACHE_MODEL_RE = re.compile(
|
|
rf"(?:^|[./:])(?:{'|'.join(re.escape(name) for name in _EXTENDED_PROMPT_CACHE_MODELS)})"
|
|
r"(?:-\d{4}-\d{2}-\d{2})?$"
|
|
)
|
|
|
|
|
|
def _default_prompt_cache_retention_for_request(model: str, base_url: Any) -> Optional[str]:
|
|
"""Return ``24h`` for supported hosts/models (Bedrock Mantle, Meta)."""
|
|
from utils import base_url_hostname
|
|
|
|
hostname = base_url_hostname(str(base_url or "")).lower()
|
|
# Meta Model API: caching is opt-in via prompt_cache_retention (0% hits without).
|
|
if hostname == "api.meta.ai":
|
|
return "24h"
|
|
parts = hostname.split(".")
|
|
is_bedrock_mantle = len(parts) == 4 and parts[0] == "bedrock-mantle" and bool(parts[1]) and parts[2:] == ["api", "aws"]
|
|
if not is_bedrock_mantle:
|
|
return None
|
|
normalized = str(model or "").strip().lower().replace("_", "-")
|
|
return "24h" if _EXTENDED_PROMPT_CACHE_MODEL_RE.search(normalized) else None
|
|
|
|
|
|
def _content_cache_key(
|
|
instructions: str, tools: Optional[List[Dict[str, Any]]], scope_id: str = ""
|
|
) -> Optional[str]:
|
|
"""``pck_<sha256[:24]>`` of (scope_id, instructions, name-sorted tools), or None if nothing static.
|
|
|
|
The key is a routing hint only. ``scope_id`` keeps unrelated sessions off one
|
|
bucket while letting timestamped cron fires of one job share a warm prefix.
|
|
"""
|
|
if not instructions and not tools:
|
|
return None
|
|
tools_part = ""
|
|
if tools:
|
|
sorted_tools = sorted(
|
|
(t for t in tools if isinstance(t, dict)),
|
|
key=lambda t: str(t.get("name") or t.get("type") or ""),
|
|
)
|
|
tools_part = json.dumps(sorted_tools, sort_keys=True, ensure_ascii=False, separators=(",", ":"))
|
|
# \x00 separators so a boundary can't be forged by content containing the same bytes.
|
|
content = f"{scope_id}\x00{instructions or ''}\x00{tools_part}"
|
|
digest = hashlib.sha256(content.encode("utf-8", errors="replace")).hexdigest()[:24]
|
|
return f"pck_{digest}"
|
|
|
|
|
|
def _profile_declared_efforts(provider: Any, model: Optional[str], base_url: Any = None) -> Optional[tuple]:
|
|
"""Provider-profile-declared reasoning-effort vocabulary, or None (fail-open).
|
|
|
|
Resolves by provider name, then by endpoint host (a custom provider pointed
|
|
at a known host gets that host's vocabulary). Lazy import: provider plugins
|
|
import this transport during registry discovery.
|
|
"""
|
|
try:
|
|
from providers import get_provider_profile
|
|
|
|
name = str(provider or "").strip().lower()
|
|
profile = get_provider_profile(name) if name else None
|
|
declared = profile.supported_reasoning_efforts(model) if profile is not None else None
|
|
if declared is None and base_url:
|
|
from agent.model_metadata import _infer_provider_from_url
|
|
|
|
inferred = _infer_provider_from_url(str(base_url))
|
|
if inferred and inferred != name:
|
|
inferred_profile = get_provider_profile(inferred)
|
|
if inferred_profile is not None:
|
|
declared = inferred_profile.supported_reasoning_efforts(model)
|
|
except Exception as exc:
|
|
logger.debug("profile-declared efforts lookup failed: %s", exc)
|
|
return None
|
|
return None if declared is None else tuple(declared)
|
|
|
|
|
|
def _is_azure_foundry_responses(params: Dict[str, Any]) -> bool:
|
|
"""True for Microsoft Foundry's Responses API (provider id, else host match — not substring)."""
|
|
from utils import base_url_host_matches
|
|
|
|
if str(params.get("provider") or "").strip().lower() == "azure-foundry":
|
|
return True
|
|
return base_url_host_matches(str(params.get("base_url") or ""), "services.ai.azure.com")
|
|
|
|
|
|
def _is_post_tool_replay(messages: Optional[List[Dict[str, Any]]]) -> bool:
|
|
"""True when ``messages`` end on a tool-result run issued by the preceding assistant turn.
|
|
|
|
Azure Foundry rejects only this post-tool follow-up shape when encrypted
|
|
reasoning is replayed, so the check is on the *trailing* messages (a whole-
|
|
history scan would make suppression sticky). Call ids are resolved the same
|
|
way ``_chat_messages_to_responses_input`` does (``call_id``, ``id``,
|
|
composite ``call_x|fc_y``, bare ``fc_`` item ids).
|
|
"""
|
|
from agent.codex_responses_adapter import _canonical_call_id_from_fc, _split_responses_tool_id
|
|
|
|
def _pair_ids(raw: Any, explicit: Any = None) -> set:
|
|
embedded_call_id, item_id = _split_responses_tool_id(raw)
|
|
ids = {embedded_call_id} if embedded_call_id else set()
|
|
if isinstance(explicit, str) and explicit.strip():
|
|
ids.add(explicit.strip())
|
|
if not ids and isinstance(raw, str) and raw.strip():
|
|
ids.add(raw.strip())
|
|
canonical = _canonical_call_id_from_fc(item_id)
|
|
if canonical:
|
|
ids.add(canonical)
|
|
return ids
|
|
|
|
trailing = set()
|
|
for msg in reversed(messages or ()):
|
|
role = msg.get("role") if isinstance(msg, dict) else None
|
|
if role == "system":
|
|
continue
|
|
if role == "tool":
|
|
ids = _pair_ids(msg.get("tool_call_id"))
|
|
if not ids:
|
|
return False
|
|
trailing |= ids
|
|
continue
|
|
# First non-tool message must be the assistant turn that issued the run.
|
|
if role != "assistant":
|
|
return False
|
|
return any(
|
|
trailing & _pair_ids(call.get("id"), call.get("call_id"))
|
|
for call in msg.get("tool_calls") or []
|
|
if isinstance(call, dict)
|
|
)
|
|
return False
|
|
|
|
|
|
def _native_compaction_active(context_management: Any) -> bool:
|
|
"""True only when the caller's eligibility gate produced a non-empty payload.
|
|
|
|
Every native-compaction wire effect hangs off this one predicate, so a
|
|
persisted checkpoint cannot keep reshaping requests after the gate closes.
|
|
"""
|
|
return isinstance(context_management, list) and bool(context_management)
|
|
|
|
|
|
class ResponsesApiTransport(ProviderTransport):
|
|
"""Transport for api_mode='codex_responses'."""
|
|
|
|
# Codex response.status -> OpenAI finish_reason (caller checks incomplete_details).
|
|
_STOP_REASON_MAP = {"completed": "stop", "incomplete": "length", "failed": "stop", "cancelled": "stop"}
|
|
|
|
# Issuer kind of the most recent build_kwargs/convert_messages call; fallback
|
|
# for normalize_response so captured reasoning items are stamped correctly.
|
|
_last_issuer_kind: Optional[str] = None
|
|
# ``{wire_alias: original_name}`` of the most recent build_kwargs call. None
|
|
# = no request built (use legacy map); {} = no aliases sent, no reverse rewrite.
|
|
_last_wire_aliases: Optional[Dict[str, str]] = None
|
|
|
|
@property
|
|
def api_mode(self) -> str:
|
|
return "codex_responses"
|
|
|
|
def _resolve_issuer_kind(self, params: Dict[str, Any]) -> str:
|
|
"""Classify the current Responses endpoint from transport params."""
|
|
from agent.codex_responses_adapter import _classify_responses_issuer
|
|
return _classify_responses_issuer(
|
|
is_xai_responses=params.get("is_xai_responses") is True,
|
|
is_github_responses=params.get("is_github_responses") is True,
|
|
is_codex_backend=params.get("is_codex_backend") is True,
|
|
base_url=params.get("base_url"),
|
|
)
|
|
|
|
def convert_messages(self, messages: List[Dict[str, Any]], **kwargs) -> Any:
|
|
"""Convert OpenAI chat messages to Responses API input items."""
|
|
from agent.codex_responses_adapter import _chat_messages_to_responses_input
|
|
issuer = self._resolve_issuer_kind(kwargs)
|
|
self._last_issuer_kind = issuer
|
|
return _chat_messages_to_responses_input(
|
|
messages,
|
|
is_xai_responses=kwargs.get("is_xai_responses") is True,
|
|
is_github_responses=kwargs.get("is_github_responses") is True,
|
|
replay_encrypted_reasoning=bool(kwargs.get("replay_encrypted_reasoning", True)),
|
|
current_issuer_kind=issuer,
|
|
native_compaction_eligible=_native_compaction_active(kwargs.get("context_management")),
|
|
)
|
|
|
|
def convert_tools(self, tools: List[Dict[str, Any]]) -> Any:
|
|
"""Convert OpenAI tool schemas to Responses API function definitions."""
|
|
from agent.codex_responses_adapter import _responses_tools
|
|
return _responses_tools(tools)
|
|
|
|
def build_kwargs(
|
|
self,
|
|
model: str,
|
|
messages: List[Dict[str, Any]],
|
|
tools: Optional[List[Dict[str, Any]]] = None,
|
|
**params,
|
|
) -> Dict[str, Any]:
|
|
"""Build Responses API kwargs (calls convert_messages/convert_tools internally).
|
|
|
|
params: instructions, reasoning_config ({effort, enabled}), session_id
|
|
(transcript id; Codex ``session_id`` header; cache-scope fallback),
|
|
cache_scope_id (rotation-stable logical scope, preferred for the cache
|
|
key / xAI conv header), max_tokens, timeout, request_overrides, provider,
|
|
base_url, is_github_responses, is_codex_backend, is_xai_responses,
|
|
github_reasoning_extra, context_management, replay_encrypted_reasoning.
|
|
"""
|
|
from agent.codex_responses_adapter import _chat_messages_to_responses_input, _responses_tools
|
|
from run_agent import DEFAULT_AGENT_IDENTITY
|
|
|
|
instructions = params.get("instructions", "")
|
|
payload_messages = messages
|
|
if not instructions and messages and messages[0].get("role") == "system":
|
|
instructions = str(messages[0].get("content") or "").strip()
|
|
payload_messages = messages[1:]
|
|
if not instructions:
|
|
instructions = DEFAULT_AGENT_IDENTITY
|
|
|
|
is_github_responses = params.get("is_github_responses") is True
|
|
is_codex_backend = params.get("is_codex_backend") is True
|
|
is_xai_responses = params.get("is_xai_responses") is True
|
|
replay_encrypted_reasoning = bool(params.get("replay_encrypted_reasoning", True))
|
|
# Foundry 400s on encrypted-reasoning replay only in the post-tool
|
|
# follow-up turn; suppress replay for exactly that shape.
|
|
if replay_encrypted_reasoning and _is_azure_foundry_responses(params) and _is_post_tool_replay(payload_messages):
|
|
replay_encrypted_reasoning = False
|
|
# Single source of truth: the same predicate decides whether
|
|
# context_management goes out AND whether the converter may replay a checkpoint.
|
|
context_management = params.get("context_management")
|
|
native_compaction_active = _native_compaction_active(context_management)
|
|
|
|
issuer_kind = self._resolve_issuer_kind(params)
|
|
self._last_issuer_kind = issuer_kind
|
|
reasoning_effort, reasoning_enabled = _resolve_reasoning(model, params)
|
|
response_tools, self._last_wire_aliases = _alias_wire_tools(_responses_tools(tools), params, is_xai_responses)
|
|
|
|
# Lazy: provider plugins import this transport during model_metadata init.
|
|
from agent.model_metadata import strip_codex_context_variant_suffix as _strip_ctx_variant
|
|
kwargs = {
|
|
# ``-900k`` picker variants are Hermes-side aliases; the backend knows only the base slug.
|
|
"model": _strip_ctx_variant(model),
|
|
"instructions": instructions,
|
|
"input": _chat_messages_to_responses_input(
|
|
payload_messages,
|
|
is_xai_responses=is_xai_responses,
|
|
is_github_responses=is_github_responses,
|
|
replay_encrypted_reasoning=replay_encrypted_reasoning,
|
|
current_issuer_kind=issuer_kind,
|
|
native_compaction_eligible=native_compaction_active,
|
|
),
|
|
"store": False,
|
|
}
|
|
# ``tools`` MUST be omitted when empty: the openai SDK iterates it without
|
|
# a None guard before any HTTP request (#32892).
|
|
if response_tools:
|
|
kwargs["tools"] = response_tools
|
|
kwargs["tool_choice"] = "auto"
|
|
kwargs["parallel_tool_calls"] = True
|
|
if native_compaction_active:
|
|
kwargs["context_management"] = context_management
|
|
|
|
session_id = params.get("session_id")
|
|
# Cache key is content-addressed (instructions + tools) within a logical
|
|
# scope — cache_scope_id survives compression rotation (#79017), and the
|
|
# cron per-fire timestamp is stripped. session_id itself stays untouched
|
|
# for transcript isolation (Codex ``session_id`` header below).
|
|
_cache_scope = _cache_scope_from_session_id(params.get("cache_scope_id") or session_id)
|
|
cache_key = _content_cache_key(instructions, response_tools, _cache_scope) or _cache_scope
|
|
# xAI takes prompt_cache_key in extra_body (below); GitHub Models opts out entirely.
|
|
if not is_github_responses and not is_xai_responses and cache_key:
|
|
kwargs["prompt_cache_key"] = cache_key
|
|
|
|
cache_retention = _default_prompt_cache_retention_for_request(model, params.get("base_url"))
|
|
if cache_retention:
|
|
kwargs.setdefault("prompt_cache_retention", cache_retention)
|
|
|
|
if reasoning_enabled and is_xai_responses:
|
|
from agent.model_metadata import grok_supports_reasoning_effort
|
|
|
|
kwargs["include"] = ["reasoning.encrypted_content"] if replay_encrypted_reasoning else []
|
|
# xAI 400s on ``reasoning.effort`` for models outside the allowlist.
|
|
if grok_supports_reasoning_effort(model):
|
|
kwargs["reasoning"] = {"effort": reasoning_effort}
|
|
elif reasoning_enabled:
|
|
if is_github_responses:
|
|
github_reasoning = params.get("github_reasoning_extra")
|
|
if github_reasoning is not None:
|
|
kwargs["reasoning"] = github_reasoning
|
|
else:
|
|
kwargs["reasoning"] = {"effort": reasoning_effort, "summary": "auto"}
|
|
kwargs["include"] = ["reasoning.encrypted_content"] if replay_encrypted_reasoning else []
|
|
elif not is_github_responses and not is_xai_responses:
|
|
kwargs["include"] = []
|
|
|
|
request_overrides = params.get("request_overrides")
|
|
if request_overrides:
|
|
kwargs.update(request_overrides)
|
|
|
|
_bound_prompt_cache_key_field(kwargs)
|
|
|
|
# Older xAI models reject ``service_tier`` (HTTP 400); only Grok 4.6
|
|
# accepts Priority Processing (#28490, #84799).
|
|
if is_xai_responses:
|
|
from agent.model_metadata import is_grok_46_family
|
|
|
|
if not (is_grok_46_family(model) and kwargs.get("service_tier") == "priority"):
|
|
kwargs.pop("service_tier", None)
|
|
|
|
# Forward per-request timeout to the SDK (providers.<id>.request_timeout_seconds).
|
|
timeout = kwargs.get("timeout", params.get("timeout"))
|
|
if isinstance(timeout, (int, float)) and not isinstance(timeout, bool) and 0 < float(timeout) < float("inf"):
|
|
kwargs["timeout"] = float(timeout)
|
|
else:
|
|
kwargs.pop("timeout", None)
|
|
|
|
if is_codex_backend:
|
|
# Codex rejects body-level extra_headers, but the SDK kwarg maps to
|
|
# HTTP headers. ``session_id`` = raw physical id (transcript identity);
|
|
# ``x-client-request-id`` mirrors the body cache key so both agree (#78941).
|
|
final_cache_key = kwargs.get("prompt_cache_key") or _bounded_prompt_cache_key(_cache_scope)
|
|
headers = {}
|
|
if session_id:
|
|
headers["session_id"] = str(session_id)
|
|
if final_cache_key:
|
|
headers["x-client-request-id"] = final_cache_key
|
|
if headers:
|
|
_merge_extra_headers(kwargs, **headers)
|
|
|
|
max_tokens = params.get("max_tokens")
|
|
if max_tokens is not None and not is_codex_backend:
|
|
kwargs["max_output_tokens"] = max_tokens
|
|
|
|
if is_xai_responses and session_id:
|
|
# Scoped like the body key so cron's per-fire timestamp doesn't pin
|
|
# each fire to a different xAI backend server (#78941).
|
|
_merge_extra_headers(kwargs, **{"x-grok-conv-id": _cache_scope})
|
|
# xAI reads prompt_cache_key from the body; sent via extra_body so it
|
|
# survives SDK builds whose Responses.stream() dropped the typed kwarg.
|
|
# An explicit request_overrides value (top-level kwarg) wins.
|
|
existing_extra_body = kwargs.get("extra_body")
|
|
kwargs["extra_body"] = dict(existing_extra_body) if isinstance(existing_extra_body, dict) else {}
|
|
kwargs["extra_body"].setdefault("prompt_cache_key", kwargs.get("prompt_cache_key", cache_key))
|
|
|
|
_bound_prompt_cache_key_field(kwargs.get("extra_body"))
|
|
return kwargs
|
|
|
|
def normalize_response(self, response: Any, **kwargs) -> NormalizedResponse:
|
|
"""Normalize Codex Responses API response to NormalizedResponse."""
|
|
from agent.codex_responses_adapter import _normalize_codex_response
|
|
|
|
# Explicit issuer if the caller knows it, else the stash from the paired
|
|
# build_kwargs/convert_messages call.
|
|
issuer_kind = kwargs.get("issuer_kind") or self._last_issuer_kind
|
|
msg, finish_reason = _normalize_codex_response(response, issuer_kind=issuer_kind)
|
|
|
|
tool_calls = None
|
|
if msg and msg.tool_calls:
|
|
tool_calls = []
|
|
alias_map = self._last_wire_aliases
|
|
for tc in msg.tool_calls:
|
|
provider_data = {
|
|
key: getattr(tc, key) for key in ("call_id", "response_item_id") if getattr(tc, key, None)
|
|
}
|
|
has_fn = hasattr(tc, "function")
|
|
name = tc.function.name if has_fn else getattr(tc, "name", "")
|
|
# Undo only aliases THIS request emitted; the legacy map applies
|
|
# solely to normalize-only call sites that never built a request.
|
|
if alias_map is None:
|
|
name = _LEGACY_ALIAS_FALLBACK.get(name, name)
|
|
elif name in alias_map:
|
|
name = alias_map[name]
|
|
tool_calls.append(ToolCall(
|
|
id=tc.id if hasattr(tc, "id") else (name or None),
|
|
name=name,
|
|
arguments=tc.function.arguments if has_fn else getattr(tc, "arguments", "{}"),
|
|
provider_data=provider_data or None,
|
|
))
|
|
|
|
provider_data = {}
|
|
if msg:
|
|
for key in ("codex_reasoning_items", "codex_message_items", "reasoning_details"):
|
|
value = getattr(msg, key, None)
|
|
if value:
|
|
provider_data[key] = value
|
|
|
|
return NormalizedResponse(
|
|
content=msg.content if msg else None,
|
|
tool_calls=tool_calls,
|
|
finish_reason=finish_reason or "stop",
|
|
reasoning=getattr(msg, "reasoning", None) if msg else None,
|
|
usage=None, # Codex usage is extracted separately in normalize_usage()
|
|
provider_data=provider_data or None,
|
|
)
|
|
|
|
def validate_response(self, response: Any) -> bool:
|
|
"""True if response.output is a non-empty list, or a terminal content_filter refusal.
|
|
|
|
A status=incomplete / reason=content_filter response with no output is a
|
|
provider refusal signal that must reach normalization, not a retry.
|
|
Does NOT check output_text fallback — the caller handles that.
|
|
"""
|
|
if response is None:
|
|
return False
|
|
output = getattr(response, "output", None)
|
|
if isinstance(output, list) and output:
|
|
return True
|
|
status = str(getattr(response, "status", "") or "").strip().lower()
|
|
details = getattr(response, "incomplete_details", None)
|
|
raw_reason = details.get("reason") if isinstance(details, dict) else getattr(details, "reason", "")
|
|
return status == "incomplete" and str(raw_reason or "").strip().lower() == "content_filter"
|
|
|
|
def preflight_kwargs(
|
|
self,
|
|
api_kwargs: Any,
|
|
*,
|
|
allow_stream: bool = False,
|
|
is_github_responses: bool = False,
|
|
sanitize_harmony_tokens: bool = False,
|
|
) -> dict:
|
|
"""Validate and sanitize Codex API kwargs before the call.
|
|
|
|
``sanitize_harmony_tokens`` is enabled only for the ChatGPT Codex
|
|
backend, which rejects literal reserved Harmony wire tokens in text.
|
|
"""
|
|
from agent.codex_responses_adapter import _preflight_codex_api_kwargs
|
|
|
|
normalized = _preflight_codex_api_kwargs(
|
|
api_kwargs, allow_stream=allow_stream, is_github_responses=is_github_responses,
|
|
sanitize_harmony_tokens=sanitize_harmony_tokens,
|
|
)
|
|
_bound_prompt_cache_key_field(normalized)
|
|
_bound_prompt_cache_key_field(normalized.get("extra_body"))
|
|
return normalized
|
|
|
|
|
|
# Auto-register on import
|
|
from agent.transports import register_transport # noqa: E402
|
|
|
|
register_transport("codex_responses", ResponsesApiTransport)
|