diff --git a/agent/transports/__init__.py b/agent/transports/__init__.py index b606da7fec..23557be795 100644 --- a/agent/transports/__init__.py +++ b/agent/transports/__init__.py @@ -1,21 +1,20 @@ -"""Transport layer types and registry for provider response normalization. +"""Transport registry for provider response normalization. -Usage: - from agent.transports import get_transport transport = get_transport("anthropic_messages") result = transport.normalize_response(raw_response) """ -from agent.transports.types import ( +from agent.transports.types import ( # noqa: F401 NormalizedResponse, ToolCall, Usage, build_tool_call, map_finish_reason, -) # noqa: F401 +) _REGISTRY: dict = {} _discovered: bool = False +_TRANSPORT_MODULES = ("anthropic", "codex", "chat_completions", "bedrock") def register_transport(api_mode: str, transport_cls: type) -> None: @@ -24,45 +23,26 @@ def register_transport(api_mode: str, transport_cls: type) -> None: def get_transport(api_mode: str): - """Get a transport instance for the given api_mode. - - Returns None if no transport is registered for this api_mode. - This allows gradual migration — call sites can check for None - and fall back to the legacy code path. - """ - global _discovered + """Return a transport instance for ``api_mode``, or None so callers can fall back to the legacy path.""" if not _discovered: _discover_transports() cls = _REGISTRY.get(api_mode) if cls is None: - # The registry can be partially populated when a specific transport - # module was imported directly (for example chat_completions before - # codex). Discover on misses, not only when the registry is empty, so - # test/order-dependent imports do not make valid api_modes unavailable. + # A directly-imported transport module leaves the registry partially + # populated; discover on misses so import order can't hide a valid api_mode. _discover_transports() cls = _REGISTRY.get(api_mode) - if cls is None: - return None - return cls() + return None if cls is None else cls() def _discover_transports() -> None: """Import all transport modules to trigger auto-registration.""" global _discovered _discovered = True - try: - import agent.transports.anthropic # noqa: F401 - except ImportError: - pass - try: - import agent.transports.codex # noqa: F401 - except ImportError: - pass - try: - import agent.transports.chat_completions # noqa: F401 - except ImportError: - pass - try: - import agent.transports.bedrock # noqa: F401 - except ImportError: - pass + import importlib + + for name in _TRANSPORT_MODULES: + try: + importlib.import_module(f"agent.transports.{name}") + except ImportError: + pass diff --git a/agent/transports/anthropic.py b/agent/transports/anthropic.py index 3c42d7ded2..d6740ac8af 100644 --- a/agent/transports/anthropic.py +++ b/agent/transports/anthropic.py @@ -1,36 +1,57 @@ """Anthropic Messages API transport. -Delegates to the existing adapter functions in agent/anthropic_adapter.py. -This transport owns format conversion and normalization — NOT client lifecycle. +Delegates format conversion to agent/anthropic_adapter.py; owns normalization, +not client lifecycle. """ from typing import Any, Dict, List, Optional from agent.transports.base import ProviderTransport -from agent.transports.types import NormalizedResponse +from agent.transports.types import NormalizedResponse, ToolCall + +_MCP_PREFIX = "mcp__" + + +def _unprefix_oauth_tool_name(name: str) -> str: + """Reverse the OAuth-wire ``mcp__`` prefix back to the registered tool name. + + Two originals map onto one wire name (``mcp__read_file`` <- ``read_file``; + ``mcp__linear_get_issue`` <- ``mcp_linear_get_issue``), so resolve by registry + lookup, never rewriting a name that already resolves natively (GH-25255). + OAuth wire aliases (e.g. chat_history_lookup -> session_search) are checked + LAST so a real tool registered under the wire name still wins. + """ + from agent.anthropic_adapter import _OAUTH_TOOL_NAME_REVERSE_ALIASES + from tools.registry import registry as _tool_registry + + bare = name[len(_MCP_PREFIX):] + for candidate in (name, "mcp_" + bare, bare): + if _tool_registry.get_entry(candidate): + return candidate + return _OAUTH_TOOL_NAME_REVERSE_ALIASES.get(bare, name) class AnthropicTransport(ProviderTransport): - """Transport for api_mode='anthropic_messages'. + """Transport for api_mode='anthropic_messages'.""" - Wraps the existing functions in anthropic_adapter.py behind the - ProviderTransport ABC. Each method delegates — no logic is duplicated. - """ + _STOP_REASON_MAP = { + "end_turn": "stop", + "tool_use": "tool_calls", + "max_tokens": "length", + "stop_sequence": "stop", + "refusal": "content_filter", + "model_context_window_exceeded": "length", + } @property def api_mode(self) -> str: return "anthropic_messages" def convert_messages(self, messages: List[Dict[str, Any]], **kwargs) -> Any: - """Convert OpenAI messages to Anthropic (system, messages) tuple. - - kwargs: - base_url: Optional[str] — affects thinking signature handling. - """ + """Convert OpenAI messages to an Anthropic (system, messages) tuple; ``base_url`` affects thinking-signature handling.""" from agent.anthropic_adapter import convert_messages_to_anthropic - base_url = kwargs.get("base_url") - return convert_messages_to_anthropic(messages, base_url=base_url) + return convert_messages_to_anthropic(messages, base_url=kwargs.get("base_url")) def convert_tools(self, tools: List[Dict[str, Any]]) -> Any: """Convert OpenAI tool schemas to Anthropic input_schema format.""" @@ -45,21 +66,7 @@ class AnthropicTransport(ProviderTransport): tools: Optional[List[Dict[str, Any]]] = None, **params, ) -> Dict[str, Any]: - """Build Anthropic messages.create() kwargs. - - Calls convert_messages and convert_tools internally. - - params (all optional): - max_tokens: int - reasoning_config: dict | None - tool_choice: str | None - is_oauth: bool - preserve_dots: bool - context_length: int | None - base_url: str | None - fast_mode: bool - drop_context_1m_beta: bool - """ + """Build Anthropic messages.create() kwargs (converts messages and tools internally).""" from agent.anthropic_adapter import build_anthropic_kwargs return build_anthropic_kwargs( @@ -78,46 +85,24 @@ class AnthropicTransport(ProviderTransport): ) def normalize_response(self, response: Any, **kwargs) -> NormalizedResponse: - """Normalize Anthropic response to NormalizedResponse. - - Parses content blocks (text, thinking, tool_use), maps stop_reason - to OpenAI finish_reason, and collects reasoning_details in provider_data. - """ + """Parse content blocks (text/thinking/tool_use), map stop_reason, collect reasoning_details.""" import json - from agent.anthropic_adapter import ( - _OAUTH_TOOL_NAME_REVERSE_ALIASES, - _sanitize_replay_block, - _to_plain_data, - ) - from agent.transports.types import ToolCall + from agent.anthropic_adapter import _sanitize_replay_block, _to_plain_data strip_tool_prefix = kwargs.get("strip_tool_prefix", False) - _MCP_PREFIX = "mcp__" - - text_parts = [] - reasoning_parts = [] - reasoning_details = [] - tool_calls = [] - # Verbatim, order-preserving copy of every content block in the turn. - # Anthropic signs each thinking block against the turn content that - # PRECEDES it at its position; when a turn interleaves thinking and - # tool_use (adaptive/interleaved thinking, Claude 4.6+), the parallel - # reasoning_details + tool_calls lists below lose that cross-type - # ordering. Replaying the latest assistant message in the wrong order - # invalidates the signatures -> HTTP 400 "thinking ... blocks in the - # latest assistant message cannot be modified". Preserve the exact - # block sequence here so the adapter can replay it unchanged. See - # tests/agent/test_anthropic_thinking_block_order.py. + text_parts, reasoning_parts, reasoning_details, tool_calls = [], [], [], [] + # Anthropic signs each thinking block against the blocks that PRECEDE it. + # When thinking interleaves with tool_use, the parallel reasoning_details + + # tool_calls lists lose that ordering and replay -> HTTP 400 "thinking ... + # blocks cannot be modified". Keep the exact sequence for the adapter. ordered_blocks = [] for block in response.content: block_dict = _to_plain_data(block) clean_block = None if isinstance(block_dict, dict): - # Sanitize at capture so output-only SDK fields (parsed_output, - # caller, citations=None, …) never persist to state.db and leak - # back as request input on replay → HTTP 400 "Extra inputs are - # not permitted". Defence-in-depth with the replay-side sanitize. + # Sanitize at capture so output-only SDK fields never persist to + # state.db and leak back as request input on replay (HTTP 400). clean_block = _sanitize_replay_block(block_dict) if clean_block is not None: ordered_blocks.append(clean_block) @@ -126,9 +111,7 @@ class AnthropicTransport(ProviderTransport): elif block.type in ("thinking", "redacted_thinking"): if block.type == "thinking": reasoning_parts.append(block.thinking) - # Use the sanitized block (clean_block) for reasoning_details too, - # since _extract_preserved_thinking_blocks replays these on the - # non-ordered path. Falls back to raw only if sanitize dropped it. + # Prefer the sanitized block (replayed on the non-ordered path); raw only if sanitize dropped it. if isinstance(clean_block, dict): reasoning_details.append(clean_block) elif isinstance(block_dict, dict): @@ -136,126 +119,49 @@ class AnthropicTransport(ProviderTransport): elif block.type == "tool_use": name = block.name if strip_tool_prefix and name.startswith(_MCP_PREFIX): - # On the OAuth wire every tool carries a double-underscore - # ``mcp__`` prefix (added in build_anthropic_kwargs to avoid - # Anthropic's single-underscore third-party classifier). - # Reverse it back to the name the registry/dispatcher knows. - # Two original forms map onto the same ``mcp__`` wire name: - # ``mcp__read_file`` <- bare native tool ``read_file`` - # ``mcp__linear_get_issue`` <- MCP server tool - # ``mcp_linear_get_issue`` - # Resolve by registry lookup, preferring whichever original - # is actually registered; never rewrite a name the LLM used - # that already resolves natively. GH-25255. - from tools.registry import registry as _tool_registry - if not _tool_registry.get_entry(name): - bare = name[len(_MCP_PREFIX):] # read_file - single = "mcp_" + bare # mcp_read_file / mcp_linear_get_issue - if _tool_registry.get_entry(single): - name = single - elif _tool_registry.get_entry(bare): - name = bare - elif bare in _OAUTH_TOOL_NAME_REVERSE_ALIASES: - # OAuth wire alias (e.g. chat_history_lookup -> - # session_search, #65365). Checked LAST so a real - # tool actually registered under the wire name - # still wins — same GH-25255 precedence. - name = _OAUTH_TOOL_NAME_REVERSE_ALIASES[bare] - tool_calls.append( - ToolCall( - id=block.id, - name=name, - arguments=json.dumps(block.input), - ) - ) - - finish_reason = self._STOP_REASON_MAP.get(response.stop_reason, "stop") + name = _unprefix_oauth_tool_name(name) + tool_calls.append(ToolCall(id=block.id, name=name, arguments=json.dumps(block.input))) provider_data = {} if reasoning_details: provider_data["reasoning_details"] = reasoning_details - # Only worth carrying the ordered-blocks channel when the turn - # actually interleaves signed thinking with tool_use — that's the - # only shape the parallel lists reconstruct incorrectly. A turn that - # is purely text, or thinking-then-tools with a single leading - # thinking block, replays correctly without it. + # Carry the ordered channel only for the one shape the parallel lists + # reconstruct wrongly: signed thinking interleaved with tool_use. _has_signed_thinking = any( - isinstance(b, dict) - and b.get("type") in ("thinking", "redacted_thinking") - and (b.get("signature") or b.get("data")) + isinstance(b, dict) and b.get("type") in ("thinking", "redacted_thinking") and (b.get("signature") or b.get("data")) for b in ordered_blocks ) - _has_tool_use = any( - isinstance(b, dict) and b.get("type") == "tool_use" - for b in ordered_blocks - ) - if _has_signed_thinking and _has_tool_use: + if _has_signed_thinking and any(isinstance(b, dict) and b.get("type") == "tool_use" for b in ordered_blocks): provider_data["anthropic_content_blocks"] = ordered_blocks return NormalizedResponse( content="\n".join(text_parts) if text_parts else None, tool_calls=tool_calls or None, - finish_reason=finish_reason, + finish_reason=self.map_finish_reason(response.stop_reason), reasoning="\n\n".join(reasoning_parts) if reasoning_parts else None, usage=None, provider_data=provider_data or None, ) def validate_response(self, response: Any) -> bool: - """Check Anthropic response structure is valid. - - An empty content list is legitimate for terminal stop reasons that - carry no text payload: - - - ``end_turn`` — the model's canonical "nothing more to add" after a - tool turn that already delivered the user-facing text. - - ``refusal`` — the model declined to respond (Claude 4.5+). The - Messages API returns an empty ``content`` list with this stop - reason. Treating it as invalid sends a deterministic refusal into - the invalid-response retry loop, which reproduces the refusal on - every attempt and surfaces a misleading "rate limited / invalid - response" error instead of the refusal. ``normalize_response`` maps - ``refusal`` → ``content_filter`` so the agent loop's refusal handler - can surface it. - - Treating either as invalid falsely retries a completed response. - """ - if response is None: - return False - content_blocks = getattr(response, "content", None) + """Structural check. An empty content list is legitimate for ``end_turn`` (nothing to add + after a tool turn) and ``refusal`` (Claude 4.5+ declines with empty content); treating + either as invalid would retry a completed/deterministic response forever.""" + content_blocks = getattr(response, "content", None) if response is not None else None if not isinstance(content_blocks, list): return False - if not content_blocks: - return getattr(response, "stop_reason", None) in {"end_turn", "refusal"} - return True + return bool(content_blocks) or getattr(response, "stop_reason", None) in {"end_turn", "refusal"} def extract_cache_stats(self, response: Any) -> Optional[Dict[str, int]]: - """Extract Anthropic cache_read and cache_creation token counts.""" + """Anthropic cache_read / cache_creation token counts.""" usage = getattr(response, "usage", None) if usage is None: return None cached = getattr(usage, "cache_read_input_tokens", 0) or 0 written = getattr(usage, "cache_creation_input_tokens", 0) or 0 - if cached or written: - return {"cached_tokens": cached, "creation_tokens": written} - return None - - # Promote the adapter's canonical mapping to module level so it's shared - _STOP_REASON_MAP = { - "end_turn": "stop", - "tool_use": "tool_calls", - "max_tokens": "length", - "stop_sequence": "stop", - "refusal": "content_filter", - "model_context_window_exceeded": "length", - } - - def map_finish_reason(self, raw_reason: str) -> str: - """Map Anthropic stop_reason to OpenAI finish_reason.""" - return self._STOP_REASON_MAP.get(raw_reason, "stop") + return {"cached_tokens": cached, "creation_tokens": written} if cached or written else None -# Auto-register on import from agent.transports import register_transport # noqa: E402 register_transport("anthropic_messages", AnthropicTransport) diff --git a/agent/transports/base.py b/agent/transports/base.py index b516967b6a..aae72b5ee0 100644 --- a/agent/transports/base.py +++ b/agent/transports/base.py @@ -1,10 +1,9 @@ """Abstract base for provider transports. A transport owns the data path for one api_mode: - convert_messages → convert_tools → build_kwargs → normalize_response - -It does NOT own: client construction, streaming, credential refresh, -prompt caching, interrupt handling, or retry logic. Those stay on AIAgent. + convert_messages -> convert_tools -> build_kwargs -> normalize_response +It does NOT own client construction, streaming, credential refresh, prompt +caching, interrupt handling, or retry logic — those stay on AIAgent. """ from abc import ABC, abstractmethod @@ -16,28 +15,22 @@ from agent.transports.types import NormalizedResponse class ProviderTransport(ABC): """Base class for provider-specific format conversion and normalization.""" + # Provider stop_reason -> OpenAI finish_reason. ``None`` means the provider + # already speaks OpenAI vocabulary and map_finish_reason passes through. + _STOP_REASON_MAP: Optional[Dict[str, str]] = None + @property @abstractmethod def api_mode(self) -> str: """The api_mode string this transport handles (e.g. 'anthropic_messages').""" - ... @abstractmethod def convert_messages(self, messages: List[Dict[str, Any]], **kwargs) -> Any: - """Convert OpenAI-format messages to provider-native format. - - Returns provider-specific structure (e.g. (system, messages) for Anthropic, - or the messages list unchanged for chat_completions). - """ - ... + """Convert OpenAI-format messages to the provider-native structure (e.g. (system, messages) for Anthropic).""" @abstractmethod def convert_tools(self, tools: List[Dict[str, Any]]) -> Any: - """Convert OpenAI-format tool definitions to provider-native format. - - Returns provider-specific tool list (e.g. Anthropic input_schema format). - """ - ... + """Convert OpenAI-format tool definitions to provider-native format.""" @abstractmethod def build_kwargs( @@ -47,43 +40,20 @@ class ProviderTransport(ABC): tools: Optional[List[Dict[str, Any]]] = None, **params, ) -> Dict[str, Any]: - """Build the complete API call kwargs dict. - - This is the primary entry point — it typically calls convert_messages() - and convert_tools() internally, then adds model-specific config. - - Returns a dict ready to be passed to the provider's SDK client. - """ - ... + """Primary entry point: convert messages/tools and return kwargs ready for the provider SDK.""" @abstractmethod def normalize_response(self, response: Any, **kwargs) -> NormalizedResponse: - """Normalize a raw provider response to the shared NormalizedResponse type. - - This is the only method that returns a transport-layer type. - """ - ... + """Normalize a raw provider response to NormalizedResponse (the only transport-layer return type).""" def validate_response(self, response: Any) -> bool: - """Optional: check if the raw response is structurally valid. - - Returns True if valid, False if the response should be treated as invalid. - Default implementation always returns True. - """ + """Optional structural validity check; default accepts everything.""" return True def extract_cache_stats(self, response: Any) -> Optional[Dict[str, int]]: - """Optional: extract provider-specific cache hit/creation stats. - - Returns dict with 'cached_tokens' and 'creation_tokens', or None. - Default returns None. - """ + """Optional: ``{'cached_tokens', 'creation_tokens'}`` or None (default).""" return None def map_finish_reason(self, raw_reason: str) -> str: - """Optional: map provider-specific stop reason to OpenAI equivalent. - - Default returns the raw reason unchanged. Override for providers - with different stop reason vocabularies. - """ - return raw_reason + """Map a provider stop reason via ``_STOP_REASON_MAP`` (unknown -> 'stop'); passthrough when no map.""" + return raw_reason if self._STOP_REASON_MAP is None else self._STOP_REASON_MAP.get(raw_reason, "stop") diff --git a/agent/transports/bedrock.py b/agent/transports/bedrock.py index d9caf0f1e3..e7aa316ac9 100644 --- a/agent/transports/bedrock.py +++ b/agent/transports/bedrock.py @@ -1,9 +1,7 @@ """AWS Bedrock Converse API transport. -Delegates to the existing adapter functions in agent/bedrock_adapter.py. -Bedrock uses its own boto3 client (not the OpenAI SDK), so the transport -owns format conversion and normalization, while client construction and -boto3 calls stay on AIAgent. +Delegates format conversion to agent/bedrock_adapter.py. Bedrock uses its own +boto3 client, so client construction and calls stay on AIAgent. """ from typing import Any, Dict, List, Optional @@ -15,6 +13,16 @@ from agent.transports.types import NormalizedResponse, ToolCall, Usage class BedrockTransport(ProviderTransport): """Transport for api_mode='bedrock_converse'.""" + # The adapter already maps inside normalize_converse_response; this serves raw-response access. + _STOP_REASON_MAP = { + "end_turn": "stop", + "tool_use": "tool_calls", + "max_tokens": "length", + "stop_sequence": "stop", + "guardrail_intervened": "content_filter", + "content_filtered": "content_filter", + } + @property def api_mode(self) -> str: return "bedrock_converse" @@ -36,76 +44,36 @@ class BedrockTransport(ProviderTransport): tools: Optional[List[Dict[str, Any]]] = None, **params, ) -> Dict[str, Any]: - """Build Bedrock converse() kwargs. - - Calls convert_messages and convert_tools internally. - - params: - max_tokens: int — output token limit (default 4096) - temperature: float | None - guardrail_config: dict | None — Bedrock guardrails - region: str — AWS region (default 'us-east-1') - """ + """Build converse() kwargs; params: max_tokens (4096), temperature, guardrail_config, region ('us-east-1').""" from agent.bedrock_adapter import build_converse_kwargs - region = params.get("region", "us-east-1") - guardrail = params.get("guardrail_config") - kwargs = build_converse_kwargs( model=model, messages=messages, tools=tools, max_tokens=params.get("max_tokens", 4096), temperature=params.get("temperature"), - guardrail_config=guardrail, + guardrail_config=params.get("guardrail_config"), ) # Sentinel keys for dispatch — agent pops these before the boto3 call kwargs["__bedrock_converse__"] = True - kwargs["__bedrock_region__"] = region + kwargs["__bedrock_region__"] = params.get("region", "us-east-1") return kwargs def normalize_response(self, response: Any, **kwargs) -> NormalizedResponse: - """Normalize Bedrock response to NormalizedResponse. - - Handles two shapes: - 1. Raw boto3 dict (from direct converse() calls) - 2. Already-normalized SimpleNamespace with .choices (from dispatch site) - """ + """Normalize either a raw boto3 dict or an already-normalized SimpleNamespace with .choices.""" from agent.bedrock_adapter import normalize_converse_response - # Normalize to OpenAI-compatible SimpleNamespace - if hasattr(response, "choices") and response.choices: - # Already normalized at dispatch site - ns = response - else: - # Raw boto3 dict - ns = normalize_converse_response(response) - + ns = response if hasattr(response, "choices") and response.choices else normalize_converse_response(response) choice = ns.choices[0] msg = choice.message - finish_reason = choice.finish_reason or "stop" tool_calls = None if msg.tool_calls: - tool_calls = [ - ToolCall( - id=tc.id, - name=tc.function.name, - arguments=tc.function.arguments, - ) - for tc in msg.tool_calls - ] - + tool_calls = [ToolCall(id=tc.id, name=tc.function.name, arguments=tc.function.arguments) for tc in msg.tool_calls] usage = None if hasattr(ns, "usage") and ns.usage: - u = ns.usage - usage = Usage( - prompt_tokens=getattr(u, "prompt_tokens", 0) or 0, - completion_tokens=getattr(u, "completion_tokens", 0) or 0, - total_tokens=getattr(u, "total_tokens", 0) or 0, - ) - - reasoning = getattr(msg, "reasoning", None) or getattr(msg, "reasoning_content", None) + usage = Usage.from_openai(ns.usage) provider_data = {} if getattr(msg, "reasoning_details", None): @@ -116,46 +84,19 @@ class BedrockTransport(ProviderTransport): return NormalizedResponse( content=msg.content, tool_calls=tool_calls, - finish_reason=finish_reason, - reasoning=reasoning, + finish_reason=choice.finish_reason or "stop", + reasoning=getattr(msg, "reasoning", None) or getattr(msg, "reasoning_content", None), usage=usage, provider_data=provider_data or None, ) def validate_response(self, response: Any) -> bool: - """Check Bedrock response structure. - - After normalize_converse_response, the response has OpenAI-compatible - .choices — same check as chat_completions. - """ - if response is None: - return False - # Raw Bedrock dict response — check for 'output' key + """Raw Bedrock dict needs an 'output' key; a normalized namespace needs non-empty .choices.""" if isinstance(response, dict): return "output" in response - # Already-normalized SimpleNamespace - if hasattr(response, "choices"): - return bool(response.choices) - return False - - def map_finish_reason(self, raw_reason: str) -> str: - """Map Bedrock stop reason to OpenAI finish_reason. - - The adapter already does this mapping inside normalize_converse_response, - so this is only used for direct access to raw responses. - """ - _MAP = { - "end_turn": "stop", - "tool_use": "tool_calls", - "max_tokens": "length", - "stop_sequence": "stop", - "guardrail_intervened": "content_filter", - "content_filtered": "content_filter", - } - return _MAP.get(raw_reason, "stop") + return bool(getattr(response, "choices", None)) if response is not None else False -# Auto-register on import from agent.transports import register_transport # noqa: E402 register_transport("bedrock_converse", BedrockTransport) diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index 25b41be44d..541a3af56f 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -1,12 +1,7 @@ -"""OpenAI Chat Completions transport. +"""OpenAI Chat Completions transport (default api_mode for OpenAI-compatible providers). -Handles the default api_mode ('chat_completions') used by ~16 OpenAI-compatible -providers (OpenRouter, Nous, NVIDIA, Qwen, Ollama, DeepSeek, xAI, Kimi, etc.). - -Messages and tools are already in OpenAI format — convert_messages and -convert_tools are near-identity. The complexity lives in build_kwargs -which has provider-specific conditionals for max_tokens defaults, -reasoning configuration, temperature handling, and extra_body assembly. +Messages/tools are already OpenAI-shaped, so convert_* are near-identity; the +provider-specific work lives in build_kwargs (max_tokens, reasoning, extra_body). """ import json @@ -27,70 +22,42 @@ from agent.prompt_builder import DEVELOPER_ROLE_MODELS from agent.transports.base import ProviderTransport from agent.transports.types import NormalizedResponse, ToolCall, Usage -# xAI's chat-completions API reserves the function name ``tool_search`` for -# its own server-side tool and rejects any request declaring a client -# function with that name (HTTP 400 "The function name tool_search is -# reserved for the tool_search tool", #95003). The Tool Search bridge -# (tools/tool_search.py) assembles its client-side discovery tool under the -# same literal name for every provider, so Grok providers are unusable -# whenever the bridge is active. Mirror the web_search treatment in -# transports/codex.py (_rename_client_web_search_for_xai): alias the wire -# declaration and map the alias back in normalize_response. The alias value -# matches _CODEX_TOOL_SEARCH_ALIAS from the Codex-side fix for the same -# reserved-name class (#83122) so the two transports stay consistent. +# xAI reserves the function name ``tool_search`` for its server-side tool and +# rejects client declarations of it (HTTP 400, #95003); alias it on the wire and +# map back in normalize_response. Value matches the Codex-side alias (#83122). _XAI_TOOL_SEARCH_ALIAS = "hermes_tool_search" +# Persistence-only / cross-transport message keys that strict OpenAI-compatible +# providers reject with HTTP 400 ("Extra inputs are not permitted"). +_STRIP_MSG_KEYS = ( + "codex_reasoning_items", "codex_message_items", "tool_name", "effect_disposition", "timestamp", + "platform_message_id", "api_content", "anthropic_content_blocks", "bedrock_content_blocks", +) +_STRIP_TC_KEYS = ("call_id", "response_item_id") +_HIGH_EFFORTS = {"high", "xhigh", "max", "ultra"} + def _rename_tool_search_bridge_for_xai( tools: list[dict[str, Any]], ) -> tuple[list[dict[str, Any]], dict[str, str]]: - """Rename the client ``tool_search`` bridge declaration to a wire alias. + """Alias the client ``tool_search`` declaration for xAI. - Only the wire name changes: descriptions, schemas, and the other two - bridge names (``tool_describe`` / ``tool_call`` — not reserved by xAI) - pass through untouched. Returns ``(rewritten_tools, alias_map)`` where - ``alias_map`` maps each alias THIS request emits back to the original - name; the caller stashes it on the transport so ``normalize_response`` - only reverses aliases that were actually sent. If a real tool already - occupies ``hermes_tool_search``, the bridge takes a ``_2``/``_3`` - suffix instead of duplicating a wire name. + Returns ``(rewritten_tools, alias_map)`` where alias_map maps each alias + emitted by THIS request back to ``tool_search``. If a real tool already + holds ``hermes_tool_search``, the bridge takes a ``_2``/``_3`` suffix. """ - rewritten: list[dict[str, Any]] = [] - alias_map: dict[str, str] = {} - taken = { - (tool.get("function") or {}).get("name") - for tool in tools - if isinstance(tool, dict) - } - taken.discard(None) - for tool in tools: - if ( - isinstance(tool, dict) - and (tool.get("function") or {}).get("name") == "tool_search" - ): - alias = _XAI_TOOL_SEARCH_ALIAS - suffix = 2 - while alias in taken: - alias = f"{_XAI_TOOL_SEARCH_ALIAS}_{suffix}" - suffix += 1 - taken.add(alias) - alias_map[alias] = "tool_search" - aliased = dict(tool) - aliased["function"] = {**tool["function"], "name": alias} - rewritten.append(aliased) - else: - rewritten.append(tool) - return rewritten, alias_map + from agent.transports.codex import _alias_reserved_tools + + return _alias_reserved_tools( + tools, + ("tool_search",), + name_of=lambda t: (t.get("function") or {}).get("name"), + rename=lambda t, alias: {**t, "function": {**t["function"], "name": alias}}, + ) def _static_prompt_instructions(messages: list[dict[str, Any]]) -> str: - """Return the stable system/developer prefix used for cache routing. - - Chat Completions carries instructions in its message list rather than a - separate ``instructions`` field. Only a leading system/developer message - is static by contract; later messages are conversation state and must not - split a warm prefix bucket on every turn. - """ + """Stable leading system/developer prefix used for cache routing (later messages are conversation state).""" if not messages or not isinstance(messages[0], dict): return "" first = messages[0] @@ -114,51 +81,28 @@ def _add_prompt_cache_key( session_id: str | None = None, cache_scope_id: str | None = None, ) -> None: - """Add a content-addressed key only for an explicitly capable endpoint. + """Add a content-addressed ``prompt_cache_key`` only for a capable endpoint. - ``cache_scope_id``, when provided, is the rotation-stable logical scope - (compression-lineage root — agent/prompt_cache_scope.py) and takes - precedence over the physical ``session_id`` so the key survives - context-compression session rotation (#79017). + ``cache_scope_id`` (compression-lineage root) beats ``session_id`` so the key + survives context-compression session rotation (#79017). A caller-supplied key + is authoritative but is bounded to OpenAI's 64-char wire cap in place. """ - # An explicit caller body field is authoritative — do not add a duplicate - # top-level field whose SDK merge precedence could overwrite it. But it - # must still respect the wire constraint: OpenAI caps ``prompt_cache_key`` - # at 64 chars (DeepSeek and Zai inherit the same limit via their - # OpenAI-compatible APIs) and rejects longer values with HTTP 400. Bound - # caller keys in place with the same hash shape the Responses transport - # uses (``_bounded_prompt_cache_key`` in agent/transports/codex.py), so - # both transports behave identically for over-length keys. - from agent.transports.codex import _bounded_prompt_cache_key + # Share the Responses transport's hash + scope normalization so equivalent + # prefixes hit one bucket across modes without merging unrelated sessions (#78941). + from agent.transports.codex import ( + _bound_prompt_cache_key_field, + _cache_scope_from_session_id, + _content_cache_key, + ) extra_body = api_kwargs.get("extra_body") - caller_supplied = "prompt_cache_key" in api_kwargs or ( - isinstance(extra_body, dict) and "prompt_cache_key" in extra_body - ) - if caller_supplied: - if "prompt_cache_key" in api_kwargs: - bounded = _bounded_prompt_cache_key(api_kwargs["prompt_cache_key"]) - if bounded: - api_kwargs["prompt_cache_key"] = bounded - else: - api_kwargs.pop("prompt_cache_key", None) - if isinstance(extra_body, dict) and "prompt_cache_key" in extra_body: - bounded = _bounded_prompt_cache_key(extra_body["prompt_cache_key"]) - if bounded: - extra_body["prompt_cache_key"] = bounded - else: - extra_body.pop("prompt_cache_key", None) + containers = [c for c in (api_kwargs, extra_body) if isinstance(c, dict) and "prompt_cache_key" in c] + if containers: + for c in containers: + _bound_prompt_cache_key_field(c) return - if not supports_prompt_cache_key: return - - # Reuse the Responses transport's single authoritative hash algorithm and - # session-scope normalization so equivalent static prefixes route to the - # same cache bucket across modes, without concentrating unrelated - # sessions into one shared bucket (see #78941). - from agent.transports.codex import _cache_scope_from_session_id, _content_cache_key - cache_key = _content_cache_key( _static_prompt_instructions(messages), tools, @@ -169,16 +113,7 @@ def _add_prompt_cache_key( def _reasoning_config_for_model(model: str, reasoning_config: dict | None) -> dict | None: - """Return the model's wire-compatible reasoning config. - - Hermes' internal effort set extends the wire vocabulary with ``ultra`` - (the /reasoning command documents none..xhigh|max|ultra). OpenAI- - compatible wires — OpenRouter chief among them — accept exactly - max|xhigh|high|medium|low|minimal|none and reject the extension with - HTTP 400 (#89503). Clamp against the declared wire vocabulary via the - shared policy in ``agent.reasoning_effort``; provider profiles with - narrower sets clamp again downstream. - """ + """Clamp Hermes' extended effort set (``ultra``) to the OpenAI-compat wire vocabulary (#89503).""" if not isinstance(reasoning_config, dict): return reasoning_config effort = str(reasoning_config.get("effort") or "").strip().lower() @@ -186,9 +121,7 @@ def _reasoning_config_for_model(model: str, reasoning_config: dict | None) -> di return reasoning_config clamped = clamp_effort(effort, OPENAI_COMPAT_WIRE_EFFORTS) if clamped != effort: - normalized = dict(reasoning_config) - normalized["effort"] = clamped - return normalized + return {**reasoning_config, "effort": clamped} return reasoning_config @@ -196,55 +129,32 @@ def _build_gemini_thinking_config(model: str, reasoning_config: dict | None) -> """Translate Hermes/OpenRouter-style reasoning config to Gemini thinkingConfig.""" if reasoning_config is None or not isinstance(reasoning_config, dict): return None - normalized_model = (model or "").strip().lower() if normalized_model.startswith("google/"): normalized_model = normalized_model.split("/", 1)[1] - - # ``thinking_config`` is a Gemini-only request parameter. The same - # ``gemini`` provider also serves Gemma (and historically PaLM/Bard); - # those reject the field with HTTP 400 "Unknown name 'thinking_config': - # Cannot find field" — including the polite ``{"includeThoughts": False}`` - # form. Omit the field entirely on non-Gemini models. (#17426) + # ``thinking_config`` is Gemini-only; Gemma/PaLM on the same provider reject + # the field with HTTP 400 even as ``{"includeThoughts": False}`` (#17426). if not normalized_model.startswith("gemini"): return None - if reasoning_config.get("enabled") is False: - # Gemini can hide thought parts even when internal thinking still - # happens; omit thinkingLevel to avoid model-specific validation quirks. return {"includeThoughts": False} - effort = str(reasoning_config.get("effort", "medium") or "medium").strip().lower() if effort == "none": return {"includeThoughts": False} - thinking_config: Dict[str, Any] = {"includeThoughts": True} - - # Gemini 2.5 accepts thinkingBudget; don't guess a budget from Hermes' - # coarse effort levels. ``includeThoughts`` alone is enough to surface - # thought parts without risking request validation errors. + # Gemini 2.5 takes thinkingBudget; don't guess one from coarse effort levels. if normalized_model.startswith("gemini-2.5-"): return thinking_config - if effort not in {"minimal", "low", "medium", "high", "xhigh", "max", "ultra"}: effort = "medium" - - # Gemini 3 Flash documents low/medium/high thinking levels; Gemini 3 Pro - # is stricter (low/high). Clamp Hermes' wider effort set to what each - # family accepts so we never forward an undocumented level verbatim. + # Gemini 3 Flash documents low/medium/high; Gemini 3 Pro only low/high. if normalized_model.startswith(("gemini-3", "gemini-3.1")): if "flash" in normalized_model: - if effort in {"minimal", "low"}: - thinking_config["thinkingLevel"] = "low" - elif effort in {"high", "xhigh", "max", "ultra"}: - thinking_config["thinkingLevel"] = "high" - else: - thinking_config["thinkingLevel"] = "medium" - elif "pro" in normalized_model: thinking_config["thinkingLevel"] = ( - "high" if effort in {"high", "xhigh", "max", "ultra"} else "low" + "low" if effort in {"minimal", "low"} else "high" if effort in _HIGH_EFFORTS else "medium" ) - + elif "pro" in normalized_model: + thinking_config["thinkingLevel"] = "high" if effort in _HIGH_EFFORTS else "low" return thinking_config @@ -252,29 +162,19 @@ def _snake_case_gemini_thinking_config(config: dict | None) -> dict | None: """Convert Gemini thinking config keys to the OpenAI-compat field names.""" if not isinstance(config, dict) or not config: return None - translated: Dict[str, Any] = {} - if isinstance(config.get("includeThoughts"), bool): - translated["include_thoughts"] = config["includeThoughts"] - if isinstance(config.get("thinkingLevel"), str) and config["thinkingLevel"].strip(): - translated["thinking_level"] = config["thinkingLevel"].strip().lower() - if isinstance(config.get("thinkingBudget"), (int, float)): - translated["thinking_budget"] = int(config["thinkingBudget"]) + include, level, budget = config.get("includeThoughts"), config.get("thinkingLevel"), config.get("thinkingBudget") + if isinstance(include, bool): + translated["include_thoughts"] = include + if isinstance(level, str) and level.strip(): + translated["thinking_level"] = level.strip().lower() + if isinstance(budget, (int, float)): + translated["thinking_budget"] = int(budget) return translated or None -def _raise_gemini_thinking_max_tokens( - model: str, - reasoning_config: dict | None, - requested: Any, -) -> Any: - """Raise Gemini output caps that thinking tokens would otherwise consume. - - Gemini bills thought tokens against maxOutputTokens / max_tokens. A - global Hermes cap of 4096 is enough for visible text, but Ultra/high - thinking can exhaust it on the first request and abort after four - length-continuations. - """ +def _raise_gemini_thinking_max_tokens(model: str, reasoning_config: dict | None, requested: Any) -> Any: + """Raise Gemini output caps that thinking tokens (billed against max_tokens) would otherwise exhaust.""" thinking_config = _build_gemini_thinking_config(model, reasoning_config) if not thinking_config: return requested @@ -285,260 +185,164 @@ def _raise_gemini_thinking_max_tokens( def _is_gemini_openai_compat_base_url(base_url: Any) -> bool: normalized = str(base_url or "").strip().rstrip("/").lower() - if not normalized: - return False - if "generativelanguage.googleapis.com" not in normalized: - return False - return normalized.endswith("/openai") + return bool(normalized) and "generativelanguage.googleapis.com" in normalized and normalized.endswith("/openai") def _is_openai_api_base_url(base_url: Any) -> bool: - """True only for api.openai.com itself (exact host). + """True only for the exact api.openai.com host (implies ``prompt_cache_key`` support). - OpenAI documents ``prompt_cache_key`` as a first-class body field and - GPT-5.6+ docs recommend it for reliable cache routing, so the flag is - implied for the real endpoint. Deliberately NOT a substring match: - Azure OpenAI and strict OpenAI-compat endpoints may reject unknown - fields and must stay opt-in via ``supports_prompt_cache_key``. + Not a substring match: Azure / strict OpenAI-compat endpoints may reject the + field and must stay opt-in via ``supports_prompt_cache_key``. """ try: from urllib.parse import urlparse - host = (urlparse(str(base_url or "").strip()).hostname or "").lower() + return (urlparse(str(base_url or "").strip()).hostname or "").lower() == "api.openai.com" except Exception: return False - return host == "api.openai.com" def _model_consumes_thought_signature(model: Any) -> bool: - """True when the outgoing model is a Gemini family model that requires - ``extra_content`` (thought_signature) to be replayed on tool calls. + """True for Gemini-family targets, which require tool-call ``extra_content`` (thought_signature) replay. - Gemini 3 thinking models attach ``extra_content`` to each tool call and - reject subsequent requests with HTTP 400 if it is missing. Every other - strict OpenAI-compatible provider (Fireworks, Mistral, ...) rejects the - request with 400 if ``extra_content`` *is* present. So the field must be - kept only when the target model is itself Gemini-family, and stripped - otherwise — including when a non-Gemini model inherits stale Gemini - ``extra_content`` from earlier in a mixed-provider session. + Every other strict provider rejects a request containing it, so it is kept + only for Gemini targets and stripped otherwise (incl. stale inherited copies). """ m = str(model or "").lower() return "gemini" in m or "gemma" in m +def _thinking_disabled(reasoning_config: Any) -> bool: + return bool(reasoning_config and isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False) + + +def _swap_developer_role(sanitized: list, model_lower: str) -> list: + """GPT-5/Codex models take a ``developer`` role instead of ``system``.""" + if ( + sanitized + and isinstance(sanitized[0], dict) + and sanitized[0].get("role") == "system" + and any(p in model_lower for p in DEVELOPER_ROLE_MODELS) + ): + sanitized = list(sanitized) + sanitized[0] = {**sanitized[0], "role": "developer"} + return sanitized + + +def _apply_max_tokens(api_kwargs: dict, model: str, reasoning_config: Any, params: dict, profile_max: Any = None) -> None: + """Resolve max_tokens — priority: ephemeral > user > profile default > anthropic_max_output.""" + max_tokens_fn = params.get("max_tokens_param_fn") + for candidate in (params.get("ephemeral_max_output_tokens"), params.get("max_tokens")): + if candidate is not None and max_tokens_fn: + api_kwargs.update(max_tokens_fn(_raise_gemini_thinking_max_tokens(model, reasoning_config, candidate))) + return + if profile_max and max_tokens_fn: + api_kwargs.update(max_tokens_fn(_raise_gemini_thinking_max_tokens(model, reasoning_config, profile_max))) + elif params.get("anthropic_max_output") is not None: + api_kwargs["max_tokens"] = params["anthropic_max_output"] + + +def _base_kwargs(model: str, sanitized: list, tools: Any, params: dict) -> dict[str, Any]: + """Shared ``{model, messages[, timeout][, tools]}`` scaffold for both build paths.""" + api_kwargs: dict[str, Any] = {"model": model, "messages": sanitized} + timeout = params.get("timeout") + if timeout is not None: + api_kwargs["timeout"] = timeout + if tools: + # Moonshot/Kimi uses a stricter JSON Schema flavor; rewriting here also covers aggregator routes. + api_kwargs["tools"] = sanitize_moonshot_tools(tools) if is_moonshot_model(model) else tools + return api_kwargs + + +def _finish_kwargs( + api_kwargs: dict[str, Any], sanitized: list, params: dict, *, supports_prompt_cache_key: bool +) -> dict[str, Any]: + """Tail shared by both build paths: content-addressed prompt_cache_key, then return.""" + _add_prompt_cache_key( + api_kwargs, + messages=sanitized, + tools=api_kwargs.get("tools"), + supports_prompt_cache_key=supports_prompt_cache_key, + session_id=params.get("session_id"), + cache_scope_id=params.get("cache_scope_id"), + ) + return api_kwargs + + +def _msg_strip_keys(msg: dict) -> list: + """Keys to drop from a message: persistence sidecars plus any ``_``-prefixed Hermes scaffolding marker.""" + return [k for k in msg if k in _STRIP_MSG_KEYS or (isinstance(k, str) and k.startswith("_"))] + + +def _tc_strip_keys(tc: dict, strip_extra_content: bool) -> list: + keys = [k for k in _STRIP_TC_KEYS if k in tc] + if strip_extra_content and "extra_content" in tc: + keys.append("extra_content") + return keys + + +def _invalid_assistant_tool_calls(msg: dict, tool_calls: Any) -> bool: + """``tool_calls: []`` / ``tool_calls: null`` on an assistant message — strict providers reject both (#58755).""" + return ( + msg.get("role") == "assistant" + and "tool_calls" in msg + and (tool_calls is None or (isinstance(tool_calls, list) and not tool_calls)) + ) + + class ChatCompletionsTransport(ProviderTransport): - """Transport for api_mode='chat_completions'. + """Transport for api_mode='chat_completions'.""" - The default path for OpenAI-compatible providers. - """ - - # Wire-alias provenance of the most recent request built for this - # transport: ``{alias_sent_on_wire: original_tool_name}``. ``None`` - # means no request recorded provenance (normalize-only call sites) — - # fall back to the static alias constant. An empty dict means the last - # request emitted no aliases, so no reverse rewrite may run (#95003). + # Wire-alias provenance of the most recent request: ``{alias: original}``. + # ``None`` = no request recorded (normalize-only call sites) -> fall back to + # the static alias; ``{}`` = last request emitted no aliases (#95003). _last_wire_aliases: dict[str, str] | None = None @property def api_mode(self) -> str: return "chat_completions" - def convert_messages( - self, messages: list[dict[str, Any]], **kwargs - ) -> list[dict[str, Any]]: - """Messages are already in OpenAI format — strip internal fields - that strict chat-completions providers reject with HTTP 400/422 - (or, in the case of some OpenAI-compatible gateways, 5xx): + def convert_messages(self, messages: list[dict[str, Any]], **kwargs) -> list[dict[str, Any]]: + """Strip internal fields that strict chat-completions providers reject (HTTP 400/422). - - Codex Responses API fields: ``codex_reasoning_items`` / - ``codex_message_items`` on the message, ``call_id`` / - ``response_item_id`` on ``tool_calls`` entries. - - ``extra_content`` on ``tool_calls`` (Gemini thought_signature) — - stripped unless the outgoing ``model`` is itself Gemini-family. - Gemini 3 thinking models attach it for replay, but strict providers - (Fireworks, Mistral) reject any payload containing it with - ``Extra inputs are not permitted, field: 'messages[N].tool_calls[M].extra_content'``. - It must be kept for Gemini targets (replay required) and dropped for - everyone else, including non-Gemini models that inherited stale - Gemini ``extra_content`` earlier in a mixed-provider session. - - ``tool_name`` on tool-result messages — written by - ``make_tool_result_message()`` for the SQLite FTS index, but not - part of the Chat Completions schema. Strict providers (Fireworks, - Moonshot/Kimi) reject any payload containing it with - ``Extra inputs are not permitted, field: 'messages[N].tool_name'``. - Permissive providers (OpenRouter, MiniMax) silently ignore the - field, which masked the bug for months. - - Hermes-internal scaffolding markers — any top-level message key - starting with ``_`` (e.g. ``_empty_recovery_synthetic``, - ``_empty_terminal_sentinel``, ``_thinking_prefill``). These are - bookkeeping flags the agent loop attaches to messages so the - persistence layer can later strip its own scaffolding; they must - never reach the wire. Permissive providers (real OpenAI, - Anthropic) silently drop unknown message keys, but strict - gateways (e.g. opencode-go, codex.nekos.me) reject with - ``Extra inputs are not permitted, field: 'messages[N]._empty_recovery_synthetic'``, - which then poisons every subsequent request in the session. - - Provider-specific ordered replay sidecars -- - ``anthropic_content_blocks`` and ``bedrock_content_blocks`` are - durable-history data for their native transports, not part of the - Chat Completions schema. They must not cross a provider boundary. + Codex sidecars, ``tool_name``, ``_``-prefixed markers, native-transport + block sidecars, and tool-call ``call_id``/``response_item_id`` are always + dropped; ``extra_content`` is dropped unless ``model`` is Gemini-family. + Returns the input list unchanged when nothing needs sanitizing. """ - strip_extra_content = not _model_consumes_thought_signature( - kwargs.get("model") - ) - needs_sanitize = False - for msg in messages: + strip_extra_content = not _model_consumes_thought_signature(kwargs.get("model")) + + def sanitize(msg: Any) -> "dict | None": + """Sanitized copy of ``msg``, or None when nothing needs stripping.""" if not isinstance(msg, dict): - continue - if ( - "codex_reasoning_items" in msg - or "codex_message_items" in msg - or "tool_name" in msg - or "effect_disposition" in msg - or "timestamp" in msg # #47868 — strict providers reject this - or "platform_message_id" in msg # gateway dedup id (persistence-only) - or "api_content" in msg # persist-what-you-send sidecar - or "anthropic_content_blocks" in msg - or "bedrock_content_blocks" in msg - ): - needs_sanitize = True - break - if any(isinstance(k, str) and k.startswith("_") for k in msg): - needs_sanitize = True - break + return None + strip_keys = _msg_strip_keys(msg) + out_msg = dict(msg) + for key in strip_keys: + out_msg.pop(key, None) tool_calls = msg.get("tool_calls") - if isinstance(tool_calls, list): - # Defense-in-depth: a strict OpenAI-compatible provider - # (e.g. onerouter / Qwen, DeepSeek v4) rejects an assistant - # message carrying ``tool_calls: []`` (empty array) with - # HTTP 400 "Empty tool_calls is not supported in message." - # The pre-API sanitizer in agent_runtime_helpers drops these, - # but only on the conversation_loop path — other routes can - # reach the wire without it. For every request that - # serializes through this transport (conversation loop and - # any caller using it), this is the last boundary, so - # normalize here. Requests built by fully separate payload - # paths (e.g. some auxiliary clients) never pass through - # this layer and are out of scope for it. (#58755 follow-up) - if ( - msg.get("role") == "assistant" - and "tool_calls" in msg - and not tool_calls - ): - needs_sanitize = True - break - for tc in tool_calls: - if isinstance(tc, dict) and ( - "call_id" in tc - or "response_item_id" in tc - or (strip_extra_content and "extra_content" in tc) - ): - needs_sanitize = True - break - if needs_sanitize: - break - elif ( - isinstance(tool_calls, type(None)) - and msg.get("role") == "assistant" - and "tool_calls" in msg - ): - # Explicit ``tool_calls: null`` is equally invalid on strict - # providers — treat it like the empty-array case. - needs_sanitize = True - break - - if not needs_sanitize: - return messages - - sanitized = list(messages) - for msg_idx, msg in enumerate(messages): - if not isinstance(msg, dict): - continue - - copied_msg: dict[str, Any] | None = None - - def mutable_msg() -> dict[str, Any]: - nonlocal copied_msg - if copied_msg is None: - copied_msg = dict(msg) - sanitized[msg_idx] = copied_msg - return copied_msg - - if ( - "codex_reasoning_items" in msg - or "codex_message_items" in msg - or "tool_name" in msg - or "effect_disposition" in msg - or "timestamp" in msg # #47868 — leak into strict providers - or "platform_message_id" in msg # gateway dedup id (persistence-only) - or "api_content" in msg # persist-what-you-send sidecar - or "anthropic_content_blocks" in msg - or "bedrock_content_blocks" in msg - ): - out_msg = mutable_msg() - out_msg.pop("codex_reasoning_items", None) - out_msg.pop("codex_message_items", None) - out_msg.pop("tool_name", None) - out_msg.pop("effect_disposition", None) - out_msg.pop("timestamp", None) # #47868 — leak into strict providers - out_msg.pop("platform_message_id", None) # gateway dedup id - out_msg.pop("api_content", None) # persist-what-you-send sidecar - out_msg.pop("anthropic_content_blocks", None) - out_msg.pop("bedrock_content_blocks", None) - - - # Drop all Hermes-internal scaffolding markers (``_``-prefixed). - # OpenAI's message schema has no ``_``-prefixed fields, so this - # is safe and future-proofs against new markers being added. - internal_keys = [k for k in msg if isinstance(k, str) and k.startswith("_")] - if internal_keys: - out_msg = mutable_msg() - for key in internal_keys: - out_msg.pop(key, None) - - tool_calls = msg.get("tool_calls") - if isinstance(tool_calls, list): - # Strip empty/invalid tool_calls arrays at the transport - # layer (see detection above). Strict OpenAI-compatible - # providers reject ``tool_calls: []`` with HTTP 400; dropping - # the key keeps the message schema-valid. Matches the - # pre-API sanitizer's behaviour so all routes agree. - if ( - msg.get("role") == "assistant" - and "tool_calls" in msg - and not tool_calls - ): - out_msg = mutable_msg() - out_msg.pop("tool_calls", None) - continue - copied_tool_calls: list[Any] | None = None + copied_tool_calls = None + if _invalid_assistant_tool_calls(msg, tool_calls): + out_msg.pop("tool_calls", None) + strip_keys.append("tool_calls") + elif isinstance(tool_calls, list): for tc_idx, tc in enumerate(tool_calls): - if isinstance(tc, dict): - should_copy_tc = ( - "call_id" in tc - or "response_item_id" in tc - or (strip_extra_content and "extra_content" in tc) - ) - if should_copy_tc: - if copied_tool_calls is None: - copied_tool_calls = list(tool_calls) - copied_tc = dict(tc) - copied_tc.pop("call_id", None) - copied_tc.pop("response_item_id", None) - if strip_extra_content: - copied_tc.pop("extra_content", None) - copied_tool_calls[tc_idx] = copied_tc + keys = _tc_strip_keys(tc, strip_extra_content) if isinstance(tc, dict) else [] + if keys: + if copied_tool_calls is None: + copied_tool_calls = list(tool_calls) + copied_tc = dict(tc) + for key in keys: + copied_tc.pop(key, None) + copied_tool_calls[tc_idx] = copied_tc if copied_tool_calls is not None: - mutable_msg()["tool_calls"] = copied_tool_calls - elif ( - isinstance(tool_calls, type(None)) - and msg.get("role") == "assistant" - and "tool_calls" in msg - ): - # Explicit ``tool_calls: null`` is invalid on strict - # providers — drop the key entirely. - mutable_msg().pop("tool_calls", None) - return sanitized + out_msg["tool_calls"] = copied_tool_calls + return out_msg if strip_keys or copied_tool_calls is not None else None + + sanitized_pairs = [(m, sanitize(m)) for m in messages] + if all(s is None for _, s in sanitized_pairs): + return messages + return [m if s is None else s for m, s in sanitized_pairs] def convert_tools(self, tools: list[dict[str, Any]]) -> list[dict[str, Any]]: """Tools are already in OpenAI format — identity.""" @@ -553,232 +357,74 @@ class ChatCompletionsTransport(ProviderTransport): ) -> dict[str, Any]: """Build chat.completions.create() kwargs. - params (all optional): - timeout: float — API call timeout - max_tokens: int | None — user-configured max tokens - ephemeral_max_output_tokens: int | None — one-shot override - max_tokens_param_fn: callable — returns {max_tokens: N} or {max_completion_tokens: N} - reasoning_config: dict | None - request_overrides: dict | None - session_id: str | None - model_lower: str — lowercase model name for pattern matching - # Provider profile path (all per-provider quirks live in providers/) - provider_profile: ProviderProfile | None — when present, delegates to - _build_kwargs_from_profile(); all flag params below are bypassed. - # Legacy-path flags — only used when provider_profile is None - # (i.e. custom / unregistered providers). Known providers all go - # through provider_profile. - is_openrouter: bool - is_nous: bool - is_qwen_portal: bool - is_github_models: bool - is_nvidia_nim: bool - is_kimi: bool - is_tokenhub: bool - is_lmstudio: bool - is_custom_provider: bool - ollama_num_ctx: int | None - # Provider routing - provider_preferences: dict | None - # Qwen-specific - qwen_prepare_fn: callable | None — runs AFTER codex sanitization - qwen_prepare_inplace_fn: callable | None — in-place variant for deepcopied lists - qwen_session_metadata: dict | None - # Temperature - fixed_temperature: Any — from _fixed_temperature_for_model() - omit_temperature: bool - # Reasoning - supports_reasoning: bool - github_reasoning_extra: dict | None - lmstudio_reasoning_options: list[str] | None # raw allowed_options from /api/v1/models - # Claude on OpenRouter/Nous max output - anthropic_max_output: int | None - extra_body_additions: dict | None - supports_prompt_cache_key: bool — explicit endpoint capability for - the top-level Chat Completions request field; defaults off. + With ``provider_profile`` every quirk comes from the profile + (_build_kwargs_from_profile). The legacy flag path below (is_kimi, + is_openrouter, is_lmstudio, ...) is only reached for unregistered providers. """ - # Codex sanitization: drop reasoning_items / call_id / response_item_id. - # Pass model so the Gemini thought_signature (extra_content) is kept for - # Gemini targets and stripped for strict non-Gemini providers. sanitized = self.convert_messages(messages, model=model) - - # ── Provider profile: single-path when present ────────────────── _profile = params.get("provider_profile") if _profile: - return self._build_kwargs_from_profile( - _profile, model, sanitized, tools, params - ) + return self._build_kwargs_from_profile(_profile, model, sanitized, tools, params) - # ── Legacy fallback (unregistered / unknown provider) ─────────── - # Reached only when get_provider_profile() returned None. - # Known providers always go through the profile path above. + sanitized = _swap_developer_role(sanitized, params.get("model_lower", (model or "").lower())) + api_kwargs = _base_kwargs(model, sanitized, tools, params) - # Developer role swap for GPT-5/Codex models - model_lower = params.get("model_lower", (model or "").lower()) - if ( - sanitized - and isinstance(sanitized[0], dict) - and sanitized[0].get("role") == "system" - and any(p in model_lower for p in DEVELOPER_ROLE_MODELS) - ): - sanitized = list(sanitized) - sanitized[0] = {**sanitized[0], "role": "developer"} - - api_kwargs: dict[str, Any] = { - "model": model, - "messages": sanitized, - } - - timeout = params.get("timeout") - if timeout is not None: - api_kwargs["timeout"] = timeout - - # Tools - if tools: - # Moonshot/Kimi uses a stricter flavored JSON Schema. Rewriting - # tool parameters here keeps aggregator routes (Nous, OpenRouter, - # etc.) compatible, in addition to direct moonshot.ai endpoints. - if is_moonshot_model(model): - tools = sanitize_moonshot_tools(tools) - api_kwargs["tools"] = tools - - # max_tokens resolution — priority: ephemeral > user > provider default - max_tokens_fn = params.get("max_tokens_param_fn") - ephemeral = params.get("ephemeral_max_output_tokens") - max_tokens = params.get("max_tokens") - anthropic_max_out = params.get("anthropic_max_output") is_kimi = params.get("is_kimi", False) - is_tokenhub = params.get("is_tokenhub", False) reasoning_config = _reasoning_config_for_model(model, params.get("reasoning_config")) + _apply_max_tokens(api_kwargs, model, reasoning_config, params) - if ephemeral is not None and max_tokens_fn: - api_kwargs.update( - max_tokens_fn( - _raise_gemini_thinking_max_tokens(model, reasoning_config, ephemeral) - ) + # Kimi / TokenHub / LM Studio: top-level reasoning_effort (unless thinking disabled). + thinking_off = _thinking_disabled(reasoning_config) + _e = requested_effort(reasoning_config) + if is_kimi and not thinking_off: + # K3 = low/high/max (server default high), K2-era = low/medium/high (default medium). + _supported = kimi_supported_efforts(model) + is_k3 = _supported is KIMI_K3_EFFORTS + api_kwargs["reasoning_effort"] = ( + ("high" if is_k3 else "medium") if _e is None + else clamp_effort(_e, _supported, KIMI_K3_OVERRIDES if is_k3 else None) ) - elif max_tokens is not None and max_tokens_fn: - api_kwargs.update( - max_tokens_fn( - _raise_gemini_thinking_max_tokens(model, reasoning_config, max_tokens) - ) - ) - elif anthropic_max_out is not None: - api_kwargs["max_tokens"] = anthropic_max_out - - # Kimi: top-level reasoning_effort (unless thinking disabled) - if is_kimi: - _kimi_thinking_off = bool( - reasoning_config - and isinstance(reasoning_config, dict) - and reasoning_config.get("enabled") is False - ) - if not _kimi_thinking_off: - # Kimi vocabularies are declared in agent.reasoning_effort: - # K3 = low/high/max (with the vendor-documented medium→high, - # xhigh→max rounding), K2-era = low/medium/high. Default when - # no effort was requested: K3's server default is high, - # K2-era's is medium. - _supported = kimi_supported_efforts(model) - _overrides = ( - KIMI_K3_OVERRIDES if _supported is KIMI_K3_EFFORTS else None - ) - _e = requested_effort(reasoning_config) - if _e is None: - _kimi_effort = ( - "high" if _supported is KIMI_K3_EFFORTS else "medium" - ) - else: - _kimi_effort = clamp_effort(_e, _supported, _overrides) - api_kwargs["reasoning_effort"] = _kimi_effort - - # Tencent TokenHub: top-level reasoning_effort (unless thinking disabled) - if is_tokenhub: - _tokenhub_thinking_off = bool( - reasoning_config - and isinstance(reasoning_config, dict) - and reasoning_config.get("enabled") is False - ) - if not _tokenhub_thinking_off: - # TokenHub accepts low/medium/high (declared in - # agent.reasoning_effort); default high when no effort was - # requested. - _e = requested_effort(reasoning_config) - _tokenhub_effort = ( - "high" if _e is None else clamp_effort(_e, TOKENHUB_EFFORTS) - ) - api_kwargs["reasoning_effort"] = _tokenhub_effort - - # LM Studio: top-level reasoning_effort. Only emit when the model - # declares reasoning support via /api/v1/models capabilities (gated - # upstream by params["supports_reasoning"]). resolve_lmstudio_effort - # is shared with run_agent's summary path so both stay in sync. + if params.get("is_tokenhub", False) and not thinking_off: + api_kwargs["reasoning_effort"] = "high" if _e is None else clamp_effort(_e, TOKENHUB_EFFORTS) if params.get("is_lmstudio", False) and params.get("supports_reasoning", False): - _lm_effort = resolve_lmstudio_effort( - reasoning_config, - params.get("lmstudio_reasoning_options"), - ) + _lm_effort = resolve_lmstudio_effort(reasoning_config, params.get("lmstudio_reasoning_options")) if _lm_effort is not None: api_kwargs["reasoning_effort"] = _lm_effort - # extra_body assembly extra_body: dict[str, Any] = {} - is_openrouter = params.get("is_openrouter", False) - is_github_models = params.get("is_github_models", False) provider_name = str(params.get("provider_name") or "").strip().lower() base_url = params.get("base_url") - provider_prefs = params.get("provider_preferences") if provider_prefs and is_openrouter: extra_body["provider"] = provider_prefs - - # Pareto Code router plugin — model-gated. Same shape as the - # profile path in plugins/model-providers/openrouter/__init__.py; - # this branch only runs when the OpenRouter profile isn't loaded. + # Pareto Code router plugin (same shape as the OpenRouter profile path). if is_openrouter and model == "openrouter/pareto-code": _pareto_score = params.get("openrouter_min_coding_score") - if _pareto_score is not None and _pareto_score != "": - try: - _pareto_score_f = float(_pareto_score) - except (TypeError, ValueError): - _pareto_score_f = None - if _pareto_score_f is not None and 0.0 <= _pareto_score_f <= 1.0: - extra_body["plugins"] = [ - {"id": "pareto-router", "min_coding_score": _pareto_score_f} - ] - - # Kimi extra_body.thinking + try: + _pareto_score_f = float(_pareto_score) if _pareto_score not in (None, "") else None + except (TypeError, ValueError): + _pareto_score_f = None + if _pareto_score_f is not None and 0.0 <= _pareto_score_f <= 1.0: + extra_body["plugins"] = [{"id": "pareto-router", "min_coding_score": _pareto_score_f}] if is_kimi: - _kimi_thinking_enabled = True - if reasoning_config and isinstance(reasoning_config, dict): - if reasoning_config.get("enabled") is False: - _kimi_thinking_enabled = False - extra_body["thinking"] = { - "type": "enabled" if _kimi_thinking_enabled else "disabled", - } + extra_body["thinking"] = {"type": "disabled" if thinking_off else "enabled"} - # Reasoning. LM Studio is handled above via top-level reasoning_effort, - # so skip emitting extra_body.reasoning for it. + # LM Studio is handled above via top-level reasoning_effort. if params.get("supports_reasoning", False) and not params.get("is_lmstudio", False): - if is_github_models: + if params.get("is_github_models", False): gh_reasoning = params.get("github_reasoning_extra") if gh_reasoning is not None: extra_body["reasoning"] = gh_reasoning else: _effort = "medium" - _enabled = True if reasoning_config and isinstance(reasoning_config, dict): _effort = reasoning_config.get("effort", "medium") or "medium" - # Honor an explicit "thinking off" (agent.reasoning_effort: - # none / the one-shot length-continuation override) the same - # way the provider-profile path does — never re-enable it. - if reasoning_config.get("enabled") is False or _effort == "none": - _enabled = False - if _enabled: - extra_body["reasoning"] = {"enabled": True, "effort": _effort} - else: + # Honor explicit "thinking off" like the profile path — never re-enable it. + if thinking_off or _effort == "none": extra_body["reasoning"] = {"enabled": False, "effort": "none"} + else: + extra_body["reasoning"] = {"enabled": True, "effort": _effort} if provider_name == "gemini": raw_thinking_config = _build_gemini_thinking_config(model, reasoning_config) @@ -793,130 +439,52 @@ class ChatCompletionsTransport(ProviderTransport): elif raw_thinking_config: extra_body["thinking_config"] = raw_thinking_config - # Merge any pre-built extra_body additions additions = params.get("extra_body_additions") if additions: extra_body.update(additions) - if extra_body: api_kwargs["extra_body"] = extra_body - - # Request overrides last (service_tier etc.) overrides = params.get("request_overrides") if overrides: api_kwargs.update(overrides) - - _add_prompt_cache_key( + return _finish_kwargs( api_kwargs, - messages=sanitized, - tools=api_kwargs.get("tools"), + sanitized, + params, supports_prompt_cache_key=bool(params.get("supports_prompt_cache_key")) or _is_openai_api_base_url(params.get("base_url")), - session_id=params.get("session_id"), - cache_scope_id=params.get("cache_scope_id"), ) - return api_kwargs - def _build_kwargs_from_profile(self, profile, model, sanitized, tools, params): - """Build API kwargs using a ProviderProfile — single path, no legacy flags. - - This method replaces the entire flag-based kwargs assembly when a - provider_profile is passed. Every quirk comes from the profile object. - """ + """Build API kwargs from a ProviderProfile — every quirk comes from the profile object.""" from providers.base import OMIT_TEMPERATURE - # Message preprocessing - sanitized = profile.prepare_messages(sanitized) - - # Developer role swap — model-name-based, applies to all providers - _model_lower = (model or "").lower() - if ( - sanitized - and isinstance(sanitized[0], dict) - and sanitized[0].get("role") == "system" - and any(p in _model_lower for p in DEVELOPER_ROLE_MODELS) - ): - sanitized = list(sanitized) - sanitized[0] = {**sanitized[0], "role": "developer"} - - api_kwargs: dict[str, Any] = { - "model": model, - "messages": sanitized, - } - - # Temperature + sanitized = _swap_developer_role(profile.prepare_messages(sanitized), (model or "").lower()) + api_kwargs: dict[str, Any] = {"model": model, "messages": sanitized} if profile.fixed_temperature is OMIT_TEMPERATURE: - pass # Don't include temperature at all + pass elif profile.fixed_temperature is not None: api_kwargs["temperature"] = profile.fixed_temperature - else: - # Use caller's temperature if provided - temp = params.get("temperature") - if temp is not None: - api_kwargs["temperature"] = temp + elif params.get("temperature") is not None: + api_kwargs["temperature"] = params["temperature"] + api_kwargs.update(_base_kwargs(model, sanitized, tools, params)) - # Timeout - timeout = params.get("timeout") - if timeout is not None: - api_kwargs["timeout"] = timeout - - # Tools — apply Moonshot/Kimi schema sanitization regardless of path - if tools: - if is_moonshot_model(model): - tools = sanitize_moonshot_tools(tools) - api_kwargs["tools"] = tools - - # max_tokens resolution — priority: ephemeral > user > profile default - max_tokens_fn = params.get("max_tokens_param_fn") - ephemeral = params.get("ephemeral_max_output_tokens") - user_max = params.get("max_tokens") - anthropic_max = params.get("anthropic_max_output") - # Per-model default cap — profiles override get_max_tokens() when - # they front several backends with different completion-token limits - # (e.g. opencode-go: mimo-v2.5-pro = 131072). - profile_max = profile.get_max_tokens(model) reasoning_config = _reasoning_config_for_model(model, params.get("reasoning_config")) + # Profiles fronting several backends override get_max_tokens() per model. + _apply_max_tokens(api_kwargs, model, reasoning_config, params, profile_max=profile.get_max_tokens(model)) - if ephemeral is not None and max_tokens_fn: - api_kwargs.update( - max_tokens_fn( - _raise_gemini_thinking_max_tokens(model, reasoning_config, ephemeral) - ) - ) - elif user_max is not None and max_tokens_fn: - api_kwargs.update( - max_tokens_fn( - _raise_gemini_thinking_max_tokens(model, reasoning_config, user_max) - ) - ) - elif profile_max and max_tokens_fn: - api_kwargs.update( - max_tokens_fn( - _raise_gemini_thinking_max_tokens(model, reasoning_config, profile_max) - ) - ) - elif anthropic_max is not None: - api_kwargs["max_tokens"] = anthropic_max - - # Provider-specific api_kwargs extras (reasoning_effort, metadata, etc.) - extra_body_from_profile, top_level_from_profile = ( - profile.build_api_kwargs_extras( - reasoning_config=reasoning_config, - supports_reasoning=params.get("supports_reasoning", False), - qwen_session_metadata=params.get("qwen_session_metadata"), - model=model, - base_url=params.get("base_url"), - ollama_num_ctx=params.get("ollama_num_ctx"), - session_id=params.get("session_id"), - ) + extra_body_from_profile, top_level_from_profile = profile.build_api_kwargs_extras( + reasoning_config=reasoning_config, + supports_reasoning=params.get("supports_reasoning", False), + qwen_session_metadata=params.get("qwen_session_metadata"), + model=model, + base_url=params.get("base_url"), + ollama_num_ctx=params.get("ollama_num_ctx"), + session_id=params.get("session_id"), ) api_kwargs.update(top_level_from_profile) - # extra_body assembly extra_body: dict[str, Any] = {} - - # Profile's extra_body (tags, provider prefs, vl_high_resolution, etc.) profile_body = profile.build_extra_body( session_id=params.get("session_id"), provider_preferences=params.get("provider_preferences"), @@ -925,19 +493,9 @@ class ChatCompletionsTransport(ProviderTransport): reasoning_config=reasoning_config, openrouter_min_coding_score=params.get("openrouter_min_coding_score"), ) - if profile_body: - extra_body.update(profile_body) - - # Profile's reasoning/thinking extra_body entries - if extra_body_from_profile: - extra_body.update(extra_body_from_profile) - - # Merge any pre-built extra_body additions from the caller - additions = params.get("extra_body_additions") - if additions: - extra_body.update(additions) - - # Request overrides (user config) + for part in (profile_body, extra_body_from_profile, params.get("extra_body_additions")): + if part: + extra_body.update(part) overrides = params.get("request_overrides") if overrides: for k, v in overrides.items(): @@ -947,134 +505,91 @@ class ChatCompletionsTransport(ProviderTransport): api_kwargs[k] = v if extra_body: - # Native Gemini (generativelanguage.googleapis.com, non-/openai) - # speaks Google's REST schema, not OpenAI's. OpenAI-style extra_body - # keys (tags, reasoning, provider, plugins, …) are unknown fields - # there and Gemini rejects the whole request with a non-retryable - # HTTP 400 ("Invalid JSON payload received. Unknown name 'tags'"). - # This happens when a profile that emits extra_body (e.g. the Nous - # profile's portal `tags`) is active but the resolved endpoint is a - # Gemini base_url — typical when only Google credentials are set and - # a fallback/aux call lands on Gemini. The native client only reads - # thinking_config from extra_body, so drop everything else here. + # Native Gemini speaks Google's REST schema: OpenAI-style extra_body + # keys (tags, reasoning, provider, ...) are unknown fields -> HTTP 400. + # The native client only reads thinking_config, so drop everything else. try: from agent.gemini_native_adapter import is_native_gemini_base_url _native_gemini = is_native_gemini_base_url(params.get("base_url")) except Exception: _native_gemini = False if _native_gemini: - extra_body = { - k: v for k, v in extra_body.items() - if k in ("thinking_config", "thinkingConfig") - } + extra_body = {k: v for k, v in extra_body.items() if k in ("thinking_config", "thinkingConfig")} if extra_body: api_kwargs["extra_body"] = extra_body - - _add_prompt_cache_key( + return _finish_kwargs( api_kwargs, - messages=sanitized, - tools=api_kwargs.get("tools"), + sanitized, + params, supports_prompt_cache_key=bool(getattr(profile, "supports_prompt_cache_key", False)), - session_id=params.get("session_id"), - cache_scope_id=params.get("cache_scope_id"), ) - return api_kwargs - def normalize_response(self, response: Any, **kwargs) -> NormalizedResponse: - """Normalize OpenAI ChatCompletion to NormalizedResponse. + """Normalize an OpenAI ChatCompletion. - For chat_completions, this is near-identity — the response is already - in OpenAI format. extra_content on tool_calls (Gemini thought_signature) - is preserved via ToolCall.provider_data. reasoning_details (OpenRouter - unified format) and reasoning_content (DeepSeek/Moonshot) are also - preserved for downstream replay. + Gemini ``extra_content`` rides on ToolCall.provider_data; ``reasoning_content`` + (DeepSeek/Moonshot) and ``reasoning_details`` (OpenRouter) are kept apart in + provider_data because downstream reads them distinctly. """ choice = response.choices[0] msg = getattr(choice, "message", None) - # Poolside returns integer finish_reason (e.g. 24) instead of string _fr = getattr(choice, "finish_reason", None) - if isinstance(_fr, int): - _fr = str(_fr) - finish_reason = _fr or "stop" + finish_reason = (str(_fr) if isinstance(_fr, int) else _fr) or "stop" # Poolside returns int finish_reason tool_calls = None message_tool_calls = getattr(msg, "tool_calls", None) if message_tool_calls: tool_calls = [] + _alias_map = self._last_wire_aliases for tc in message_tool_calls: tc_function = getattr(tc, "function", None) function_name = getattr(tc_function, "name", None) - # Match Relay's codec: skip absent function/name fields, but - # preserve an explicit blank name for Hermes's recovery path. + # Match Relay's codec: skip absent function/name, keep an explicit blank name. if tc_function is None or function_name is None: continue - # Map THIS request's wire aliases back before dispatch. - # Request-local provenance: when the paired request recorded - # its alias map, only those aliases are reversed — a real - # user/plugin/MCP tool that happens to be named - # ``hermes_tool_search`` dispatches as itself when no alias - # was emitted. The static-constant fallback covers - # normalize-only call sites with no recorded request. - _alias_map = self._last_wire_aliases + # Reverse only aliases THIS request emitted; a real tool named + # ``hermes_tool_search`` dispatches as itself when none were. if _alias_map is None: if function_name == _XAI_TOOL_SEARCH_ALIAS: function_name = "tool_search" elif function_name in _alias_map: function_name = _alias_map[function_name] function_arguments = getattr(tc_function, "arguments", None) - # Preserve provider-specific extras on the tool call. - # Gemini 3 thinking models attach extra_content with - # thought_signature — without replay on the next turn the API - # rejects the request with 400. tc_provider_data: dict[str, Any] = {} extra = getattr(tc, "extra_content", None) if extra is None and hasattr(tc, "model_extra"): extra = (tc.model_extra if isinstance(tc.model_extra, dict) else {}).get("extra_content") if extra is not None: if hasattr(extra, "model_dump"): - try: - extra = extra.model_dump(warnings=False) - except TypeError: + for dump_kwargs in ({"warnings": False}, {}): try: - extra = extra.model_dump() + extra = extra.model_dump(**dump_kwargs) + break + except TypeError: + continue # older pydantic: retry without ``warnings`` except Exception: - pass - except Exception: - pass + break tc_provider_data["extra_content"] = extra tool_calls.append( ToolCall( id=getattr(tc, "id", None), name=function_name, - arguments=( - function_arguments - if function_arguments is not None - else "{}" - ), + arguments=function_arguments if function_arguments is not None else "{}", provider_data=tc_provider_data or None, ) ) usage = None if hasattr(response, "usage") and response.usage: - u = response.usage - usage = Usage( - prompt_tokens=getattr(u, "prompt_tokens", 0) or 0, - completion_tokens=getattr(u, "completion_tokens", 0) or 0, - total_tokens=getattr(u, "total_tokens", 0) or 0, - ) + usage = Usage.from_openai(response.usage) - # Preserve reasoning fields separately. DeepSeek/Moonshot use - # ``reasoning_content``; others use ``reasoning``. Downstream code - # (_extract_reasoning, thinking-prefill retry) reads both distinctly, - # so keep them apart in provider_data rather than merging. + # Fields some SDKs park in pydantic ``model_extra`` rather than as attributes. + model_extra = getattr(msg, "model_extra", None) or {} + model_extra = model_extra if isinstance(model_extra, dict) else {} reasoning = getattr(msg, "reasoning", None) reasoning_content = getattr(msg, "reasoning_content", None) - if reasoning_content is None and hasattr(msg, "model_extra"): - model_extra = getattr(msg, "model_extra", None) or {} - if isinstance(model_extra, dict) and "reasoning_content" in model_extra: - reasoning_content = model_extra["reasoning_content"] + if reasoning_content is None: + reasoning_content = model_extra.get("reasoning_content") provider_data: Dict[str, Any] = {} if reasoning_content is not None: @@ -1083,36 +598,18 @@ class ChatCompletionsTransport(ProviderTransport): if rd: provider_data["reasoning_details"] = rd - # OpenAI structured-refusal field. When a model declines, the SDK - # populates ``message.refusal`` with the explanation and leaves - # ``content`` empty. OpenAI-compatible proxies that front Anthropic / - # Bedrock (e.g. Nous Portal) surface a Claude refusal this way — or via - # ``finish_reason="content_filter"`` — instead of the native - # ``stop_reason="refusal"``. Without capturing it the refusal looks - # like an empty response, so the agent loop retries a deterministic - # refusal three times and gives up with "no content after retries". - # Promote it to content + a ``content_filter`` finish reason so the - # loop's refusal handler surfaces it clearly and stops. ``refusal`` is - # ``None`` for normal responses, so this is a no-op in the common case. + # OpenAI structured refusal: ``message.refusal`` set, ``content`` empty. + # Proxies fronting Anthropic/Bedrock surface Claude refusals this way; without + # promotion the loop retries a deterministic refusal as an empty response. content = getattr(msg, "content", None) refusal = getattr(msg, "refusal", None) - if refusal is None and hasattr(msg, "model_extra"): - _msg_extra = getattr(msg, "model_extra", None) or {} - if isinstance(_msg_extra, dict): - refusal = _msg_extra.get("refusal") + if refusal is None: + refusal = model_extra.get("refusal") if isinstance(refusal, str) and refusal.strip(): - # Record the refusal explanation regardless — it's useful provider - # metadata even when the model also returned a usable payload. provider_data["refusal"] = refusal - _has_text = isinstance(content, str) and content.strip() - _has_tool_calls = bool(tool_calls) - # Only promote to a terminal ``content_filter`` when the refusal is - # the *sole* payload — no visible text and no tool calls. A response - # that carries real content (or tool calls) alongside a refusal note - # is a normal, usable turn: surfacing it as a failed safety refusal - # would discard the model's actual work. In the empty-payload case, - # adopt the refusal as content so the loop has something to show. - if not _has_text and not _has_tool_calls: + # Promote to a terminal ``content_filter`` only when the refusal is the + # sole payload — real text or tool calls alongside it is a usable turn. + if not (isinstance(content, str) and content.strip()) and not tool_calls: content = refusal if finish_reason in (None, "stop"): finish_reason = "content_filter" @@ -1128,33 +625,20 @@ class ChatCompletionsTransport(ProviderTransport): def validate_response(self, response: Any) -> bool: """Check that response has valid choices.""" - if response is None: - return False - if not hasattr(response, "choices") or response.choices is None: - return False - if not response.choices: - return False - return True + return bool(response is not None and getattr(response, "choices", None)) def extract_cache_stats(self, response: Any) -> dict[str, int] | None: - """Extract cache stats from prompt_tokens_details (OpenRouter/OpenAI) - or DeepSeek's native top-level prompt_cache_hit_tokens field.""" + """Cache stats from prompt_tokens_details (OpenRouter/OpenAI) or DeepSeek's top-level prompt_cache_hit_tokens.""" usage = getattr(response, "usage", None) if usage is None: return None details = getattr(usage, "prompt_tokens_details", None) cached = getattr(details, "cached_tokens", 0) or 0 if details else 0 written = getattr(details, "cache_write_tokens", 0) or 0 if details else 0 - if not cached: - # DeepSeek native API shape (api.deepseek.com): top-level - # prompt_cache_hit_tokens / prompt_cache_miss_tokens (#61871). - cached = getattr(usage, "prompt_cache_hit_tokens", 0) or 0 - if cached or written: - return {"cached_tokens": cached, "creation_tokens": written} - return None + cached = cached or getattr(usage, "prompt_cache_hit_tokens", 0) or 0 # DeepSeek native (#61871) + return {"cached_tokens": cached, "creation_tokens": written} if cached or written else None -# Auto-register on import from agent.transports import register_transport # noqa: E402 register_transport("chat_completions", ChatCompletionsTransport) diff --git a/agent/transports/codex.py b/agent/transports/codex.py index 99bd2f65e6..5772643eec 100644 --- a/agent/transports/codex.py +++ b/agent/transports/codex.py @@ -1,34 +1,14 @@ """OpenAI Responses API (Codex) transport. -Delegates to the existing adapter functions in agent/codex_responses_adapter.py. -This transport owns format conversion and normalization — NOT client lifecycle, -streaming, or the _run_codex_stream() call path. +Owns format conversion/normalization on top of agent/codex_responses_adapter.py — +NOT client lifecycle, streaming, or the _run_codex_stream() call path. """ import hashlib import json import logging import re -from typing import Any, Dict, List, Optional, Tuple - -logger = logging.getLogger(__name__) - -# Cron fires build session_id as ``cron__`` (see -# cron/scheduler.py). The trailing timestamp is per-fire noise; stripped so -# repeat fires of the same job share a cache scope (see #51395/#52295). -_CRON_SESSION_ID_RE = re.compile(r"^(cron_.+)_\d{8}_\d{6}$") - - -def _cache_scope_from_session_id(session_id: Optional[str]) -> str: - """Normalize a physical session_id into a stable logical cache scope. - - Every non-cron session_id already identifies one conversation/agent - instance (main run, a specific child/subagent, a sibling child, ...), - so it is used unchanged. Only cron's per-fire timestamp needs stripping. - """ - sid = str(session_id or "") - match = _CRON_SESSION_ID_RE.match(sid) - return match.group(1) if match else sid +from typing import Any, Callable, Dict, List, Optional, Tuple from agent.reasoning_effort import ( ACTUAL_RELAY_EFFORTS, @@ -40,76 +20,69 @@ from agent.reasoning_effort import ( from agent.transports.base import ProviderTransport from agent.transports.types import NormalizedResponse, ToolCall +logger = logging.getLogger(__name__) + +# Cron fires use ``cron__``; the per-fire timestamp is +# stripped so repeat fires of one job share a cache scope. +_CRON_SESSION_ID_RE = re.compile(r"^(cron_.+)_\d{8}_\d{6}$") + + +def _cache_scope_from_session_id(session_id: Optional[str]) -> str: + """Normalize a physical session_id into a stable logical cache scope.""" + sid = str(session_id or "") + match = _CRON_SESSION_ID_RE.match(sid) + return match.group(1) if match else sid + def _bounded_prompt_cache_key(value: Any) -> Optional[str]: - """Return a provider-safe cache key without changing session identity.""" - if value is None: - return None - key = str(value).strip() + """Return a provider-safe (<=64 char) cache key without changing session identity.""" + key = "" if value is None else str(value).strip() if not key: return None if len(key) <= 64: return key - # Match _content_cache_key's compact, collision-resistant routing-key shape. - digest = hashlib.sha256(key.encode("utf-8", errors="replace")).hexdigest()[:24] - return f"pck_{digest}" + return "pck_" + hashlib.sha256(key.encode("utf-8", errors="replace")).hexdigest()[:24] -# Wire-name used when Hermes keeps client-side web_search on xAI Responses. -# A function literally named ``web_search`` collides with Grok's native -# server-side tool (incomplete hang or HTTP 400 duplicate names); this alias -# avoids that while still dispatching through Hermes's configured provider -# (Firecrawl / Tavily / …). Mapped back to ``web_search`` in normalize_response. +def _bound_prompt_cache_key_field(container: Any) -> None: + """Bound (or drop, when empty) an in-place ``prompt_cache_key`` entry.""" + if isinstance(container, dict) and "prompt_cache_key" in container: + bounded = _bounded_prompt_cache_key(container["prompt_cache_key"]) + if bounded: + container["prompt_cache_key"] = bounded + else: + container.pop("prompt_cache_key", None) + + +def _str_headers(existing: Any) -> Dict[str, str]: + """Copy an ``extra_headers`` dict with str-coerced keys/values.""" + return {str(k): str(v) for k, v in existing.items() if k and v is not None} if isinstance(existing, dict) else {} + + +# Client-side ``web_search`` on xAI Responses collides with Grok's native +# server-side tool (incomplete hang / HTTP 400); it goes on the wire under this +# alias and is mapped back in normalize_response. _XAI_CLIENT_WEB_SEARCH_ALIAS = "hermes_web_search" -# OpenCode's /v1/responses endpoints (Zen and Go, including custom providers -# pointing at opencode.ai) reserve certain function names server-side and -# reject client tools that use them with HTTP 400 ("custom function name -# 'X' is reserved"). Reported for grok-4.5 on Go with `search_files` and -# `web_search` (#85589). Same treatment as the xAI web_search collision: -# rename on the wire (hermes_), map back in normalize_response so -# Hermes dispatch is unaffected. +# OpenCode /v1/responses rejects client tools using these names (HTTP 400 +# "custom function name 'X' is reserved", #85589); xAI rejects ``tool_search`` +# (reserved for Grok's native Tool Search, #95003). Aliased as hermes_. _OPENCODE_RESERVED_TOOL_NAMES = ("web_search", "search_files") - -# xAI reserves ``tool_search`` server-side for Grok's own native Tool Search -# and rejects *any* client function declared with that name: -# HTTP 400 {"code":"invalid-argument","error":"The function name -# tool_search is reserved for the tool_search tool"} -# Hermes's progressive-disclosure bridge registers exactly that literal -# (``tools.tool_search.TOOL_SEARCH_NAME``), and assembly is not provider -# gated, so with ``tools.tool_search.enabled: auto`` a grok turn dies the -# moment the catalog crosses the threshold — mid-session, and only for -# sessions large enough to activate the bridge. Same treatment as the two -# collisions above. ``tool_describe`` / ``tool_call`` are not reserved. -# Refs #95003. _XAI_RESERVED_TOOL_NAMES = ("tool_search",) - _RESERVED_TOOL_ALIAS_PREFIX = "hermes_" -_RESERVED_ALIAS_TO_NAME = { + +# Reverse map used ONLY when normalize_response runs on a transport that never +# built a request; real requests carry request-local ``_last_wire_aliases`` so a +# genuine tool named ``hermes_tool_search`` is never silently rewritten. +_LEGACY_ALIAS_FALLBACK = { f"{_RESERVED_TOOL_ALIAS_PREFIX}{name}": name for name in (*_OPENCODE_RESERVED_TOOL_NAMES, *_XAI_RESERVED_TOOL_NAMES) } - -# Legacy reverse map used ONLY when normalize_response runs on a transport -# instance that never built a request (normalize-only call sites / tests). -# Production requests carry request-local provenance instead — see -# ``_last_wire_aliases`` — so a real user/plugin/MCP tool that happens to be -# named ``hermes_tool_search`` is never silently rewritten to ``tool_search`` -# unless THIS request actually emitted that alias (#95003 review contract). -_LEGACY_ALIAS_FALLBACK = { - **_RESERVED_ALIAS_TO_NAME, - "hermes_web_search": "web_search", -} +_LEGACY_ALIAS_FALLBACK[_XAI_CLIENT_WEB_SEARCH_ALIAS] = "web_search" def _is_opencode_responses_backend(params: Dict[str, Any]) -> bool: - """True when this Responses request targets an OpenCode endpoint. - - Matches the built-in opencode-zen/go providers, custom ``opencode-go-*`` / - ``opencode-zen-*`` family providers, and any base_url hosted on - opencode.ai (covers custom providers with arbitrary names pointing at - the OpenCode gateway). - """ + """True for opencode-zen/go providers, ``opencode-*`` families, or opencode.ai hosts.""" try: from hermes_cli.models import opencode_provider_family @@ -128,38 +101,30 @@ def _is_opencode_responses_backend(params: Dict[str, Any]) -> bool: def _alias_reserved_tools( response_tools: List[Dict[str, Any]], reserved_names: Tuple[str, ...], + name_of: Callable[[dict], Any] = lambda t: t.get("name"), + rename: Callable[[dict, str], dict] = lambda t, alias: {**t, "name": alias}, ) -> Tuple[List[Dict[str, Any]], Dict[str, str]]: - """Alias provider-reserved client function names on the wire. + """Alias provider-reserved function names on the wire. - Single owner for every reserved-name collision on this transport. - Returns ``(rewritten_tools, alias_map)`` where ``alias_map`` maps each - wire alias emitted by THIS request back to the original tool name. - The caller stashes the map for ``normalize_response`` so the reverse - rewrite only ever applies to aliases this request actually sent — - a legitimate user/plugin/MCP tool already named ``hermes_`` is - neither shadowed (the alias picks a ``_2``/``_3`` suffix instead of - duplicating a wire name) nor mis-dispatched on the response path. + Returns ``(rewritten_tools, {alias: original_name})``. An alias that is + already taken by a real tool gets a ``_2``/``_3`` suffix rather than + duplicating a wire name. ``name_of``/``rename`` adapt the tool shape + (Responses ``{name}`` by default; chat_completions passes ``function.name``). """ rewritten: List[Dict[str, Any]] = [] alias_map: Dict[str, str] = {} - taken = { - tool.get("name") - for tool in response_tools - if isinstance(tool, dict) and tool.get("name") - } + taken = {name_of(tool) for tool in response_tools if isinstance(tool, dict) and name_of(tool)} for tool in response_tools: - if isinstance(tool, dict) and tool.get("name") in reserved_names: - base = f"{_RESERVED_TOOL_ALIAS_PREFIX}{tool['name']}" - alias = base + name = name_of(tool) if isinstance(tool, dict) else None + if name in reserved_names: + base = alias = f"{_RESERVED_TOOL_ALIAS_PREFIX}{name}" suffix = 2 while alias in taken: alias = f"{base}_{suffix}" suffix += 1 taken.add(alias) - alias_map[alias] = tool["name"] - aliased = dict(tool) - aliased["name"] = alias - rewritten.append(aliased) + alias_map[alias] = name + rewritten.append(rename(tool, alias)) else: rewritten.append(tool) return rewritten, alias_map @@ -168,12 +133,9 @@ def _alias_reserved_tools( def _xai_prefers_native_web_search() -> bool: """True when xAI Responses should use Grok's native ``web_search`` built-in. - Delegates to the web-search registry's provider resolution (which reads - ``web.search_backend`` / ``web.backend`` from config) and checks whether - the resolved provider is xAI. Falls back to the legacy ``_get_search_backend`` - probe when the registry has no providers loaded. On any resolution failure, - returns True (fail-closed to native — preserves the #48108 incomplete-hang - fix rather than risk reintroducing it). + Resolves via the web-search registry (config ``web.search_backend`` / + ``web.backend``), falling back to the legacy ``_get_search_backend`` probe. + Fails closed to native (True) on any error — preserves the #48108 fix. """ try: from agent.web_search_registry import get_active_search_provider @@ -186,36 +148,82 @@ def _xai_prefers_native_web_search() -> bool: return (_get_search_backend() or "").strip().lower() == "xai" except Exception: - # Fail closed to native — same behavior as pre-fix main. return True -def _rename_client_web_search_for_xai(response_tools: List[Dict[str, Any]]) -> List[Dict[str, Any]]: - """Rename client ``web_search`` → alias so xAI won't hijack it server-side.""" - rewritten: List[Dict[str, Any]] = [] - for tool in response_tools: - if isinstance(tool, dict) and tool.get("name") == "web_search": - aliased = dict(tool) - aliased["name"] = _XAI_CLIENT_WEB_SEARCH_ALIAS - rewritten.append(aliased) +def _alias_wire_tools( + response_tools: Any, params: Dict[str, Any], is_xai_responses: bool +) -> Tuple[Any, Dict[str, str]]: + """Apply provider-reserved tool-name aliasing; returns ``(tools, {alias: original})``. + + The alias map is wire provenance for THIS request — normalize_response + reverses only these. xAI: a client function named ``web_search`` collides + with Grok's native search (#48108); native mode swaps it 1:1 for the + built-in, client mode keeps Hermes dispatch under an alias. + """ + wire_aliases: Dict[str, str] = {} + def is_client_web_search(t: Any) -> bool: + return isinstance(t, dict) and t.get("name") == "web_search" + + if is_xai_responses and response_tools and any(is_client_web_search(t) for t in response_tools): + if _xai_prefers_native_web_search(): + response_tools = [t for t in response_tools if not is_client_web_search(t)] + [{"type": "web_search"}] else: - rewritten.append(tool) - return rewritten + response_tools = [ + {**t, "name": _XAI_CLIENT_WEB_SEARCH_ALIAS} if is_client_web_search(t) else t for t in response_tools + ] + wire_aliases[_XAI_CLIENT_WEB_SEARCH_ALIAS] = "web_search" + if response_tools and _is_opencode_responses_backend(params): + response_tools, _oc_aliases = _alias_reserved_tools(response_tools, _OPENCODE_RESERVED_TOOL_NAMES) + wire_aliases.update(_oc_aliases) + if is_xai_responses and response_tools: + response_tools, _xai_aliases = _alias_reserved_tools(response_tools, _XAI_RESERVED_TOOL_NAMES) + wire_aliases.update(_xai_aliases) + return response_tools, wire_aliases + + +def _resolve_reasoning(model: str, params: Dict[str, Any]) -> Tuple[Any, bool]: + """``(effort, enabled)`` for the request, effort clamped to the endpoint's vocabulary. + + Wire vocabularies live in agent.reasoning_effort; clamp_effort picks the + nearest weaker supported level and never escalates. A profile-declared + ``()`` means "no reasoning parameters accepted" (such backends 400 on any + reasoning field) and disables reasoning outright. + """ + reasoning_effort, reasoning_enabled = "medium", True + reasoning_config = params.get("reasoning_config") + if reasoning_config and isinstance(reasoning_config, dict): + if reasoning_config.get("enabled") is False: + reasoning_enabled = False + elif reasoning_config.get("effort"): + reasoning_effort = reasoning_config["effort"] + + if params.get("is_xai_responses", False): + from agent.model_metadata import is_grok_46_family + + # Grok 4.6 accepts xhigh; older Grok tops out at high. + supported = XAI_GROK46_EFFORTS if is_grok_46_family(model) else XAI_LEGACY_EFFORTS + elif (params.get("provider") or "").strip().lower() == "actual": + supported = ACTUAL_RELAY_EFFORTS + else: + declared = _profile_declared_efforts(params.get("provider"), model, params.get("base_url")) + if declared is not None and not declared: + reasoning_enabled = False + supported = declared or codex_supported_efforts(model) + return clamp_effort(reasoning_effort, supported), reasoning_enabled + + +def _merge_extra_headers(kwargs: Dict[str, Any], **headers: str) -> None: + """Merge str-coerced ``headers`` into ``kwargs['extra_headers']`` (SDK kwarg -> HTTP headers).""" + merged = _str_headers(kwargs.get("extra_headers")) + merged.update(headers) + kwargs["extra_headers"] = merged _EXTENDED_PROMPT_CACHE_MODELS = ( - "gpt-5.5-pro", - "gpt-5.5", - "gpt-5.4", - "gpt-5.2", - "gpt-5.1-codex-max", - "gpt-5.1-codex-mini", - "gpt-5.1-chat-latest", - "gpt-5.1-codex", - "gpt-5.1", - "gpt-5-codex", - "gpt-5", - "gpt-4.1", + "gpt-5.5-pro", "gpt-5.5", "gpt-5.4", "gpt-5.2", + "gpt-5.1-codex-max", "gpt-5.1-codex-mini", "gpt-5.1-chat-latest", "gpt-5.1-codex", "gpt-5.1", + "gpt-5-codex", "gpt-5", "gpt-4.1", ) _EXTENDED_PROMPT_CACHE_MODEL_RE = re.compile( rf"(?:^|[./:])(?:{'|'.join(re.escape(name) for name in _EXTENDED_PROMPT_CACHE_MODELS)})" @@ -223,55 +231,29 @@ _EXTENDED_PROMPT_CACHE_MODEL_RE = re.compile( ) -def _default_prompt_cache_retention_for_request( - model: str, - base_url: Any, -) -> Optional[str]: +def _default_prompt_cache_retention_for_request(model: str, base_url: Any) -> Optional[str]: """Return ``24h`` for supported hosts/models (Bedrock Mantle, Meta).""" from utils import base_url_hostname hostname = base_url_hostname(str(base_url or "")).lower() - # Meta Model API: prompt caching is opt-in via prompt_cache_retention. - # Measured 0% hits on /chat/completions vs 93-99% on /responses with 24h. + # Meta Model API: caching is opt-in via prompt_cache_retention (0% hits without). if hostname == "api.meta.ai": return "24h" - - hostname_parts = hostname.split(".") - is_bedrock_mantle = ( - len(hostname_parts) == 4 - and hostname_parts[0] == "bedrock-mantle" - and bool(hostname_parts[1]) - and hostname_parts[2:] == ["api", "aws"] - ) + parts = hostname.split(".") + is_bedrock_mantle = len(parts) == 4 and parts[0] == "bedrock-mantle" and bool(parts[1]) and parts[2:] == ["api", "aws"] if not is_bedrock_mantle: return None - normalized = str(model or "").strip().lower().replace("_", "-") - if _EXTENDED_PROMPT_CACHE_MODEL_RE.search(normalized): - return "24h" - return None + return "24h" if _EXTENDED_PROMPT_CACHE_MODEL_RE.search(normalized) else None def _content_cache_key( - instructions: str, - tools: Optional[List[Dict[str, Any]]], - scope_id: str = "", + instructions: str, tools: Optional[List[Dict[str, Any]]], scope_id: str = "" ) -> Optional[str]: - """Content-address the prompt cache key within a logical cache scope. + """``pck_`` of (scope_id, instructions, name-sorted tools), or None if nothing static. - Returns ``pck_`` of (scope_id + instructions + sorted tool - schemas), or None when there is nothing static to key on. The cache key - is a routing hint only — never a correctness boundary — so two requests - sharing a scope, system prompt, and tool set intentionally resolve to the - same warm prefix bucket. - - ``scope_id`` (pass ``_cache_scope_from_session_id(session_id)``) keeps - unrelated sessions — independent conversations, main vs. child/subagent, - sibling children — from concentrating onto the same bucket merely because - their static prefix matches (see #78941), while still letting recurring - cron fires of one job share a stable key across their timestamped - session_ids (the original #51395/#52295 fix this built on). Sorting tools - by name keeps the hash insertion-order independent. + The key is a routing hint only. ``scope_id`` keeps unrelated sessions off one + bucket while letting timestamped cron fires of one job share a warm prefix. """ if not instructions and not tools: return None @@ -281,44 +263,26 @@ def _content_cache_key( (t for t in tools if isinstance(t, dict)), key=lambda t: str(t.get("name") or t.get("type") or ""), ) - tools_part = json.dumps( - sorted_tools, sort_keys=True, ensure_ascii=False, separators=(",", ":") - ) - # \x00 separators so a scope/instructions/tools boundary can't be forged - # by content that happens to contain the same bytes. + tools_part = json.dumps(sorted_tools, sort_keys=True, ensure_ascii=False, separators=(",", ":")) + # \x00 separators so a boundary can't be forged by content containing the same bytes. content = f"{scope_id}\x00{instructions or ''}\x00{tools_part}" digest = hashlib.sha256(content.encode("utf-8", errors="replace")).hexdigest()[:24] return f"pck_{digest}" -def _profile_declared_efforts( - provider: Any, model: Optional[str], base_url: Any = None -) -> Optional[tuple]: - """Provider-profile-declared reasoning-effort vocabulary, or None. +def _profile_declared_efforts(provider: Any, model: Optional[str], base_url: Any = None) -> Optional[tuple]: + """Provider-profile-declared reasoning-effort vocabulary, or None (fail-open). - Thin, fail-open wrapper around - ``ProviderProfile.supported_reasoning_efforts`` (see providers/base.py - for the tri-state contract). Lazy import: provider plugins import this - transport during registry discovery, so a module-level import of - ``providers`` would cycle. - - Resolution is by provider name first, then by the endpoint's host: a - named custom provider pointed at a known provider's endpoint (e.g. a - ``providers.my-proxy`` entry with base_url ``https://api.router.com/v1``, - which the host mandate routes onto this transport) must get that - provider's declared vocabulary too — the host, not the config-entry - name, is what validates the request. + Resolves by provider name, then by endpoint host (a custom provider pointed + at a known host gets that host's vocabulary). Lazy import: provider plugins + import this transport during registry discovery. """ try: from providers import get_provider_profile name = str(provider or "").strip().lower() profile = get_provider_profile(name) if name else None - declared = ( - profile.supported_reasoning_efforts(model) - if profile is not None - else None - ) + declared = profile.supported_reasoning_efforts(model) if profile is not None else None if declared is None and base_url: from agent.model_metadata import _infer_provider_from_url @@ -328,71 +292,32 @@ def _profile_declared_efforts( if inferred_profile is not None: declared = inferred_profile.supported_reasoning_efforts(model) except Exception as exc: - # Fail-open by design: a broken profile hook must never block the - # request — the transport falls back to its default vocabulary. logger.debug("profile-declared efforts lookup failed: %s", exc) return None - if declared is None: - return None - return tuple(declared) + return None if declared is None else tuple(declared) def _is_azure_foundry_responses(params: Dict[str, Any]) -> bool: - """Return True for Microsoft Foundry's OpenAI-compatible Responses API. - - Matched on the registered provider id first, then on the endpoint host. - Host matching goes through ``base_url_host_matches`` rather than a - substring test, so a path or query segment carrying the Foundry domain - (``https://proxy.example.com/.services.ai.azure.com/v1``) is not - misclassified as Foundry. - """ + """True for Microsoft Foundry's Responses API (provider id, else host match — not substring).""" from utils import base_url_host_matches - provider = str(params.get("provider") or "").strip().lower() - if provider == "azure-foundry": + if str(params.get("provider") or "").strip().lower() == "azure-foundry": return True - - return base_url_host_matches( - str(params.get("base_url") or ""), "services.ai.azure.com" - ) + return base_url_host_matches(str(params.get("base_url") or ""), "services.ai.azure.com") def _is_post_tool_replay(messages: Optional[List[Dict[str, Any]]]) -> bool: - """Return True when ``messages`` end on a tool result awaiting a follow-up. + """True when ``messages`` end on a tool-result run issued by the preceding assistant turn. - Azure Foundry only rejects the *post-tool follow-up* payload — the shape - where a prior assistant ``function_call`` and its ``function_call_output`` - are replayed alongside an encrypted ``reasoning`` item (HTTP 400 - invalid_payload). Detecting that shape here keeps reasoning suppression - scoped to the failing turn, so ordinary (non-tool) Foundry multi-turn - continuity is left unchanged. - - The test is on the *trailing* messages, not on the history as a whole. - Scanning the whole history for any tool call plus any tool result makes - the predicate sticky: one tool call early in a conversation would then - suppress reasoning on every later turn, including plain user follow-ups - that Foundry accepts. The rejected payload is specifically the turn whose - last item is a tool result, so that is what this matches: the final - non-system message is a ``tool`` result, and the assistant message that - issued its ``tool_call_id`` is present. - - Tool-call identity is resolved the same way - ``_chat_messages_to_responses_input`` resolves it, because the pairing - that matters is the one that reaches the wire. A stored tool call can - carry the function call id in ``call_id``, in ``id``, or in a composite - ``"call_x|fc_y"`` id, and a bare ``fc_``-prefixed ``id`` is a response - item id that the converter turns into ``call_``. Matching only - ``id`` would miss the ``id=fc_… / call_id=call_…`` shape that resumed - legacy sessions and host-fed histories still use, and let the rejected - payload through. + Azure Foundry rejects only this post-tool follow-up shape when encrypted + reasoning is replayed, so the check is on the *trailing* messages (a whole- + history scan would make suppression sticky). Call ids are resolved the same + way ``_chat_messages_to_responses_input`` does (``call_id``, ``id``, + composite ``call_x|fc_y``, bare ``fc_`` item ids). """ - from agent.codex_responses_adapter import ( - _canonical_call_id_from_fc, - _split_responses_tool_id, - ) + from agent.codex_responses_adapter import _canonical_call_id_from_fc, _split_responses_tool_id def _pair_ids(raw: Any, explicit: Any = None) -> set: - """Every call id a stored tool id could pair on, converter-order.""" embedded_call_id, item_id = _split_responses_tool_id(raw) ids = {embedded_call_id} if embedded_call_id else set() if isinstance(explicit, str) and explicit.strip(): @@ -406,9 +331,7 @@ def _is_post_tool_replay(messages: Optional[List[Dict[str, Any]]]) -> bool: trailing = set() for msg in reversed(messages or ()): - if not isinstance(msg, dict): - return False - role = msg.get("role") + role = msg.get("role") if isinstance(msg, dict) else None if role == "system": continue if role == "tool": @@ -417,9 +340,7 @@ def _is_post_tool_replay(messages: Optional[List[Dict[str, Any]]]) -> bool: return False trailing |= ids continue - # First message before the trailing run of tool results. It must be - # the assistant turn that issued them for this to be the follow-up - # payload; a non-empty ``trailing`` is what proves the run existed. + # First non-tool message must be the assistant turn that issued the run. if role != "assistant": return False return any( @@ -427,44 +348,29 @@ def _is_post_tool_replay(messages: Optional[List[Dict[str, Any]]]) -> bool: for call in msg.get("tool_calls") or [] if isinstance(call, dict) ) - return False def _native_compaction_active(context_management: Any) -> bool: - """Is THIS request natively compacted? + """True only when the caller's eligibility gate produced a non-empty payload. - True only when the caller's eligibility gate - (``native_compaction.native_compaction_context_management``) produced a - non-empty payload. Every native-compaction side effect on the wire — - sending ``context_management``, replaying a ``type: "compaction"`` - checkpoint, restructuring the input around it — hangs off this one - predicate, so a checkpoint that outlives the gate (model swapped out of - the gpt-5.6 family, compression disabled, rejection kill switch, resumed - session) cannot keep reshaping requests on its own. + Every native-compaction wire effect hangs off this one predicate, so a + persisted checkpoint cannot keep reshaping requests after the gate closes. """ return isinstance(context_management, list) and bool(context_management) class ResponsesApiTransport(ProviderTransport): - """Transport for api_mode='codex_responses'. + """Transport for api_mode='codex_responses'.""" - Wraps the functions extracted into codex_responses_adapter.py (PR 1). - """ + # Codex response.status -> OpenAI finish_reason (caller checks incomplete_details). + _STOP_REASON_MAP = {"completed": "stop", "incomplete": "length", "failed": "stop", "cancelled": "stop"} - # Issuer kind of the most recent build_kwargs / convert_messages call. - # Used as a fallback when normalize_response is invoked without an - # explicit ``issuer_kind`` kwarg, so reasoning items captured from a - # response are stamped with the endpoint that minted them. Plain class - # attribute default; mutated on the instance, not the class. + # Issuer kind of the most recent build_kwargs/convert_messages call; fallback + # for normalize_response so captured reasoning items are stamped correctly. _last_issuer_kind: Optional[str] = None - - # Wire-alias provenance of the most recent build_kwargs call: - # ``{alias_sent_on_wire: original_tool_name}``. ``None`` means "no - # request built on this instance" (normalize-only call sites), in which - # case normalize_response falls back to the static legacy map. An empty - # dict means the last request emitted no aliases, so no reverse rewrite - # is permitted (#95003 provenance contract). + # ``{wire_alias: original_name}`` of the most recent build_kwargs call. None + # = no request built (use legacy map); {} = no aliases sent, no reverse rewrite. _last_wire_aliases: Optional[Dict[str, str]] = None @property @@ -490,13 +396,9 @@ class ResponsesApiTransport(ProviderTransport): messages, is_xai_responses=kwargs.get("is_xai_responses") is True, is_github_responses=kwargs.get("is_github_responses") is True, - replay_encrypted_reasoning=bool( - kwargs.get("replay_encrypted_reasoning", True) - ), + replay_encrypted_reasoning=bool(kwargs.get("replay_encrypted_reasoning", True)), current_issuer_kind=issuer, - native_compaction_eligible=_native_compaction_active( - kwargs.get("context_management") - ), + native_compaction_eligible=_native_compaction_active(kwargs.get("context_management")), ) def convert_tools(self, tools: List[Dict[str, Any]]) -> Any: @@ -511,225 +413,48 @@ class ResponsesApiTransport(ProviderTransport): tools: Optional[List[Dict[str, Any]]] = None, **params, ) -> Dict[str, Any]: - """Build Responses API kwargs. + """Build Responses API kwargs (calls convert_messages/convert_tools internally). - Calls convert_messages and convert_tools internally. - - params: - instructions: str — system prompt (extracted from messages[0] if not given) - reasoning_config: dict | None — {effort, enabled} - session_id: str | None — transcript/session id; drives the Codex - ``session_id`` header, and is the cache-scope fallback when no - ``cache_scope_id`` is given - cache_scope_id: str | None — rotation-stable logical scope id - (compression-lineage root; see agent/prompt_cache_scope.py). - Preferred over session_id when deriving the prompt_cache_key - content hash and the xAI x-grok-conv-id header; the Codex - x-client-request-id header mirrors the resulting body key. - Keeps the cache warm across context-compression session - rotation (#79017) - max_tokens: int | None — max_output_tokens - timeout: float | None — per-request timeout forwarded to the SDK - request_overrides: dict | None — extra kwargs merged in - provider: str | None — provider name for backend-specific logic - base_url: str | None — endpoint URL - base_url_hostname: str | None — hostname for backend detection - is_github_responses: bool — Copilot/GitHub models backend - is_codex_backend: bool — chatgpt.com/backend-api/codex - is_xai_responses: bool — xAI/Grok backend - github_reasoning_extra: dict | None — Copilot reasoning params + params: instructions, reasoning_config ({effort, enabled}), session_id + (transcript id; Codex ``session_id`` header; cache-scope fallback), + cache_scope_id (rotation-stable logical scope, preferred for the cache + key / xAI conv header), max_tokens, timeout, request_overrides, provider, + base_url, is_github_responses, is_codex_backend, is_xai_responses, + github_reasoning_extra, context_management, replay_encrypted_reasoning. """ - from agent.codex_responses_adapter import ( - _chat_messages_to_responses_input, - _responses_tools, - ) - + from agent.codex_responses_adapter import _chat_messages_to_responses_input, _responses_tools from run_agent import DEFAULT_AGENT_IDENTITY instructions = params.get("instructions", "") payload_messages = messages - if not instructions: - if messages and messages[0].get("role") == "system": - instructions = str(messages[0].get("content") or "").strip() - payload_messages = messages[1:] + if not instructions and messages and messages[0].get("role") == "system": + instructions = str(messages[0].get("content") or "").strip() + payload_messages = messages[1:] if not instructions: instructions = DEFAULT_AGENT_IDENTITY is_github_responses = params.get("is_github_responses") is True is_codex_backend = params.get("is_codex_backend") is True is_xai_responses = params.get("is_xai_responses") is True - replay_encrypted_reasoning = bool( - params.get("replay_encrypted_reasoning", True) - ) - if replay_encrypted_reasoning and _is_azure_foundry_responses(params): - # Microsoft Foundry accepts the initial Responses function-call - # request and ordinary (non-tool) multi-turn continuity, but - # rejects the post-tool follow-up payload that carries prior - # encrypted reasoning items alongside function_call / - # function_call_output, with HTTP 400 invalid_payload. Scope the - # suppression to that follow-up turn: keep function_call / - # function_call_output continuity intact and drop only the - # encrypted reasoning replay for this endpoint. - if _is_post_tool_replay(payload_messages): - replay_encrypted_reasoning = False - # Native server-side compaction (gpt-5.6 on direct OpenAI/Codex routes - # only). The caller resolves eligibility via - # agent.native_compaction.native_compaction_context_management(); - # None means the field is never added to the request. + replay_encrypted_reasoning = bool(params.get("replay_encrypted_reasoning", True)) + # Foundry 400s on encrypted-reasoning replay only in the post-tool + # follow-up turn; suppress replay for exactly that shape. + if replay_encrypted_reasoning and _is_azure_foundry_responses(params) and _is_post_tool_replay(payload_messages): + replay_encrypted_reasoning = False + # Single source of truth: the same predicate decides whether + # context_management goes out AND whether the converter may replay a checkpoint. context_management = params.get("context_management") - # Single source of truth for "this request is natively compacted": - # the same value decides whether the field goes out AND whether the - # converter may replay/prune around a compaction checkpoint. Keeping - # them derived from one expression is what stops a persisted - # checkpoint from restructuring the wire after the gate closes. native_compaction_active = _native_compaction_active(context_management) - # Resolve the issuing endpoint for this call. Stashed on the - # transport so normalize_response can stamp it onto reasoning - # items captured from the response, and passed to the input - # converter so foreign-issuer reasoning blocks in history are - # dropped before the API rejects them. issuer_kind = self._resolve_issuer_kind(params) self._last_issuer_kind = issuer_kind + reasoning_effort, reasoning_enabled = _resolve_reasoning(model, params) + response_tools, self._last_wire_aliases = _alias_wire_tools(_responses_tools(tools), params, is_xai_responses) - # Resolve reasoning effort - reasoning_effort = "medium" - reasoning_enabled = True - reasoning_config = params.get("reasoning_config") - if reasoning_config and isinstance(reasoning_config, dict): - if reasoning_config.get("enabled") is False: - reasoning_enabled = False - elif reasoning_config.get("effort"): - reasoning_effort = reasoning_config["effort"] - - # Wire vocabularies are declared in agent.reasoning_effort; the shared - # clamp policy (nearest weaker supported level, never escalate, - # never invert the ladder) replaces the per-backend hand maps that - # repeatedly leaked internal levels like "ultra" to the wire - # (#89503 class) or clamped one rung below a model's real ceiling - # (#87279). - if params.get("is_xai_responses", False): - from agent.model_metadata import is_grok_46_family - - # Grok 4.6 accepts xhigh as a wire value; older Grok tops out - # at high. - _supported = ( - XAI_GROK46_EFFORTS if is_grok_46_family(model) - else XAI_LEGACY_EFFORTS - ) - elif (params.get("provider") or "").strip().lower() == "actual": - # Actual Computer relays to SGLang/vLLM backends: - # none/low/medium/high/max. - _supported = ACTUAL_RELAY_EFFORTS - else: - # Profile-declared vocabulary first: gateways that validate - # reasoning.effort per model (Ramp Router reads its live catalog) - # declare it via ProviderProfile.supported_reasoning_efforts. - # ``()`` is the definitive "this model takes no reasoning - # parameters" verdict — such backends 400 on any reasoning field - # rather than ignoring it, so suppress reasoning entirely. - _supported = None - _declared = _profile_declared_efforts( - params.get("provider"), model, params.get("base_url") - ) - if _declared is not None: - if not _declared: - reasoning_enabled = False - else: - _supported = _declared - if _supported is None: - # OpenAI/Codex Responses backend — per-model vocabulary - # (live-verified: "max" is gpt-5.6-only, "minimal" always - # rejected). #68365 premise confirmed. - _supported = codex_supported_efforts(model) - reasoning_effort = clamp_effort(reasoning_effort, _supported) - - response_tools = _responses_tools(tools) - - # xAI server-side web search vs Hermes web providers. - # - # grok models on xAI's /v1/responses surface have a *native*, - # server-executed web search. A client-side function literally named - # ``web_search`` collides with that engine: declared as a plain - # ``function`` rather than ``{"type": "web_search"}``, the search - # dispatches but never reconciles → incomplete turn + 3 retries. - # Verified live against grok-composer-2.5-fast (2026-06); see #48108. - # - # Two modes, chosen by the user's web-search backend config: - # - # 1. **Native** (active/configured backend is ``xai``, or resolution - # fails): drop the client ``web_search`` function and declare - # xAI's built-in instead. 1:1 swap only when client ``web_search`` - # was already present — never an additive grant. - # 2. **Client** (Firecrawl / Tavily / Exa / … configured or resolved): - # keep Hermes dispatch so ``web.backend`` / ``web.search_backend`` - # is honored, but rename the wire tool to - # ``hermes_web_search`` so Grok cannot hijack the name. The alias - # is mapped back to ``web_search`` in ``normalize_response``. - # Request-local alias provenance: every wire alias THIS request - # emits is recorded here and stashed on the transport, so the - # reverse rewrite in ``normalize_response`` applies only to aliases - # that were actually sent (never to a real tool that merely shares - # an alias-shaped name). - wire_aliases: Dict[str, str] = {} - if is_xai_responses and response_tools: - has_client_web_search = any( - isinstance(t, dict) and t.get("name") == "web_search" - for t in response_tools - ) - if has_client_web_search: - if _xai_prefers_native_web_search(): - filtered = [ - t for t in response_tools - if not (isinstance(t, dict) and t.get("name") == "web_search") - ] - filtered.append({"type": "web_search"}) - response_tools = filtered - else: - response_tools = _rename_client_web_search_for_xai(response_tools) - wire_aliases[_XAI_CLIENT_WEB_SEARCH_ALIAS] = "web_search" - - # OpenCode Responses backends reserve web_search / search_files as - # function names (HTTP 400 "custom function name 'X' is reserved", - # #85589). Alias them on the wire; normalize_response maps them back. - if response_tools and _is_opencode_responses_backend(params): - response_tools, _oc_aliases = _alias_reserved_tools( - response_tools, _OPENCODE_RESERVED_TOOL_NAMES - ) - wire_aliases.update(_oc_aliases) - - # xAI reserves ``tool_search`` for its native server-side tool and - # rejects the client declaration outright (#95003). Alias it on the - # wire; normalize_response maps it back before dispatch. - if is_xai_responses and response_tools: - response_tools, _xai_aliases = _alias_reserved_tools( - response_tools, _XAI_RESERVED_TOOL_NAMES - ) - wire_aliases.update(_xai_aliases) - - # Stash for normalize_response (same request/response pairing model - # as ``_last_issuer_kind``). An empty dict is meaningful: it means - # this request emitted NO aliases, so no reverse rewrite may run. - self._last_wire_aliases = wire_aliases - - # ``tools`` MUST be omitted entirely when there are no functions to - # expose: the openai SDK's ``responses.stream()`` / ``responses.parse()`` - # eagerly call ``_make_tools(tools)`` which does ``for tool in tools`` - # without a None guard, so passing ``tools=None`` raises - # ``TypeError: 'NoneType' object is not iterable`` before any HTTP - # request is issued (openai==2.24.0). Reported for the - # ``openai-codex`` / ``gpt-5.5`` combo on chatgpt.com/backend-api/codex - # (#32892) when the agent runs without external tools registered. - # Function-level import: agent.model_metadata is imported lazily - # because provider plugins import this transport during - # model_metadata's own module init (circular otherwise). - from agent.model_metadata import ( - strip_codex_context_variant_suffix as _strip_ctx_variant, - ) + # Lazy: provider plugins import this transport during model_metadata init. + from agent.model_metadata import strip_codex_context_variant_suffix as _strip_ctx_variant kwargs = { - # ``-900k`` large-context picker variants are Hermes-side aliases - # (gpt-5.6-sol-900k etc.) — the Codex/OpenAI backend only knows - # the base slug, so strip the suffix before it hits the wire. + # ``-900k`` picker variants are Hermes-side aliases; the backend knows only the base slug. "model": _strip_ctx_variant(model), "instructions": instructions, "input": _chat_messages_to_responses_input( @@ -742,6 +467,8 @@ class ResponsesApiTransport(ProviderTransport): ), "store": False, } + # ``tools`` MUST be omitted when empty: the openai SDK iterates it without + # a None guard before any HTTP request (#32892). if response_tools: kwargs["tools"] = response_tools kwargs["tool_choice"] = "auto" @@ -750,53 +477,25 @@ class ResponsesApiTransport(ProviderTransport): kwargs["context_management"] = context_management session_id = params.get("session_id") - # prompt_cache_key is content-addressed from the static prefix - # (instructions + tools) scoped by session, NOT the raw session_id — - # recurring cron jobs carry a per-fire timestamp in session_id - # (cron__) that made every run cache-cold, so the scope strips - # that suffix (see _cache_scope_from_session_id). session_id is left - # untouched for transcript isolation (the Codex ``session_id`` header - # below). Falls back to session_id when there is no static content to - # hash. - # - # cache_scope_id, when provided, is the rotation-stable logical scope - # (compression-lineage root — agent/prompt_cache_scope.py): legacy - # ``compression.in_place: false`` compaction rotates session_id - # mid-conversation, and scoping by the physical id went cache-cold at - # every rotation boundary (#79017). - _cache_scope = _cache_scope_from_session_id( - params.get("cache_scope_id") or session_id - ) - cache_key = _content_cache_key( - instructions, response_tools, _cache_scope - ) or _cache_scope - # xAI Responses takes prompt_cache_key in extra_body (set further - # down); GitHub Models opts out of cache-key routing entirely. + # Cache key is content-addressed (instructions + tools) within a logical + # scope — cache_scope_id survives compression rotation (#79017), and the + # cron per-fire timestamp is stripped. session_id itself stays untouched + # for transcript isolation (Codex ``session_id`` header below). + _cache_scope = _cache_scope_from_session_id(params.get("cache_scope_id") or session_id) + cache_key = _content_cache_key(instructions, response_tools, _cache_scope) or _cache_scope + # xAI takes prompt_cache_key in extra_body (below); GitHub Models opts out entirely. if not is_github_responses and not is_xai_responses and cache_key: kwargs["prompt_cache_key"] = cache_key - cache_retention = _default_prompt_cache_retention_for_request( - model, - params.get("base_url"), - ) + cache_retention = _default_prompt_cache_retention_for_request(model, params.get("base_url")) if cache_retention: kwargs.setdefault("prompt_cache_retention", cache_retention) if reasoning_enabled and is_xai_responses: from agent.model_metadata import grok_supports_reasoning_effort - # Ask xAI to echo back encrypted reasoning items so we can - # replay them on subsequent turns for cross-turn coherence. - # See agent/codex_responses_adapter._chat_messages_to_responses_input - # for the May 2026 reversal of the earlier suppression gate. - kwargs["include"] = ( - ["reasoning.encrypted_content"] if replay_encrypted_reasoning else [] - ) - # xAI rejects `reasoning.effort` on grok-4 / grok-4-fast / grok-3 - # / grok-code-fast / grok-4.20-0309-* with HTTP 400 even though - # those models reason natively. Only send the effort dial when - # the target model is on the allowlist; otherwise send no - # `reasoning` key at all and let the model reason on its own. + kwargs["include"] = ["reasoning.encrypted_content"] if replay_encrypted_reasoning else [] + # xAI 400s on ``reasoning.effort`` for models outside the allowlist. if grok_supports_reasoning_effort(model): kwargs["reasoning"] = {"effort": reasoning_effort} elif reasoning_enabled: @@ -806,9 +505,7 @@ class ResponsesApiTransport(ProviderTransport): kwargs["reasoning"] = github_reasoning else: kwargs["reasoning"] = {"effort": reasoning_effort, "summary": "auto"} - kwargs["include"] = ( - ["reasoning.encrypted_content"] if replay_encrypted_reasoning else [] - ) + kwargs["include"] = ["reasoning.encrypted_content"] if replay_encrypted_reasoning else [] elif not is_github_responses and not is_xai_responses: kwargs["include"] = [] @@ -816,204 +513,118 @@ class ResponsesApiTransport(ProviderTransport): if request_overrides: kwargs.update(request_overrides) - if "prompt_cache_key" in kwargs: - bounded_cache_key = _bounded_prompt_cache_key(kwargs["prompt_cache_key"]) - if bounded_cache_key: - kwargs["prompt_cache_key"] = bounded_cache_key - else: - kwargs.pop("prompt_cache_key", None) + _bound_prompt_cache_key_field(kwargs) - # Older xAI Responses models reject ``service_tier`` (HTTP 400 - # "Argument not supported: service_tier"). Grok 4.6 accepts Priority - # Processing, but continue stripping stale or unsupported tier values - # on every other xAI path. See #28490 and #84799. + # Older xAI models reject ``service_tier`` (HTTP 400); only Grok 4.6 + # accepts Priority Processing (#28490, #84799). if is_xai_responses: from agent.model_metadata import is_grok_46_family - if not ( - is_grok_46_family(model) - and kwargs.get("service_tier") == "priority" - ): + if not (is_grok_46_family(model) and kwargs.get("service_tier") == "priority"): kwargs.pop("service_tier", None) - # Forward per-request timeout to the SDK so OpenAI/Anthropic clients - # honor it. Without this, ``providers..request_timeout_seconds`` - # is silently dropped on the main agent Codex path while the - # chat_completions path and auxiliary Codex adapter both forward it. + # Forward per-request timeout to the SDK (providers..request_timeout_seconds). timeout = kwargs.get("timeout", params.get("timeout")) - if ( - isinstance(timeout, (int, float)) - and not isinstance(timeout, bool) - and 0 < float(timeout) < float("inf") - ): + if isinstance(timeout, (int, float)) and not isinstance(timeout, bool) and 0 < float(timeout) < float("inf"): kwargs["timeout"] = float(timeout) else: kwargs.pop("timeout", None) if is_codex_backend: - # The Codex backend rejects body-level ``extra_headers`` with - # HTTP 400, but the OpenAI SDK's ``extra_headers`` kwarg maps - # to actual HTTP request headers (not body fields). ``session_id`` - # carries the raw physical session id — transcript/identity, per - # the #57012 contract — while ``x-client-request-id`` mirrors the - # body's effective ``prompt_cache_key`` so header and body always - # agree on the same routing bucket instead of diverging (#78941). + # Codex rejects body-level extra_headers, but the SDK kwarg maps to + # HTTP headers. ``session_id`` = raw physical id (transcript identity); + # ``x-client-request-id`` mirrors the body cache key so both agree (#78941). final_cache_key = kwargs.get("prompt_cache_key") or _bounded_prompt_cache_key(_cache_scope) - if session_id or final_cache_key: - existing_extra_headers = kwargs.get("extra_headers") - merged_extra_headers: Dict[str, str] = {} - if isinstance(existing_extra_headers, dict): - merged_extra_headers.update( - { - str(key): str(value) - for key, value in existing_extra_headers.items() - if key and value is not None - } - ) - if session_id: - merged_extra_headers["session_id"] = str(session_id) - if final_cache_key: - merged_extra_headers["x-client-request-id"] = final_cache_key - kwargs["extra_headers"] = merged_extra_headers + headers = {} + if session_id: + headers["session_id"] = str(session_id) + if final_cache_key: + headers["x-client-request-id"] = final_cache_key + if headers: + _merge_extra_headers(kwargs, **headers) max_tokens = params.get("max_tokens") if max_tokens is not None and not is_codex_backend: kwargs["max_output_tokens"] = max_tokens if is_xai_responses and session_id: - existing_extra_headers = kwargs.get("extra_headers") - merged_extra_headers: Dict[str, str] = {} - if isinstance(existing_extra_headers, dict): - merged_extra_headers.update( - { - str(key): str(value) - for key, value in existing_extra_headers.items() - if key and value is not None - } - ) - # Scoped like the body cache key below — otherwise cron's - # per-fire timestamp in session_id (cron__) pins every - # fire of the same job to a different xAI backend server (#78941). - merged_extra_headers["x-grok-conv-id"] = _cache_scope - kwargs["extra_headers"] = merged_extra_headers - - # xAI Responses cache-routing — body-level field per - # https://docs.x.ai/developers/advanced-api-usage/prompt-caching/maximizing-cache-hits. - # Sent via extra_body (not the typed kwarg) so it survives openai - # SDK builds whose Responses.stream() signature has dropped the field. - # A caller's request_overrides={"prompt_cache_key": ...} lands on - # the top-level kwarg set above — read it back here so an explicit - # override actually governs the field xAI reads, instead of being - # silently outrun by the auto-derived cache_key (#78941). + # Scoped like the body key so cron's per-fire timestamp doesn't pin + # each fire to a different xAI backend server (#78941). + _merge_extra_headers(kwargs, **{"x-grok-conv-id": _cache_scope}) + # xAI reads prompt_cache_key from the body; sent via extra_body so it + # survives SDK builds whose Responses.stream() dropped the typed kwarg. + # An explicit request_overrides value (top-level kwarg) wins. existing_extra_body = kwargs.get("extra_body") - merged_extra_body: Dict[str, Any] = {} - if isinstance(existing_extra_body, dict): - merged_extra_body.update(existing_extra_body) - merged_extra_body.setdefault( - "prompt_cache_key", kwargs.get("prompt_cache_key", cache_key) - ) - kwargs["extra_body"] = merged_extra_body - - extra_body = kwargs.get("extra_body") - if isinstance(extra_body, dict) and "prompt_cache_key" in extra_body: - bounded_cache_key = _bounded_prompt_cache_key(extra_body["prompt_cache_key"]) - if bounded_cache_key: - extra_body["prompt_cache_key"] = bounded_cache_key - else: - extra_body.pop("prompt_cache_key", None) + kwargs["extra_body"] = dict(existing_extra_body) if isinstance(existing_extra_body, dict) else {} + kwargs["extra_body"].setdefault("prompt_cache_key", kwargs.get("prompt_cache_key", cache_key)) + _bound_prompt_cache_key_field(kwargs.get("extra_body")) return kwargs def normalize_response(self, response: Any, **kwargs) -> NormalizedResponse: """Normalize Codex Responses API response to NormalizedResponse.""" - from agent.codex_responses_adapter import ( - _normalize_codex_response, - ) + from agent.codex_responses_adapter import _normalize_codex_response - # Issuer for this response = explicit kwarg if the caller knows it, - # otherwise the stash from the matching build_kwargs/convert_messages - # call. Either way it gets stamped onto reasoning items so future - # turns can detect a model swap and drop foreign-issuer blobs. + # Explicit issuer if the caller knows it, else the stash from the paired + # build_kwargs/convert_messages call. issuer_kind = kwargs.get("issuer_kind") or self._last_issuer_kind - # _normalize_codex_response returns (SimpleNamespace, finish_reason_str) msg, finish_reason = _normalize_codex_response(response, issuer_kind=issuer_kind) tool_calls = None if msg and msg.tool_calls: tool_calls = [] + alias_map = self._last_wire_aliases for tc in msg.tool_calls: - provider_data = {} - if hasattr(tc, "call_id") and tc.call_id: - provider_data["call_id"] = tc.call_id - if hasattr(tc, "response_item_id") and tc.response_item_id: - provider_data["response_item_id"] = tc.response_item_id - name = tc.function.name if hasattr(tc, "function") else getattr(tc, "name", "") - # Undo THIS request's wire aliases before Hermes dispatch. - # Request-local provenance: only aliases the paired - # build_kwargs call actually emitted are rewritten, so a - # legitimate tool that happens to be named - # ``hermes_tool_search`` etc. is dispatched as itself when - # no alias was sent. The static legacy map is used only for - # normalize-only call sites that never built a request on - # this transport instance. - alias_map = self._last_wire_aliases + provider_data = { + key: getattr(tc, key) for key in ("call_id", "response_item_id") if getattr(tc, key, None) + } + has_fn = hasattr(tc, "function") + name = tc.function.name if has_fn else getattr(tc, "name", "") + # Undo only aliases THIS request emitted; the legacy map applies + # solely to normalize-only call sites that never built a request. if alias_map is None: - if name == _XAI_CLIENT_WEB_SEARCH_ALIAS: - name = "web_search" - elif name in _LEGACY_ALIAS_FALLBACK: - name = _LEGACY_ALIAS_FALLBACK[name] + name = _LEGACY_ALIAS_FALLBACK.get(name, name) elif name in alias_map: name = alias_map[name] tool_calls.append(ToolCall( id=tc.id if hasattr(tc, "id") else (name or None), name=name, - arguments=tc.function.arguments if hasattr(tc, "function") else getattr(tc, "arguments", "{}"), + arguments=tc.function.arguments if has_fn else getattr(tc, "arguments", "{}"), provider_data=provider_data or None, )) - # Extract reasoning items for provider_data provider_data = {} - if msg and hasattr(msg, "codex_reasoning_items") and msg.codex_reasoning_items: - provider_data["codex_reasoning_items"] = msg.codex_reasoning_items - if msg and hasattr(msg, "codex_message_items") and msg.codex_message_items: - provider_data["codex_message_items"] = msg.codex_message_items - if msg and hasattr(msg, "reasoning_details") and msg.reasoning_details: - provider_data["reasoning_details"] = msg.reasoning_details + if msg: + for key in ("codex_reasoning_items", "codex_message_items", "reasoning_details"): + value = getattr(msg, key, None) + if value: + provider_data[key] = value return NormalizedResponse( content=msg.content if msg else None, tool_calls=tool_calls, finish_reason=finish_reason or "stop", - reasoning=msg.reasoning if msg and hasattr(msg, "reasoning") else None, + reasoning=getattr(msg, "reasoning", None) if msg else None, usage=None, # Codex usage is extracted separately in normalize_usage() provider_data=provider_data or None, ) def validate_response(self, response: Any) -> bool: - """Check Codex Responses API response has valid output structure. + """True if response.output is a non-empty list, or a terminal content_filter refusal. - Returns True only if response.output is a non-empty list. Also treats - terminal content-filter incomplete responses as valid: the Responses API - may return status=incomplete with incomplete_details.reason='content_filter' - and no output items. That is a provider refusal signal, not a malformed - response, and must reach normalization so the agent loop can use the - content-policy / fallback path instead of invalid-response retries. - - Does NOT check output_text fallback — the caller handles that with - diagnostic logging for stream backfill recovery. + A status=incomplete / reason=content_filter response with no output is a + provider refusal signal that must reach normalization, not a retry. + Does NOT check output_text fallback — the caller handles that. """ if response is None: return False output = getattr(response, "output", None) - if not isinstance(output, list) or not output: - status = str(getattr(response, "status", "") or "").strip().lower() - incomplete_details = getattr(response, "incomplete_details", None) - if isinstance(incomplete_details, dict): - reason = str(incomplete_details.get("reason") or "").strip().lower() - else: - reason = str(getattr(incomplete_details, "reason", "") or "").strip().lower() - return status == "incomplete" and reason == "content_filter" - return True + if isinstance(output, list) and output: + return True + status = str(getattr(response, "status", "") or "").strip().lower() + details = getattr(response, "incomplete_details", None) + raw_reason = details.get("reason") if isinstance(details, dict) else getattr(details, "reason", "") + return status == "incomplete" and str(raw_reason or "").strip().lower() == "content_filter" def preflight_kwargs( self, @@ -1025,49 +636,19 @@ class ResponsesApiTransport(ProviderTransport): ) -> dict: """Validate and sanitize Codex API kwargs before the call. - Normalizes input items, strips unsupported fields, validates structure. ``sanitize_harmony_tokens`` is enabled only for the ChatGPT Codex backend, which rejects literal reserved Harmony wire tokens in text. """ from agent.codex_responses_adapter import _preflight_codex_api_kwargs normalized = _preflight_codex_api_kwargs( - api_kwargs, - allow_stream=allow_stream, - is_github_responses=is_github_responses, + api_kwargs, allow_stream=allow_stream, is_github_responses=is_github_responses, sanitize_harmony_tokens=sanitize_harmony_tokens, ) - if "prompt_cache_key" in normalized: - bounded = _bounded_prompt_cache_key(normalized["prompt_cache_key"]) - if bounded: - normalized["prompt_cache_key"] = bounded - else: - normalized.pop("prompt_cache_key", None) - extra_body = normalized.get("extra_body") - if isinstance(extra_body, dict) and "prompt_cache_key" in extra_body: - bounded = _bounded_prompt_cache_key(extra_body["prompt_cache_key"]) - if bounded: - extra_body["prompt_cache_key"] = bounded - else: - extra_body.pop("prompt_cache_key", None) + _bound_prompt_cache_key_field(normalized) + _bound_prompt_cache_key_field(normalized.get("extra_body")) return normalized - def map_finish_reason(self, raw_reason: str) -> str: - """Map Codex response.status to OpenAI finish_reason. - - Codex uses response.status ('completed', 'incomplete') + - response.incomplete_details.reason for granular mapping. - This method handles the simple status string; the caller - should check incomplete_details separately for 'max_output_tokens'. - """ - _MAP = { - "completed": "stop", - "incomplete": "length", - "failed": "stop", - "cancelled": "stop", - } - return _MAP.get(raw_reason, "stop") - # Auto-register on import from agent.transports import register_transport # noqa: E402 diff --git a/agent/transports/codex_app_server.py b/agent/transports/codex_app_server.py index c23ff836ed..9686118c59 100644 --- a/agent/transports/codex_app_server.py +++ b/agent/transports/codex_app_server.py @@ -1,17 +1,10 @@ """Codex app-server JSON-RPC client. -Speaks the protocol documented in codex-rs/app-server/README.md (codex 0.125+). -Transport is newline-delimited JSON-RPC 2.0 over stdio: spawn `codex app-server`, -do an `initialize` handshake, then drive `thread/start` + `turn/start` and -consume streaming `item/*` notifications until `turn/completed`. - -This module is the wire-level speaker only. Higher-level concerns (event -projection into Hermes' display, approval bridging, transcript projection into -AIAgent.messages, plugin migration) live in sibling modules. - -Status: optional opt-in runtime gated behind `model.openai_runtime == -"codex_app_server"`. Hermes' default tool dispatch is unchanged when this -runtime is not selected. +Newline-delimited JSON-RPC 2.0 over stdio to ``codex app-server`` (codex 0.125+): +``initialize`` handshake, then ``thread/start`` + ``turn/start`` with streaming +``item/*`` notifications until ``turn/completed``. Wire-level speaker only — +projection, approvals and transcript handling live in sibling modules. +Opt-in runtime gated behind ``model.openai_runtime == "codex_app_server"``. """ from __future__ import annotations @@ -19,6 +12,7 @@ from __future__ import annotations import json import os import queue +import re import subprocess import threading import time @@ -27,8 +21,6 @@ from typing import Any, Optional from tools.environments.local import hermes_subprocess_env -# Default minimum codex version we test against. The PR sets this from the -# `codex --version` parsed at install time; bumping is a one-line change here. MIN_CODEX_VERSION = (0, 125, 0) @@ -52,20 +44,12 @@ class _Pending: class CodexAppServerClient: - """Minimal JSON-RPC 2.0 client for `codex app-server` over stdio. + """Minimal synchronous JSON-RPC 2.0 client for ``codex app-server`` over stdio. - Threading model: - - Spawning thread (caller) drives request/response pairs synchronously. - - One reader thread parses stdout, dispatches replies to the right - pending future, and routes notifications + server-initiated requests - to bounded queues that the caller drains on their own cadence. - - One reader thread captures stderr for diagnostics; codex emits - tracing logs there at RUST_LOG-controlled levels. - - Intentionally NOT async. AIAgent.run_conversation() is synchronous and - runs on the main thread; layering asyncio just to drive a stdio child - creates surprising interrupt semantics. We use blocking queues with - timeouts and rely on `turn/interrupt` for cancellation. + The caller drives request/response pairs; one reader thread routes replies + to pending queues and notifications / server requests to bounded queues; + another captures stderr. Intentionally NOT async: AIAgent.run_conversation() + is synchronous and cancellation goes through ``turn/interrupt``. """ def __init__( @@ -76,17 +60,8 @@ class CodexAppServerClient: env: Optional[dict[str, str]] = None, ) -> None: self._codex_bin = codex_bin - # codex app-server is a model-driving CLI executor: it runs a - # model-chosen agentic loop that executes shell commands, so it - # legitimately needs LLM provider credentials (inherit_credentials=True) - # to authenticate against the model endpoint. But the previous - # `os.environ.copy()` also handed it every Tier-1 Hermes secret — gateway - # bot tokens, GitHub auth, Modal/Daytona infra tokens, the dashboard - # session token, AUXILIARY_* side-LLM keys, GATEWAY_RELAY_* auth — none - # of which a coding subprocess has any use for. Route through the - # centralized helper so Tier-1 + dynamic-internal secrets are always - # stripped while provider creds still flow, matching copilot_acp_client - # (#29157 sibling spawn-site gap). + # codex needs LLM provider creds (inherit_credentials=True) but must not + # receive Tier-1 Hermes secrets (gateway/GitHub/infra tokens) — #29157. spawn_env = hermes_subprocess_env(inherit_credentials=True) if env: spawn_env.update(env) @@ -94,51 +69,28 @@ class CodexAppServerClient: spawn_env["CODEX_HOME"] = codex_home app_server_args = list(extra_args or []) - # Kanban workers must be able to write their handoff/status back to - # the board DB, which lives outside the per-task workspace. Keep the - # Codex sandbox on, but add the Kanban root as the only extra writable - # root. Without this, codex-runtime workers finish their actual work - # but crash/block when kanban_complete/kanban_block writes SQLite. + # Kanban workers must write handoff/status to the board DB outside the + # workspace: keep the sandbox on, add the Kanban root as writable. if spawn_env.get("HERMES_KANBAN_TASK"): kanban_db = spawn_env.get("HERMES_KANBAN_DB") - kanban_root = ( - os.path.dirname(kanban_db) - if kanban_db - else spawn_env.get( - "HERMES_KANBAN_ROOT", - os.path.join( - spawn_env.get("HERMES_HOME", os.path.expanduser("~/.hermes")), - "kanban", - ), - ) - ) - app_server_args.extend( - [ - "-c", - 'sandbox_mode="workspace-write"', - "-c", - f'sandbox_workspace_write.writable_roots=["{kanban_root}"]', - "-c", - "sandbox_workspace_write.network_access=false", - ] - ) + default_root = os.path.join(spawn_env.get("HERMES_HOME", os.path.expanduser("~/.hermes")), "kanban") + kanban_root = os.path.dirname(kanban_db) if kanban_db else spawn_env.get("HERMES_KANBAN_ROOT", default_root) + app_server_args.extend([ + "-c", 'sandbox_mode="workspace-write"', + "-c", f'sandbox_workspace_write.writable_roots=["{kanban_root}"]', + "-c", "sandbox_workspace_write.network_access=false", + ]) cmd = [codex_bin, "app-server"] + app_server_args # Codex emits tracing to stderr; default WARN keeps it quiet for users. spawn_env.setdefault("RUST_LOG", "warn") - # Hide the console the codex child would otherwise flash on Windows - # (#56747). Hide-only — stdio pipes stay intact for the app-server wire. + # Hide the console flash on Windows (#56747); stdio pipes stay intact. from hermes_cli._subprocess_compat import windows_hide_flags self._proc = subprocess.Popen( - cmd, - stdin=subprocess.PIPE, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - bufsize=0, - env=spawn_env, - creationflags=windows_hide_flags(), + cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, + bufsize=0, env=spawn_env, creationflags=windows_hide_flags(), ) self._next_id = 1 self._pending: dict[int, _Pending] = {} @@ -151,12 +103,10 @@ class CodexAppServerClient: self._initialized = False self._reader = threading.Thread(target=self._read_stdout, daemon=True) - self._reader.start() self._stderr_reader = threading.Thread(target=self._read_stderr, daemon=True) + self._reader.start() self._stderr_reader.start() - # ---------- lifecycle ---------- - def initialize( self, client_name: str = "hermes", @@ -165,16 +115,11 @@ class CodexAppServerClient: capabilities: Optional[dict] = None, timeout: float = 10.0, ) -> dict: - """Send `initialize` + `initialized` handshake. Returns the server's - InitializeResponse (userAgent, codexHome, platformFamily, platformOs).""" + """Send ``initialize`` + ``initialized``; return the server's InitializeResponse.""" if self._initialized: raise RuntimeError("already initialized") params = { - "clientInfo": { - "name": client_name, - "title": client_title, - "version": client_version, - }, + "clientInfo": {"name": client_name, "title": client_title, "version": client_version}, "capabilities": capabilities or {}, } result = self.request("initialize", params, timeout=timeout) @@ -208,16 +153,8 @@ class CodexAppServerClient: def __exit__(self, *exc: Any) -> None: self.close() - # ---------- send/receive ---------- - - def request( - self, - method: str, - params: Optional[dict] = None, - timeout: float = 30.0, - ) -> dict: - """Send a JSON-RPC request and block on the response. Returns `result`, - raises CodexAppServerError on `error`.""" + def request(self, method: str, params: Optional[dict] = None, timeout: float = 30.0) -> dict: + """Send a request and block for ``result``; raise CodexAppServerError on ``error``.""" rid = self._take_id() q: queue.Queue = queue.Queue(maxsize=1) with self._pending_lock: @@ -228,16 +165,10 @@ class CodexAppServerClient: except queue.Empty: with self._pending_lock: self._pending.pop(rid, None) - raise TimeoutError( - f"codex app-server method {method!r} timed out after {timeout}s" - ) + raise TimeoutError(f"codex app-server method {method!r} timed out after {timeout}s") if "error" in msg: err = msg["error"] - raise CodexAppServerError( - code=err.get("code", -1), - message=err.get("message", ""), - data=err.get("data"), - ) + raise CodexAppServerError(code=err.get("code", -1), message=err.get("message", ""), data=err.get("data")) return msg.get("result", {}) def notify(self, method: str, params: Optional[dict] = None) -> None: @@ -248,37 +179,29 @@ class CodexAppServerClient: """Reply to a server-initiated request (e.g. approval prompts).""" self._send({"id": request_id, "result": result}) - def respond_error( - self, request_id: Any, code: int, message: str, data: Optional[Any] = None - ) -> None: + def respond_error(self, request_id: Any, code: int, message: str, data: Optional[Any] = None) -> None: """Reply to a server-initiated request with an error.""" err: dict[str, Any] = {"code": code, "message": message} if data is not None: err["data"] = data self._send({"id": request_id, "error": err}) - def take_notification(self, timeout: float = 0.0) -> Optional[dict]: - """Pop the next streaming notification, or return None on timeout. - - timeout=0.0 means non-blocking. Use small positive timeouts inside the - AIAgent turn loop to interleave reads with interrupt checks.""" + @staticmethod + def _take(q: queue.Queue, timeout: float) -> Optional[dict]: try: if timeout <= 0: - return self._notifications.get_nowait() - return self._notifications.get(timeout=timeout) + return q.get_nowait() + return q.get(timeout=timeout) except queue.Empty: return None + def take_notification(self, timeout: float = 0.0) -> Optional[dict]: + """Pop the next streaming notification, or None on timeout (0 = non-blocking).""" + return self._take(self._notifications, timeout) + def take_server_request(self, timeout: float = 0.0) -> Optional[dict]: """Pop the next server-initiated request (e.g. exec/applyPatch approval).""" - try: - if timeout <= 0: - return self._server_requests.get_nowait() - return self._server_requests.get(timeout=timeout) - except queue.Empty: - return None - - # ---------- diagnostics ---------- + return self._take(self._server_requests, timeout) def stderr_tail(self, n: int = 20) -> list[str]: """Return last n lines of codex's stderr (for error reports).""" @@ -288,12 +211,7 @@ class CodexAppServerClient: def is_alive(self) -> bool: return self._proc.poll() is None - # ---------- internals ---------- - def _take_id(self) -> int: - # JSON-RPC ids only need to be unique per-connection. A simple - # monotonically increasing int is the common choice and matches what - # codex's own clients use. rid = self._next_id self._next_id += 1 return rid @@ -307,38 +225,34 @@ class CodexAppServerClient: self._proc.stdin.write((json.dumps(obj) + "\n").encode("utf-8")) self._proc.stdin.flush() except (BrokenPipeError, ValueError) as exc: - raise RuntimeError( - f"codex app-server stdin closed unexpectedly: {exc}" - ) from exc + raise RuntimeError(f"codex app-server stdin closed unexpectedly: {exc}") from exc + + def _append_stderr(self, line: str) -> None: + with self._stderr_lock: + self._stderr_lines.append(line) + if len(self._stderr_lines) > 500: # bound memory + self._stderr_lines = self._stderr_lines[-500:] def _read_stdout(self) -> None: if self._proc.stdout is None: return try: for line in iter(self._proc.stdout.readline, b""): - if not line: - break line = line.strip() if not line: continue try: msg = json.loads(line) except json.JSONDecodeError: - # Non-JSON output is unexpected on stdout; tracing belongs - # on stderr. Surface it via stderr buffer for diagnostics. - with self._stderr_lock: - self._stderr_lines.append( - f" {line[:200]!r}" - ) + # Non-JSON stdout is unexpected; surface it via the stderr buffer. + self._append_stderr(f" {line[:200]!r}") continue self._dispatch(msg) except Exception as exc: - with self._stderr_lock: - self._stderr_lines.append(f" {exc}") + self._append_stderr(f" {exc}") def _dispatch(self, msg: dict) -> None: - # Reply (has id + result/error, no method) - if "id" in msg and ("result" in msg or "error" in msg): + if "id" in msg and ("result" in msg or "error" in msg): # reply with self._pending_lock: pending = self._pending.pop(msg["id"], None) if pending is not None: @@ -346,63 +260,36 @@ class CodexAppServerClient: pending.queue.put_nowait(msg) except queue.Full: # pragma: no cover - defensive pass - return - # Server-initiated request (has id + method) - if "id" in msg and "method" in msg: - self._server_requests.put(msg) - return - # Notification (no id) - if "method" in msg: - self._notifications.put(msg) + elif "method" in msg: # server-initiated request (has id) or notification + (self._server_requests if "id" in msg else self._notifications).put(msg) def _read_stderr(self) -> None: if self._proc.stderr is None: return try: for line in iter(self._proc.stderr.readline, b""): - if not line: - break - with self._stderr_lock: - self._stderr_lines.append( - line.decode("utf-8", "replace").rstrip() - ) - # Bound memory: keep last 500 lines. - if len(self._stderr_lines) > 500: - self._stderr_lines = self._stderr_lines[-500:] + self._append_stderr(line.decode("utf-8", "replace").rstrip()) except Exception: # pragma: no cover pass def parse_codex_version(output: str) -> Optional[tuple[int, int, int]]: - """Parse `codex --version` output. Returns (major, minor, patch) or None.""" - # Output format: "codex-cli 0.130.0" possibly followed by metadata. - import re - + """Parse ``codex --version`` output ("codex-cli 0.130.0 ...") into (major, minor, patch).""" match = re.search(r"(\d+)\.(\d+)\.(\d+)", output or "") - if not match: - return None - return (int(match.group(1)), int(match.group(2)), int(match.group(3))) + return tuple(int(g) for g in match.groups()) if match else None def check_codex_binary( codex_bin: str = "codex", min_version: tuple[int, int, int] = MIN_CODEX_VERSION ) -> tuple[bool, str]: - """Verify codex CLI is installed and meets minimum version. - - Returns (ok, message). Used by setup wizard and runtime startup.""" + """Verify codex CLI is installed and meets minimum version. Returns (ok, message).""" try: proc = subprocess.run( - [codex_bin, "--version"], - capture_output=True, - text=True, encoding='utf-8', errors='replace', - timeout=10, - stdin=subprocess.DEVNULL, + [codex_bin, "--version"], capture_output=True, text=True, encoding='utf-8', errors='replace', + timeout=10, stdin=subprocess.DEVNULL, ) except FileNotFoundError: - return False, ( - f"codex CLI not found at {codex_bin!r}. Install with: " - f"npm i -g @openai/codex" - ) + return False, f"codex CLI not found at {codex_bin!r}. Install with: npm i -g @openai/codex" except subprocess.TimeoutExpired: return False, "codex --version timed out" if proc.returncode != 0: @@ -410,9 +297,7 @@ def check_codex_binary( version = parse_codex_version(proc.stdout) if version is None: return False, f"could not parse codex version from: {proc.stdout!r}" + have, need = ".".join(map(str, version)), ".".join(map(str, min_version)) if version < min_version: - return False, ( - f"codex {'.'.join(map(str, version))} is older than required " - f"{'.'.join(map(str, min_version))}. Run: npm i -g @openai/codex" - ) - return True, ".".join(map(str, version)) + return False, f"codex {have} is older than required {need}. Run: npm i -g @openai/codex" + return True, have diff --git a/agent/transports/codex_app_server_session.py b/agent/transports/codex_app_server_session.py index 384a7de8ab..17aeeb3a0e 100644 --- a/agent/transports/codex_app_server_session.py +++ b/agent/transports/codex_app_server_session.py @@ -1,25 +1,11 @@ """Session adapter for codex app-server runtime. -Owns one Codex thread per Hermes session. Drives `turn/start`, consumes -streaming notifications via CodexEventProjector, handles server-initiated -approval requests (apply_patch, exec command), translates cancellation, -and returns a clean turn result that AIAgent.run_conversation() can splice -into its `messages` list. - -Lifecycle: - session = CodexAppServerSession(cwd="/home/x/proj") - session.ensure_started() # spawns + handshake + thread/start - result = session.run_turn(user_input="hello") # blocks until turn/completed - # result.final_text → assistant text returned to caller - # result.projected_messages → list of {role, content, ...} for messages list - # result.tool_iterations → how many tool-shaped items completed (skill nudge counter) - # result.interrupted → True if Ctrl+C / interrupt_requested fired mid-turn - session.close() # tears down subprocess - -Threading model: the adapter is single-threaded from the caller's perspective. -The underlying CodexAppServerClient owns its own reader threads but exposes -blocking-with-timeout queues that this adapter polls in a loop, so the run_turn -call is synchronous and behaves like AIAgent's existing chat_completions loop. +Owns one Codex thread per Hermes session: drives ``turn/start``, consumes +streaming notifications via CodexEventProjector, bridges server-initiated +approval requests, translates cancellation, and returns a TurnResult that +AIAgent.run_conversation() splices into ``messages``. Synchronous from the +caller's view: the client's reader threads feed blocking-with-timeout queues +that this adapter polls, matching AIAgent's chat_completions loop. """ from __future__ import annotations @@ -37,27 +23,20 @@ from agent.transports.codex_app_server import ( CodexAppServerClient, CodexAppServerError, ) -from agent.transports.codex_event_projector import CodexEventProjector +from agent.transports.codex_event_projector import CodexEventProjector, ProjectionResult logger = logging.getLogger(__name__) -# How many tailing stderr lines from the codex subprocess to attach to a -# user-facing error when we don't have a more specific classification (OAuth, -# wedge watchdog, etc.). Small enough to keep error messages legible, large -# enough to surface a config/provider/auth diagnostic. +# Tail of codex stderr attached to generic user-facing errors: small enough to +# stay legible, large enough to surface a config/provider/auth diagnostic. _STDERR_TAIL_LINES = 12 - -# Permission profile mapping mirrors the docstring in PR proposal: -# Hermes' tools.terminal.security_mode → Codex's permissions profile id. -# Defaults if config is missing → workspace-write (matches Codex's own default). +# Hermes' tools.terminal.security_mode -> Codex permissions profile id. +# Missing config -> workspace-write (Codex's own default). _HERMES_TO_CODEX_PERMISSION_PROFILE = { - "auto": "workspace-write", - "approval-required": "read-only-with-approval", - "unrestricted": "full-access", - # Backstop alias used by some skills/tests. - "yolo": "full-access", + "auto": "workspace-write", "approval-required": "read-only-with-approval", + "unrestricted": "full-access", "yolo": "full-access", # yolo: backstop alias used by some skills/tests } @@ -73,190 +52,116 @@ class TurnResult: turn_id: Optional[str] = None thread_id: Optional[str] = None token_usage_last: Optional[dict[str, Any]] = None - token_usage_total: Optional[dict[str, Any]] = None model_context_window: Optional[int] = None compacted: bool = False - # Hint to the caller that the underlying codex subprocess is likely - # wedged (turn-level timeout fired, post-tool watchdog tripped, or - # token-refresh failure killed the child). The caller should retire - # the session so the next turn respawns codex from scratch instead - # of riding a CPU-spinning or auth-broken process. Mirrors openclaw - # beta.8's "retire timed-out app-server clients" fix. + # Codex subprocess is likely wedged (turn timeout, post-tool watchdog, token + # refresh failure): caller should retire the session so the next turn respawns. should_retire: bool = False -# Markers we accept as terminal even when codex never emits turn/completed. -# Some codex versions stream `` as raw text in agentMessage -# items when an interrupt or upstream error tears the turn down before the -# normal completion path fires. Mirrors openclaw beta.8 fix. +# Some codex versions stream ```` as raw agentMessage text when an +# interrupt/upstream error tears the turn down without emitting turn/completed. _TURN_ABORTED_MARKERS = ("", "") -def _notification_scope_ids( - note: dict, -) -> tuple[Optional[str], Optional[str]]: - """Extract the thread/turn identity carried by a notification.""" - if not isinstance(note, dict): - return None, None - params = note.get("params") or {} +def _notification_scope_ids(note: dict) -> tuple[Optional[str], Optional[str]]: + """Extract the thread/turn identity carried by a notification (top-level, then turn/item).""" + params = (note.get("params") or {}) if isinstance(note, dict) else None if not isinstance(params, dict): return None, None + turn, item = params.get("turn") or {}, params.get("item") or {} - nested_turn = params.get("turn") or {} - nested_item = params.get("item") or {} + def first(*lookups: tuple[Any, str, str]) -> Any: + """``src.get(a) or src.get(b)`` over successive dict sources until one is not None.""" + for src, primary, fallback in lookups: + if isinstance(src, dict): + observed = src.get(primary) or src.get(fallback) + if observed is not None: + return observed + return None - observed_thread_id = params.get("threadId") or params.get("thread_id") - if observed_thread_id is None and isinstance(nested_turn, dict): - observed_thread_id = ( - nested_turn.get("threadId") - or nested_turn.get("thread_id") - ) - if observed_thread_id is None and isinstance(nested_item, dict): - observed_thread_id = ( - nested_item.get("threadId") - or nested_item.get("thread_id") - ) - - observed_turn_id = params.get("turnId") or params.get("turn_id") - if observed_turn_id is None and isinstance(nested_turn, dict): - observed_turn_id = nested_turn.get("id") or nested_turn.get("turnId") - if observed_turn_id is None and isinstance(nested_item, dict): - observed_turn_id = ( - nested_item.get("turnId") - or nested_item.get("turn_id") - ) - - return observed_thread_id, observed_turn_id + return ( + first((params, "threadId", "thread_id"), (turn, "threadId", "thread_id"), (item, "threadId", "thread_id")), + first((params, "turnId", "turn_id"), (turn, "id", "turnId"), (item, "turnId", "turn_id")), + ) -def _notification_belongs_to_turn( - note: dict, - *, - thread_id: Optional[str], - turn_id: Optional[str], -) -> bool: - """Return whether a multiplexed notification belongs to this turn. +def _notification_belongs_to_turn(note: dict, *, thread_id: Optional[str], turn_id: Optional[str]) -> bool: + """Whether a multiplexed notification belongs to this turn. - Codex app-server can carry parent and hosted subagent threads over one - JSON-RPC connection. An explicitly foreign child or - stale-turn event must not mutate the active parent's transcript or mark - its turn complete. Unscoped notifications remain accepted for protocol - compatibility. + One JSON-RPC connection can carry parent and hosted subagent threads; an + explicitly foreign thread/turn event must not mutate this transcript. + Unscoped notifications remain accepted for protocol compatibility. """ if not isinstance(note, dict): return False - observed_thread_id, observed_turn_id = _notification_scope_ids(note) - - if ( - thread_id is not None - and observed_thread_id is not None - and str(observed_thread_id) != str(thread_id) - ): + if thread_id is not None and observed_thread_id is not None and str(observed_thread_id) != str(thread_id): return False - - if ( - turn_id is not None - and observed_turn_id is not None - and str(observed_turn_id) != str(turn_id) - ): - return False - - return True + return not (turn_id is not None and observed_turn_id is not None and str(observed_turn_id) != str(turn_id)) def _coerce_turn_input_text(user_input: Any) -> str: - """Collapse Hermes/OpenAI rich content into app-server text input. + """Collapse Hermes/OpenAI rich content parts into app-server text input. - The current `turn/start` path sends text items only. TUI image attachment - can hand us OpenAI-style content parts, so keep the text/path hints and - replace opaque image payloads with a small marker instead of putting a - Python list into the `text` field. + ``turn/start`` sends text items only; TUI image attachments arrive as content + parts, so keep text and replace image payloads with a marker. """ if isinstance(user_input, str): return user_input - if isinstance(user_input, list): - parts: list[str] = [] - for item in user_input: - if isinstance(item, str): - if item.strip(): - parts.append(item) - continue - if not isinstance(item, dict): - if item is not None: - parts.append(str(item)) - continue - item_type = item.get("type") - if item_type in {"text", "input_text"}: - text = item.get("text") or item.get("content") or "" - if text: - parts.append(str(text)) - elif item_type in {"image", "image_url", "input_image"}: - parts.append("[image attached]") - text = "\n\n".join(p for p in parts if p).strip() - return text or "What do you see in this image?" - return "" if user_input is None else str(user_input) + if not isinstance(user_input, list): + return "" if user_input is None else str(user_input) + parts: list[str] = [] + for item in user_input: + if not isinstance(item, dict): + keep = item.strip() if isinstance(item, str) else item is not None + if keep: + parts.append(str(item)) + continue + item_type = item.get("type") + if item_type in {"text", "input_text"}: + text = item.get("text") or item.get("content") or "" + if text: + parts.append(str(text)) + elif item_type in {"image", "image_url", "input_image"}: + parts.append("[image attached]") + text = "\n\n".join(p for p in parts if p).strip() + return text or "What do you see in this image?" -# Substrings in codex stderr / JSON-RPC error messages that signal the -# subprocess died because its OAuth credentials are no longer valid. -# Kept conservative: we only redirect users to `codex login` when we're -# reasonably sure that's the actual failure, otherwise we surface the -# original error verbatim. Mirrors openclaw beta.8's auth-refresh -# classification. +# Substrings in codex stderr / JSON-RPC errors signalling expired OAuth creds. +# Conservative on purpose: only redirect to `codex login` on a strong signal, +# otherwise the original error surfaces verbatim. _OAUTH_REFRESH_FAILURE_HINTS = ( - "invalid_grant", - "invalid grant", - "refresh token", - "refresh_token", - "token refresh", - "token_refresh", - "token has expired", - "expired_token", - "expired token", - "not authenticated", - "unauthenticated", - "unauthorized", - "401 unauthorized", - "re-authenticate", - "reauthenticate", - "please log in", - "please login", - "auth profile", - "no auth profile", - "oauth", + "invalid_grant", "invalid grant", + "refresh token", "refresh_token", "token refresh", "token_refresh", + "token has expired", "expired_token", "expired token", + "not authenticated", "unauthenticated", "unauthorized", "401 unauthorized", + "re-authenticate", "reauthenticate", "please log in", "please login", + "auth profile", "no auth profile", "oauth", +) + +_OAUTH_REAUTH_HINT = ( + "Codex authentication failed — your ChatGPT/Codex login looks expired or invalid. Run `codex login` to refresh, " + "then retry. (Fall back to default runtime with `/codex-runtime auto` if the issue persists.)" ) def _classify_oauth_failure(*parts: str) -> Optional[str]: - """Return a user-friendly re-auth hint if any of the provided strings - look like a codex OAuth/token-refresh failure; otherwise None. - - Used for both `turn/start` JSON-RPC errors and post-mortem stderr - inspection when the subprocess exits unexpectedly. Conservative on - purpose — we only redirect users to `codex login` when the signal - is strong, so unrelated runtime failures still surface verbatim. - """ + """Re-auth hint if any part looks like a codex OAuth/token-refresh failure, else None.""" haystack = " ".join(p for p in parts if p).lower() - if not haystack: - return None - for needle in _OAUTH_REFRESH_FAILURE_HINTS: - if needle in haystack: - return ( - "Codex authentication failed — your ChatGPT/Codex login " - "looks expired or invalid. Run `codex login` to refresh, " - "then retry. (Fall back to default runtime with " - "`/codex-runtime auto` if the issue persists.)" - ) + if haystack and any(needle in haystack for needle in _OAUTH_REFRESH_FAILURE_HINTS): + return _OAUTH_REAUTH_HINT return None @dataclass class _ServerRequestRouting: - """Default policies for codex-side approval requests when no interactive - callback is wired in. These are only used by tests + cron / non-interactive - contexts; the live CLI path passes an approval_callback that defers to - tools.approval.prompt_dangerous_approval().""" + """Default approval policies when no interactive callback is wired in. + + Used by tests + cron / non-interactive contexts; the live CLI passes an + approval_callback that defers to tools.approval.prompt_dangerous_approval(). + """ auto_approve_exec: bool = False auto_approve_apply_patch: bool = False @@ -265,10 +170,8 @@ class _ServerRequestRouting: class CodexAppServerSession: """One Codex thread per Hermes session, lifetime owned by AIAgent. - Not thread-safe — one caller drives it at a time, matching how AIAgent's - run_conversation() loop is structured today. The codex client itself can - handle interleaved reads/writes via its own threads, but the adapter's - state (projector, thread_id, turn counter) is owned by the caller thread. + Not thread-safe: one caller drives it at a time (projector, thread_id and + turn state are owned by the caller thread), like AIAgent.run_conversation(). """ def __init__( @@ -286,11 +189,8 @@ class CodexAppServerSession: self._cwd = cwd or os.getcwd() self._codex_bin = codex_bin self._codex_home = codex_home - self._permission_profile = ( - permission_profile or _HERMES_TO_CODEX_PERMISSION_PROFILE.get( - os.environ.get("HERMES_TERMINAL_SECURITY_MODE", "auto"), - "workspace-write", - ) + self._permission_profile = permission_profile or _HERMES_TO_CODEX_PERMISSION_PROFILE.get( + os.environ.get("HERMES_TERMINAL_SECURITY_MODE", "auto"), "workspace-write" ) self._approval_callback = approval_callback self._on_event = on_event # Display hook (kawaii spinner ticks etc.) @@ -302,75 +202,38 @@ class CodexAppServerSession: self._interrupt_event = threading.Event() self._active_turn_id: Optional[str] = None self._active_turn_lock = threading.Lock() - # Pending file-change items, keyed by item id. Populated on - # item/started for fileChange items; consumed by the approval - # bridge when codex sends item/fileChange/requestApproval. The - # approval params don't carry the changeset, so we cache here - # to surface a real summary in the approval prompt (quirk #4). + # In-progress fileChange items by id (item/started -> item/completed): + # approval params don't carry the changeset, so this feeds the prompt summary. self._pending_file_changes: dict[str, str] = {} self._closed = False - # ---------- lifecycle ---------- - def ensure_started(self) -> str: - """Spawn the subprocess, do the initialize handshake, and start a - thread. Returns the codex thread id. Idempotent — repeated calls - return the same thread id.""" + """Spawn, handshake, and ``thread/start``; idempotent, returns the codex thread id.""" if self._thread_id is not None: return self._thread_id if self._client is None: - self._client = self._client_factory( - codex_bin=self._codex_bin, codex_home=self._codex_home - ) - self._client.initialize( - client_name="hermes", - client_title="Hermes Agent", - client_version=_get_hermes_version(), - ) - # Permission selection is intentionally NOT sent on thread/start. - # Two reasons (live-tested against codex 0.130.0): - # 1. `thread/start.permissions` is gated behind the experimentalApi - # capability on this codex version — we'd have to opt in during - # initialize and accept the unstable surface. - # 2. Even with experimentalApi declared and the correct shape - # (`{"type": "profile", "id": "..."}`, not `{"profileId": ...}`), - # codex requires a matching `[permissions]` table in - # ~/.codex/config.toml or it fails the request with - # 'default_permissions requires a [permissions] table'. - # Letting codex pick its default (`:read-only` unless the user has - # configured otherwise in their codex config.toml) is the standard - # codex CLI workflow and avoids fighting codex's own validation. - # Users who want a write-capable profile configure it in their - # ~/.codex/config.toml the same way they would for any codex usage. - params: dict[str, Any] = {"cwd": self._cwd} - result = self._client.request("thread/start", params, timeout=15) - # Cross-fill thread.id/sessionId — different codex versions have - # serialized this under either key. Mirrors openclaw beta.8's - # tolerance fix so future codex drops/renames don't KeyError us - # at handshake time. + self._client = self._client_factory(codex_bin=self._codex_bin, codex_home=self._codex_home) + self._client.initialize(client_name="hermes", client_title="Hermes Agent", client_version=_get_hermes_version()) + # Permissions are intentionally NOT sent on thread/start: on codex 0.130 + # ``thread/start.permissions`` is gated behind experimentalApi and also + # requires a matching ``[permissions]`` table in ~/.codex/config.toml. + # Users configure a write-capable profile there, as for any codex use. + result = self._client.request("thread/start", {"cwd": self._cwd}, timeout=15) + # Different codex versions serialize the id under thread.id / sessionId / threadId. thread_obj = result.get("thread") or {} thread_id = ( - thread_obj.get("id") - or thread_obj.get("sessionId") - or result.get("sessionId") - or result.get("threadId") + thread_obj.get("id") or thread_obj.get("sessionId") or result.get("sessionId") or result.get("threadId") ) if not thread_id: raise CodexAppServerError( code=-32603, - message=( - "codex thread/start returned no thread id " - f"(payload keys: {sorted(result.keys())})" - ), + message=f"codex thread/start returned no thread id (payload keys: {sorted(result.keys())})", ) self._thread_id = thread_id logger.info( - "codex app-server thread started: id=%s profile=%s cwd=%s", - self._thread_id[:8], - self._permission_profile, - self._cwd, + "codex app-server thread started: id=%s profile=%s cwd=%s", thread_id[:8], self._permission_profile, self._cwd ) - return self._thread_id + return thread_id def close(self) -> None: if self._closed: @@ -383,7 +246,7 @@ class CodexAppServerSession: self._client.close() except Exception: # pragma: no cover - best-effort cleanup pass - self._client = None + self._client = None self._thread_id = None def __enter__(self) -> "CodexAppServerSession": @@ -392,11 +255,8 @@ class CodexAppServerSession: def __exit__(self, *exc: Any) -> None: self.close() - # ---------- interrupt ---------- - def request_interrupt(self) -> None: - """Idempotent: signal the active turn loop to issue turn/interrupt - and unwind. Called by AIAgent's _interrupt_requested path.""" + """Idempotent: signal the active turn loop to issue turn/interrupt and unwind.""" self._interrupt_event.set() def request_steer(self, text: str) -> bool: @@ -413,11 +273,7 @@ class CodexAppServerSession: try: response = client.request( "turn/steer", - { - "threadId": thread_id, - "input": [{"type": "text", "text": cleaned}], - "expectedTurnId": turn_id, - }, + {"threadId": thread_id, "input": [{"type": "text", "text": cleaned}], "expectedTurnId": turn_id}, timeout=10, ) except (CodexAppServerError, TimeoutError): @@ -426,46 +282,107 @@ class CodexAppServerSession: accepted_turn_id = response.get("turnId") if isinstance(response, dict) else None return accepted_turn_id in {None, turn_id} - # ---------- diagnostics ---------- + def _format_error_with_stderr(self, prefix: str, exc: Any = "", *, tail_lines: int = _STDERR_TAIL_LINES) -> str: + """User-facing error string for generic codex failures. - def _format_error_with_stderr( - self, - prefix: str, - exc: Any = "", - *, - tail_lines: int = _STDERR_TAIL_LINES, - ) -> str: - """Build a user-facing error string for codex failures. - - Appends the last few lines of codex's stderr buffer when available, - passed through agent.redact with force=True so secrets in provider - error responses (auth headers, query-string tokens, sk-* keys) never - leak into chat output or trajectories. The codex CLI's own error - text ('Internal error', 'turn/start failed: ...') is otherwise - opaque and forces users to re-run with verbose flags to diagnose - config / provider / auth-bridge problems. - - Use this for the generic / catch-all branches. Specific - classifications (OAuth via _classify_oauth_failure, post-tool wedge - watchdog) already produce a clean hint and should be used instead. + Appends the redacted (force=True) stderr tail so provider/auth secrets + never leak into chat output, while codex's opaque 'Internal error' text + becomes diagnosable. Specific classifications (OAuth, wedge watchdog) + produce their own clean hint instead. """ exc_str = str(exc) if exc != "" and exc is not None else "" base = f"{prefix}: {exc_str}" if exc_str else prefix - if self._client is None: - return base try: - tail = self._client.stderr_tail(tail_lines) + tail = self._client.stderr_tail(tail_lines) if self._client is not None else [] except Exception: # pragma: no cover - diagnostic best-effort return base - if not tail: - return base joined = "\n".join(line.rstrip() for line in tail if line) if not joined.strip(): return base redacted = redact_sensitive_text(joined, force=True) return f"{base}\ncodex stderr (last {len(tail)} lines):\n{redacted}" - # ---------- per-turn ---------- + def _stderr_blob(self, n: int) -> str: + return "\n".join(self._client.stderr_tail(n)) + + @staticmethod + def _retire(result: TurnResult, error: str) -> None: + """Record a terminal error and flag the session for respawn on the next turn.""" + result.error = error + result.should_retire = True + + def _set_classified_error(self, result: TurnResult, prefix: str, classify_text: str, detail: Any) -> None: + """OAuth failures become the re-auth hint AND retire the session (token store + is broken even though JSON-RPC is fine); everything else gets the stderr tail.""" + hint = _classify_oauth_failure(classify_text, self._stderr_blob(40)) + if hint is not None: + self._retire(result, hint) + else: + result.error = self._format_error_with_stderr(prefix, detail) + + def _start_for(self, result: TurnResult) -> bool: + """ensure_started() into ``result``; startup failures surface as TurnResult.error + (retiring the session) instead of raw codex exceptions reaching AIAgent.""" + try: + self.ensure_started() + except (CodexAppServerError, TimeoutError) as exc: + self._retire(result, self._format_error_with_stderr("codex app-server startup failed", exc)) + return False + assert self._client is not None and self._thread_id is not None + result.thread_id = self._thread_id + return True + + def _request_for(self, result: TurnResult, method: str, params: dict, label: str) -> Optional[dict]: + """Issue ``method``; on failure fill ``result.error`` and return None. + A timeout is a strong wedge signal and always retires the session.""" + try: + return self._client.request(method, params, timeout=10) + except CodexAppServerError as exc: + self._set_classified_error(result, f"{label} failed", exc.message, exc) + except TimeoutError as exc: + hint = _classify_oauth_failure(self._stderr_blob(40)) + self._retire(result, hint or self._format_error_with_stderr(f"{label} timed out", exc)) + return None + + def _subprocess_died(self, result: TurnResult) -> bool: + """Bail out early (rather than waiting on the deadline) when codex exited.""" + if self._client.is_alive(): + return False + hint = _classify_oauth_failure(self._stderr_blob(60)) + self._retire( + result, + hint or self._format_error_with_stderr("codex app-server subprocess exited unexpectedly", tail_lines=20), + ) + return True + + def _absorb_notification( + self, result: TurnResult, projector: CodexEventProjector, note: dict + ) -> tuple[ProjectionResult, bool]: + """Fan one in-scope notification out to display, accounting, file-change + tracking and the projector. Returns (projection, aborted) where aborted + means the agent text carried a ```` marker (terminal).""" + if self._on_event is not None: + try: + self._on_event(note) + except Exception: # pragma: no cover - display callback + logger.debug("on_event callback raised", exc_info=True) + _apply_token_usage_notification(result, note) + _apply_compaction_notification(result, note) + self._track_pending_file_change(note) + projection = projector.project(note) + if projection.messages: + result.projected_messages.extend(projection.messages) + if projection.is_tool_iteration: + result.tool_iterations += 1 + aborted = False + if projection.final_text is not None: + # Multiple agentMessage items per turn: the last one is canonical. + result.final_text = projection.final_text + if _has_turn_aborted_marker(projection.final_text): + aborted = True + result.interrupted = True + result.error = result.error or "codex reported turn_aborted" + return projection, aborted def run_turn( self, @@ -475,290 +392,174 @@ class CodexAppServerSession: notification_poll_timeout: float = 0.25, post_tool_quiet_timeout: float = 90.0, ) -> TurnResult: - """Send a user message and block until turn/completed, while - forwarding server-initiated approval requests and projecting items - into Hermes' messages shape. + """Send a user message and block until turn/completed, bridging approvals + and projecting items into Hermes' messages shape. - post_tool_quiet_timeout: if codex emits a tool completion and then - goes quiet for this many seconds without emitting another item or - `turn/completed`, fast-fail and mark the session for retirement. - Mirrors openclaw beta.8's post-tool completion watchdog (#81697) - so a wedged codex doesn't burn the full turn deadline. + post_tool_quiet_timeout: if codex completes a tool and then stays silent + this long, fast-fail and retire instead of burning the full turn deadline. """ - # Pre-create the result so startup failures (codex subprocess can't - # spawn, initialize handshake rejects, thread/start blows up) surface - # the same way per-turn failures do — with a TurnResult.error string - # the caller can render — instead of bubbling raw codex exceptions - # up to AIAgent.run_conversation. result = TurnResult() - try: - self.ensure_started() - except (CodexAppServerError, TimeoutError) as exc: - result.error = self._format_error_with_stderr( - "codex app-server startup failed", exc - ) - # Subprocess almost certainly unhealthy — retire so the next - # turn re-spawns cleanly. - result.should_retire = True + if not self._start_for(result): self._interrupt_event.clear() return result - assert self._client is not None and self._thread_id is not None - result.thread_id = self._thread_id - - # Do not clear here: a hard stop can arrive while ensure_started() is - # spawning/initializing the subprocess. Honor it before launching a - # Codex turn instead of erasing the signal. + # Do not clear first: a hard stop arriving during ensure_started() must + # be honored before launching a Codex turn. if self._interrupt_event.is_set(): result.interrupted = True self._interrupt_event.clear() return result projector = CodexEventProjector() - user_input_text = _coerce_turn_input_text(user_input) - - # Send turn/start with the user input. Text-only for now (codex - # supports rich content but Hermes' text path is the common case). - try: - ts = self._client.request( - "turn/start", - { - "threadId": self._thread_id, - "input": [{"type": "text", "text": user_input_text}], - }, - timeout=10, - ) - except CodexAppServerError as exc: - # Classify auth/refresh failures so the user gets a clear - # `codex login` pointer instead of a raw RPC error string. - stderr_blob = "\n".join(self._client.stderr_tail(40)) - hint = _classify_oauth_failure(exc.message, stderr_blob) - if hint is not None: - result.error = hint - # Subprocess is fine on a JSON-RPC level here, but the - # token store is broken — retire so the next turn does a - # clean handshake (and the user has a chance to re-auth - # via `codex login` between turns). - result.should_retire = True - else: - result.error = self._format_error_with_stderr( - "turn/start failed", exc - ) - self._interrupt_event.clear() - return result - except TimeoutError as exc: - # turn/start hanging is a strong signal the subprocess is wedged. - stderr_blob = "\n".join(self._client.stderr_tail(40)) - hint = _classify_oauth_failure(stderr_blob) - result.error = hint or self._format_error_with_stderr( - "turn/start timed out", exc - ) - result.should_retire = True + ts = self._request_for( + result, + "turn/start", + {"threadId": self._thread_id, "input": [{"type": "text", "text": _coerce_turn_input_text(user_input)}]}, + "turn/start", + ) + if ts is None: self._interrupt_event.clear() return result result.turn_id = (ts.get("turn") or {}).get("id") with self._active_turn_lock: self._active_turn_id = result.turn_id - deadline = time.monotonic() + turn_timeout - turn_complete = False - # Post-tool watchdog state. last_tool_completion_at is set whenever - # a tool-shaped item completes; if no further notification arrives - # within post_tool_quiet_timeout and the turn hasn't completed, we - # fast-fail and retire the session. + # Post-tool watchdog: armed on each tool completion, cleared by any other activity. last_tool_completion_at: Optional[float] = None + def watchdog_tripped() -> bool: + if ( + last_tool_completion_at is not None + and (time.monotonic() - last_tool_completion_at) > post_tool_quiet_timeout + ): + self._issue_interrupt(result.turn_id) + result.interrupted = True + self._retire( + result, + f"codex went silent for {post_tool_quiet_timeout:.0f}s after a tool result; " + f"retiring app-server session.", + ) + return True + return False + + def on_server_request(sreq: dict) -> bool: + nonlocal last_tool_completion_at + # Drain pending notifications first (bounded) so per-turn state such + # as _pending_file_changes is current for the approval decision, and + # so display events around approvals still reach on_event. + turn_complete = False + for _ in range(8): + pending = self._client.take_notification(timeout=0) + if pending is None: + break + if not _notification_belongs_to_turn( + pending, thread_id=self._thread_id, turn_id=result.turn_id + ): + logger.debug( + "ignoring foreign codex notification while draining " + "server request: method=%s", + pending.get("method"), + ) + continue + proj, aborted = self._absorb_notification(result, projector, pending) + if proj.is_tool_iteration: + last_tool_completion_at = time.monotonic() + if aborted: + turn_complete = True + self._handle_server_request(sreq) + # An approval round-trip is live signal — don't let it trip the watchdog. + last_tool_completion_at = None + return turn_complete + + def on_note(note: dict, method: str) -> bool: + nonlocal last_tool_completion_at + projection, aborted = self._absorb_notification(result, projector, note) + if projection.is_tool_iteration: + last_tool_completion_at = time.monotonic() + elif projection.messages or projection.final_text is not None: + last_tool_completion_at = None + if method != "turn/completed": + return aborted + turn_obj = (note.get("params") or {}).get("turn") or {} + turn_status = turn_obj.get("status") + if turn_status and turn_status not in {"completed", "interrupted"}: + err_obj = turn_obj.get("error") + if err_obj: + err_msg = _format_responses_error(err_obj, str(turn_status)) + self._set_classified_error( + result, f"turn ended status={turn_status}", err_msg, err_msg + ) + return True + + self._drive_turn( + result, + turn_timeout=turn_timeout, + notification_poll_timeout=notification_poll_timeout, + timeout_label="turn", + before_poll=watchdog_tripped, + on_server_request=on_server_request, + on_note=on_note, + accept_final_text_at_deadline=True, + ) + with self._active_turn_lock: + self._active_turn_id = None + self._interrupt_event.clear() + return result + + def _drive_turn( + self, + result: TurnResult, + *, + turn_timeout: float, + notification_poll_timeout: float, + timeout_label: str, + on_server_request: Callable[[dict], bool], + on_note: Callable[[dict, str], bool], + before_poll: Optional[Callable[[], bool]] = None, + pre_scope_filter: Optional[Callable[[dict, str], bool]] = None, + accept_final_text_at_deadline: bool = False, + ) -> None: + """Shared poll loop for run_turn / compact_thread until turn/completed or deadline. + + Per iteration: interrupt -> subprocess death -> ``before_poll`` (watchdog) -> + server requests (answered before notifications so codex isn't blocked) -> + one notification, filtered by ``pre_scope_filter`` then by turn scope, handed + to ``on_note``. Hooks return True to mark the turn complete / stop the loop. + A deadline without completion interrupts and retires the session — a turn + that never finished is a strong sign the next turn shouldn't inherit this codex. + """ + deadline = time.monotonic() + turn_timeout + turn_complete = False while time.monotonic() < deadline and not turn_complete: if self._interrupt_event.is_set(): self._issue_interrupt(result.turn_id) result.interrupted = True break - - # Detect a dead subprocess between iterations. If codex exited - # (e.g. crashed, segfaulted, or its auth refresh thread killed - # the process), we won't get any more notifications — bail out - # rather than waiting for the full turn deadline. - if not self._client.is_alive(): - stderr_blob = "\n".join(self._client.stderr_tail(60)) - hint = _classify_oauth_failure(stderr_blob) - if hint is not None: - result.error = hint - else: - result.error = self._format_error_with_stderr( - "codex app-server subprocess exited unexpectedly", - tail_lines=20, - ) - result.should_retire = True + if self._subprocess_died(result): break - - # Post-tool watchdog: if a tool completion was the most recent - # signal and codex has been silent past the quiet timeout, give - # up on this turn instead of waiting for the outer deadline. - if ( - last_tool_completion_at is not None - and (time.monotonic() - last_tool_completion_at) - > post_tool_quiet_timeout - ): - self._issue_interrupt(result.turn_id) - result.interrupted = True - result.error = ( - f"codex went silent for " - f"{post_tool_quiet_timeout:.0f}s after a tool result; " - f"retiring app-server session." - ) - result.should_retire = True + if before_poll is not None and before_poll(): break - - # Drain any server-initiated requests (approvals) before - # reading notifications, so the codex side isn't blocked. sreq = self._client.take_server_request(timeout=0) if sreq is not None: - # Drain any pending notifications first so per-turn state - # (e.g. _pending_file_changes for fileChange approvals) is - # up to date when we make the approval decision. Bounded - # to avoid starving the server-request response. - for _ in range(8): - pending = self._client.take_notification(timeout=0) - if pending is None: - break - if not _notification_belongs_to_turn( - pending, - thread_id=self._thread_id, - turn_id=result.turn_id, - ): - logger.debug( - "ignoring foreign codex notification while draining " - "server request: method=%s", - pending.get("method"), - ) - continue - # Mirror the main notification-handling block below so - # display events surface and stay in step with projector - # state. Without this, item/started / item/completed - # events drained as part of the approval-roundtrip - # preamble are projected into messages but never reach - # the tool-progress display, silently hiding tool - # bubbles around approvals. - if self._on_event is not None: - try: - self._on_event(pending) - except Exception: # pragma: no cover - display callback - logger.debug( - "on_event callback raised", exc_info=True - ) - _apply_token_usage_notification(result, pending) - _apply_compaction_notification(result, pending) - self._track_pending_file_change(pending) - proj = projector.project(pending) - if proj.messages: - result.projected_messages.extend(proj.messages) - if proj.is_tool_iteration: - result.tool_iterations += 1 - last_tool_completion_at = time.monotonic() - if proj.final_text is not None: - result.final_text = proj.final_text - if _has_turn_aborted_marker(proj.final_text): - turn_complete = True - result.interrupted = True - result.error = ( - result.error - or "codex reported turn_aborted" - ) - self._handle_server_request(sreq) - # Activity counts as live signal — reset the post-tool - # quiet timer so an approval round-trip doesn't trip it. - last_tool_completion_at = None + if on_server_request(sreq): + turn_complete = True continue - - note = self._client.take_notification( - timeout=notification_poll_timeout - ) + note = self._client.take_notification(timeout=notification_poll_timeout) if note is None: continue - method = note.get("method", "") - if not _notification_belongs_to_turn( - note, - thread_id=self._thread_id, - turn_id=result.turn_id, - ): - logger.debug( - "ignoring foreign codex notification: method=%s", method - ) + if pre_scope_filter is not None and not pre_scope_filter(note, method): continue - - if self._on_event is not None: - try: - self._on_event(note) - except Exception: # pragma: no cover - display callback - logger.debug("on_event callback raised", exc_info=True) - - _apply_token_usage_notification(result, note) - _apply_compaction_notification(result, note) - - # Track in-progress fileChange items so the approval bridge - # can surface a real change summary when codex requests - # approval (the approval params themselves don't carry the - # changeset). Quirk #4 fix. - self._track_pending_file_change(note) - - # Project into messages - projection = projector.project(note) - if projection.messages: - result.projected_messages.extend(projection.messages) - if projection.is_tool_iteration: - result.tool_iterations += 1 - # Arm/refresh the post-tool quiet watchdog whenever a - # tool-shaped item completes. - last_tool_completion_at = time.monotonic() - else: - # Any non-tool projected activity (assistant message, - # status update, etc.) means codex is still producing - # output — clear the quiet timer so we don't fast-fail. - if projection.messages or projection.final_text is not None: - last_tool_completion_at = None - if projection.final_text is not None: - # Codex can emit multiple agentMessage items in one turn - # (e.g. partial then final). Take the last one as canonical. - result.final_text = projection.final_text - # Some codex builds tear a turn down by emitting a - # `` marker in the agent message text and - # never sending turn/completed. Treat the marker itself - # as terminal so we don't burn the full deadline. - if _has_turn_aborted_marker(projection.final_text): - turn_complete = True - result.interrupted = True - result.error = ( - result.error or "codex reported turn_aborted" - ) - - if method == "turn/completed": + if not _notification_belongs_to_turn( + note, thread_id=self._thread_id, turn_id=result.turn_id + ): + logger.debug("ignoring foreign codex notification: method=%s", method) + continue + if on_note(note, method): turn_complete = True - turn_status = ( - (note.get("params") or {}).get("turn") or {} - ).get("status") - if turn_status and turn_status not in {"completed", "interrupted"}: - err_obj = ( - (note.get("params") or {}).get("turn") or {} - ).get("error") - if err_obj: - err_msg = _format_responses_error(err_obj, str(turn_status)) - # If the turn failed for an auth/refresh reason, - # rewrite the error into a re-auth hint AND mark - # the session for retirement. - stderr_blob = "\n".join( - self._client.stderr_tail(40) - ) - hint = _classify_oauth_failure(err_msg, stderr_blob) - if hint is not None: - result.error = hint - result.should_retire = True - else: - result.error = self._format_error_with_stderr( - f"turn ended status={turn_status}", err_msg - ) if ( - not turn_complete + accept_final_text_at_deadline + and not turn_complete and not result.interrupted and result.final_text and result.error is None @@ -771,23 +572,12 @@ class CodexAppServerSession: turn_complete = True if not turn_complete and not result.interrupted: - # Hit the deadline. Issue interrupt to stop wasted compute, and - # tell the caller to retire the session — a turn that never - # finished is a strong sign codex is wedged in a way the next - # turn shouldn't inherit. self._issue_interrupt(result.turn_id) result.interrupted = True if not result.error: - result.error = self._format_error_with_stderr( - f"turn timed out after {turn_timeout}s" - ) + result.error = self._format_error_with_stderr(f"{timeout_label} timed out after {turn_timeout}s") result.should_retire = True - with self._active_turn_lock: - self._active_turn_id = None - self._interrupt_event.clear() - return result - def compact_thread( self, *, @@ -796,199 +586,79 @@ class CodexAppServerSession: ) -> TurnResult: """Trigger Codex-native history compaction for the current thread. - `thread/compact/start` returns immediately; the actual compaction - progress streams through the same turn/item notifications as a normal - turn. We wait for the matching `turn/completed` so callers can treat a - successful return as a completed compaction boundary. + ``thread/compact/start`` returns immediately (with no turn id); progress + streams through normal turn/item notifications, so we wait for the + matching ``turn/completed`` to treat the return as a compaction boundary. """ result = TurnResult() - try: - self.ensure_started() - except (CodexAppServerError, TimeoutError) as exc: - result.error = self._format_error_with_stderr( - "codex app-server startup failed", exc - ) - result.should_retire = True + if not self._start_for(result): return result - - assert self._client is not None and self._thread_id is not None - result.thread_id = self._thread_id self._interrupt_event.clear() projector = CodexEventProjector() - try: - self._client.request( - "thread/compact/start", - {"threadId": self._thread_id}, - timeout=10, - ) - except CodexAppServerError as exc: - stderr_blob = "\n".join(self._client.stderr_tail(40)) - hint = _classify_oauth_failure(exc.message, stderr_blob) - if hint is not None: - result.error = hint - result.should_retire = True - else: - result.error = self._format_error_with_stderr( - "thread/compact/start failed", exc - ) - return result - except TimeoutError as exc: - stderr_blob = "\n".join(self._client.stderr_tail(40)) - hint = _classify_oauth_failure(stderr_blob) - result.error = hint or self._format_error_with_stderr( - "thread/compact/start timed out", exc - ) - result.should_retire = True + if self._request_for( + result, "thread/compact/start", {"threadId": self._thread_id}, "thread/compact/start" + ) is None: return result - deadline = time.monotonic() + turn_timeout - turn_complete = False - - while time.monotonic() < deadline and not turn_complete: - if self._interrupt_event.is_set(): - self._issue_interrupt(result.turn_id) - result.interrupted = True - break - - if not self._client.is_alive(): - stderr_blob = "\n".join(self._client.stderr_tail(60)) - hint = _classify_oauth_failure(stderr_blob) - if hint is not None: - result.error = hint - else: - result.error = self._format_error_with_stderr( - "codex app-server subprocess exited unexpectedly", - tail_lines=20, - ) - result.should_retire = True - break - - sreq = self._client.take_server_request(timeout=0) - if sreq is not None: - self._handle_server_request(sreq) - continue - - note = self._client.take_notification( - timeout=notification_poll_timeout - ) - if note is None: - continue - - method = note.get("method", "") + def pre_scope_filter(note: dict, method: str) -> bool: + if result.turn_id is not None: + return True observed_thread_id, observed_turn_id = _notification_scope_ids(note) - if result.turn_id is None: - if method == "turn/started": - if ( - observed_thread_id is not None - and str(observed_thread_id) != str(self._thread_id) - ): - logger.debug( - "ignoring foreign compact turn/started: thread=%s", - observed_thread_id, - ) - continue - if observed_turn_id is None: - logger.debug( - "ignoring compact turn/started without a turn id" - ) - continue - result.turn_id = str(observed_turn_id) - elif observed_turn_id is not None or method in { - "item/completed", - "turn/completed", - }: - # thread/compact/start does not return a turn id. Until the - # new turn/started arrives, any terminal/projectable event - # is stale or cannot be safely attributed to this compaction. - logger.debug( - "ignoring codex notification before compact turn start: " - "method=%s", - method, - ) - continue - - if not _notification_belongs_to_turn( - note, - thread_id=self._thread_id, - turn_id=result.turn_id, - ): - logger.debug( - "ignoring foreign codex notification: method=%s", method - ) - continue - - if self._on_event is not None: - try: - self._on_event(note) - except Exception: # pragma: no cover - display callback - logger.debug("on_event callback raised", exc_info=True) - - _apply_token_usage_notification(result, note) - _apply_compaction_notification(result, note) - self._track_pending_file_change(note) - - projection = projector.project(note) - if projection.messages: - result.projected_messages.extend(projection.messages) - if projection.is_tool_iteration: - result.tool_iterations += 1 - if projection.final_text is not None: - result.final_text = projection.final_text - if _has_turn_aborted_marker(projection.final_text): - turn_complete = True - result.interrupted = True - result.error = ( - result.error or "codex reported turn_aborted" - ) - if method == "turn/started": - turn_obj = (note.get("params") or {}).get("turn") or {} - result.turn_id = turn_obj.get("id") or result.turn_id - elif method == "turn/completed": - turn_complete = True - turn_obj = (note.get("params") or {}).get("turn") or {} - result.turn_id = turn_obj.get("id") or result.turn_id - turn_status = turn_obj.get("status") - if turn_status == "interrupted": - result.interrupted = True - result.error = result.error or "compact turn interrupted" - elif turn_status and turn_status != "completed": - err_obj = turn_obj.get("error") - err_msg = _format_responses_error(err_obj, str(turn_status)) - stderr_blob = "\n".join(self._client.stderr_tail(40)) - hint = _classify_oauth_failure(err_msg, stderr_blob) - if hint is not None: - result.error = hint - result.should_retire = True - else: - result.error = self._format_error_with_stderr( - f"compact turn ended status={turn_status}", - err_msg, - ) + if observed_thread_id is not None and str(observed_thread_id) != str(self._thread_id): + logger.debug("ignoring foreign compact turn/started: thread=%s", observed_thread_id) + return False + if observed_turn_id is None: + logger.debug("ignoring compact turn/started without a turn id") + return False + result.turn_id = str(observed_turn_id) + elif observed_turn_id is not None or method in {"item/completed", "turn/completed"}: + # Until the new turn/started arrives, terminal/projectable + # events are stale or can't be attributed to this compaction. + logger.debug("ignoring codex notification before compact turn start: method=%s", method) + return False + return True - if not turn_complete and not result.interrupted: - self._issue_interrupt(result.turn_id) - result.interrupted = True - if not result.error: - result.error = self._format_error_with_stderr( - f"compact turn timed out after {turn_timeout}s" + def on_note(note: dict, method: str) -> bool: + _, aborted = self._absorb_notification(result, projector, note) + if method not in {"turn/started", "turn/completed"}: + return aborted + turn_obj = (note.get("params") or {}).get("turn") or {} + result.turn_id = turn_obj.get("id") or result.turn_id + if method == "turn/started": + return aborted + turn_status = turn_obj.get("status") + if turn_status == "interrupted": + result.interrupted = True + result.error = result.error or "compact turn interrupted" + elif turn_status and turn_status != "completed": + err_msg = _format_responses_error(turn_obj.get("error"), str(turn_status)) + self._set_classified_error( + result, f"compact turn ended status={turn_status}", err_msg, err_msg ) - result.should_retire = True + return True + def on_server_request(sreq: dict) -> bool: + self._handle_server_request(sreq) + return False + + self._drive_turn( + result, + turn_timeout=turn_timeout, + notification_poll_timeout=notification_poll_timeout, + timeout_label="compact turn", + on_server_request=on_server_request, + on_note=on_note, + pre_scope_filter=pre_scope_filter, + ) return result - # ---------- internals ---------- - def _issue_interrupt(self, turn_id: Optional[str]) -> None: if self._client is None or self._thread_id is None or turn_id is None: return try: - self._client.request( - "turn/interrupt", - {"threadId": self._thread_id, "turnId": turn_id}, - timeout=5, - ) + self._client.request("turn/interrupt", {"threadId": self._thread_id, "turnId": turn_id}, timeout=5) except CodexAppServerError as exc: # "no active turn to interrupt" is fine — already done. logger.debug("turn/interrupt non-fatal: %s", exc) @@ -996,290 +666,163 @@ class CodexAppServerSession: logger.warning("turn/interrupt timed out") def _handle_server_request(self, req: dict) -> None: - """Translate a codex server request (approval) into Hermes' approval - flow, then send the response. + """Answer a codex server request (approval / elicitation) via Hermes' approval flow. - Method names verified live against codex 0.130.0 (Apr 2026): - item/commandExecution/requestApproval — exec approvals - item/fileChange/requestApproval — apply_patch approvals - item/permissions/requestApproval — permissions changes - (we decline; user controls - permission profile in - ~/.codex/config.toml). + Method names verified live against codex 0.130.0. Permission escalations + are always declined: the user chose their profile in ~/.codex/config.toml. + Unknown methods get a clean JSON-RPC error so codex doesn't hang. """ if self._client is None: return method = req.get("method", "") rid = req.get("id") params = req.get("params") or {} - - if method == "item/commandExecution/requestApproval": - decision = self._decide_exec_approval(params) - self._client.respond(rid, {"decision": decision}) - elif method == "item/fileChange/requestApproval": - decision = self._decide_apply_patch_approval(params) - self._client.respond(rid, {"decision": decision}) - elif method == "item/permissions/requestApproval": - # Codex sometimes asks to escalate permissions mid-turn. We - # always decline — the user already chose their permission - # profile in ~/.codex/config.toml and surprise escalations - # shouldn't be silently accepted. - self._client.respond(rid, {"decision": "decline"}) - elif method == "mcpServer/elicitation/request": - # Codex's MCP layer asks the user for structured input on - # behalf of an MCP server (e.g. tool-call confirmation, - # OAuth, form data). For our own hermes-tools callback we - # auto-accept — the user already approved Hermes' tools - # by enabling the runtime, and we never expose anything - # codex's built-in shell can't already do. For other MCP - # servers we decline so the user explicitly opts in via - # codex's own auth flow. - server_name = params.get("serverName") or "" - if server_name == "hermes-tools": - self._client.respond( - rid, - {"action": "accept", "content": None, "_meta": None}, - ) - else: - self._client.respond( - rid, - {"action": "decline", "content": None, "_meta": None}, - ) - else: - # Unknown server request — codex can extend this surface. Reject - # cleanly so codex doesn't hang waiting for us. + handler = self._SERVER_REQUEST_HANDLERS.get(method) + if handler is None: logger.warning("Unknown codex server request: %s", method) - self._client.respond_error( - rid, code=-32601, message=f"Unsupported method: {method}" - ) + self._client.respond_error(rid, code=-32601, message=f"Unsupported method: {method}") + return + self._client.respond(rid, handler(self, params)) + + def _respond_elicitation(self, params: dict) -> dict: + """MCP elicitation (tool confirmation, OAuth, form data): auto-accept for our + own hermes-tools server (the user opted in by enabling the runtime; it exposes + nothing codex's shell can't already do); decline others so the user opts in + via codex's own auth flow.""" + action = "accept" if (params.get("serverName") or "") == "hermes-tools" else "decline" + return {"action": action, "content": None, "_meta": None} + + _SERVER_REQUEST_HANDLERS: dict[str, Callable[..., dict]] = { + "item/commandExecution/requestApproval": lambda self, p: {"decision": self._decide_exec_approval(p)}, + "item/fileChange/requestApproval": lambda self, p: {"decision": self._decide_apply_patch_approval(p)}, + "item/permissions/requestApproval": lambda self, p: {"decision": "decline"}, + "mcpServer/elicitation/request": _respond_elicitation, + } + + def _run_approval_callback(self, command: str, description: str, log_label: str) -> str: + try: + choice = self._approval_callback(command, description, allow_permanent=False) + return _approval_choice_to_codex_decision(choice) + except Exception: + logger.exception("approval_callback raised on %s", log_label) + return "decline" def _decide_exec_approval(self, params: dict) -> str: - """Decide a Codex exec approval request. + """Decide a Codex exec approval request — protocol-level routing only. - This is protocol-level routing only — it carries NO Hermes - approval-mode/timeout logic. The Hermes-side resolution happens - upstream: ``agent/codex_runtime.py`` derives - ``auto_approve_exec`` from the canonical - ``tools.approval.is_approval_bypass_active()`` (which reads - ``approvals.mode`` via ``tools.approval._get_approval_mode``), - and ``self._approval_callback`` itself runs the shared approval - gate (mode + ``approvals.timeout``) in ``tools/approval.py``. - Keep it that way — do not re-read approval config here. + Hermes approval mode/timeout resolution lives upstream: codex_runtime.py + derives ``auto_approve_exec`` from tools.approval, and the callback runs + the shared approval gate. Do not re-read approval config here. """ if self._routing.auto_approve_exec: return "accept" - command = params.get("command") or "" - # Codex's CommandExecutionRequestApprovalParams has cwd as Optional — - # fall back to the session's cwd when codex doesn't include it so the - # approval prompt is never empty (quirk #10 fix). - cwd = params.get("cwd") or self._cwd or "" - reason = params.get("reason") - description = f"Codex requests exec in {cwd}" - if reason: - description += f" — {reason}" - if self._approval_callback is not None: - try: - choice = self._approval_callback( - command, description, allow_permanent=False - ) - return _approval_choice_to_codex_decision(choice) - except Exception: - logger.exception("approval_callback raised on exec request") - return "decline" - return "decline" # fail-closed when no callback wired + if self._approval_callback is None: + return "decline" # fail-closed when no callback wired + # ``cwd`` is Optional on codex's side; fall back so the prompt is never empty. + description = f"Codex requests exec in {params.get('cwd') or self._cwd or ''}" + if params.get("reason"): + description += f" — {params['reason']}" + return self._run_approval_callback(params.get("command") or "", description, "exec request") def _decide_apply_patch_approval(self, params: dict) -> str: - """Decide a Codex apply_patch approval request. - - Protocol-level routing only; Hermes approval-mode/timeout - resolution is delegated to ``tools/approval.py`` upstream — see - the docstring on ``_decide_exec_approval``. - """ + """Decide a Codex apply_patch approval request (routing only; see _decide_exec_approval).""" if self._routing.auto_approve_apply_patch: return "accept" - if self._approval_callback is not None: - # FileChangeRequestApprovalParams gives us reason + grantRoot. - # The actual changeset lives on the corresponding fileChange - # item which the projector has already cached for us — look it - # up by item_id so the user sees what's actually changing. - reason = params.get("reason") - grant_root = params.get("grantRoot") - item_id = params.get("itemId") or "" - change_summary = self._lookup_pending_file_change(item_id) - description_parts = [] - if reason: - description_parts.append(reason) - if change_summary: - description_parts.append(change_summary) - if grant_root: - description_parts.append(f"grants write to {grant_root}") - description = ( - "; ".join(description_parts) - if description_parts - else "Codex requests to apply a patch" - ) - command_label = ( - f"apply_patch: {change_summary}" if change_summary - else f"apply_patch: {reason}" if reason - else "apply_patch" - ) - try: - choice = self._approval_callback( - command_label, - description, - allow_permanent=False, - ) - return _approval_choice_to_codex_decision(choice) - except Exception: - logger.exception("approval_callback raised on apply_patch") - return "decline" - return "decline" + if self._approval_callback is None: + return "decline" + # Params carry reason + grantRoot only; the changeset comes from the + # fileChange item cached by _track_pending_file_change. + reason = params.get("reason") + grant_root = params.get("grantRoot") + change_summary = self._pending_file_changes.get(params.get("itemId") or "") or None + description_parts = [p for p in (reason, change_summary) if p] + if grant_root: + description_parts.append(f"grants write to {grant_root}") + description = "; ".join(description_parts) if description_parts else "Codex requests to apply a patch" + command_label = ( + f"apply_patch: {change_summary}" if change_summary + else f"apply_patch: {reason}" if reason + else "apply_patch" + ) + return self._run_approval_callback(command_label, description, "apply_patch") def _track_pending_file_change(self, note: dict) -> None: - """Maintain self._pending_file_changes from item/started + item/completed - notifications. Lets the apply_patch approval prompt show what's - actually changing — codex's approval params don't carry the data.""" + """Maintain _pending_file_changes from item/started + item/completed so the + apply_patch approval prompt can show what's actually changing.""" method = note.get("method", "") - params = note.get("params") or {} - item = params.get("item") or {} - if item.get("type") != "fileChange": - return + item = (note.get("params") or {}).get("item") or {} item_id = item.get("id") or "" - if not item_id: + if item.get("type") != "fileChange" or not item_id: return - if method == "item/started": - changes = item.get("changes") or [] - if not changes: + if method == "item/completed": + self._pending_file_changes.pop(item_id, None) + elif method == "item/started": + raw_changes = item.get("changes") or [] + if not raw_changes: self._pending_file_changes[item_id] = "1 change pending" return + changes = [ch for ch in raw_changes if isinstance(ch, dict)] kinds: dict[str, int] = {} - paths: list[str] = [] for ch in changes: - if not isinstance(ch, dict): - continue kind = (ch.get("kind") or {}).get("type") or "update" kinds[kind] = kinds.get(kind, 0) + 1 - p = ch.get("path") or "" - if p: - paths.append(p) + paths: list[str] = [ch["path"] for ch in changes if ch.get("path")] counts = ", ".join(f"{n} {k}" for k, n in sorted(kinds.items())) preview = ", ".join(paths[:3]) if len(paths) > 3: preview += f", +{len(paths) - 3} more" - self._pending_file_changes[item_id] = ( - f"{counts}: {preview}" if preview else counts - ) - elif method == "item/completed": - self._pending_file_changes.pop(item_id, None) - - def _lookup_pending_file_change(self, item_id: str) -> Optional[str]: - """Look up an in-progress fileChange item by id and summarize its - changes for the approval prompt. Returns None when we don't have - the item cached (e.g. approval arrived before item/started, or - fileChange item content not tracked yet).""" - if not item_id: - return None - cached = self._pending_file_changes.get(item_id) - if not cached: - return None - return cached + self._pending_file_changes[item_id] = f"{counts}: {preview}" if preview else counts def _apply_token_usage_notification(result: TurnResult, note: dict) -> None: - """Capture Codex app-server token usage updates for caller accounting. - - Codex does not put token usage on turn/completed. It emits a separate - thread/tokenUsage/updated notification containing cumulative totals and - the latest turn breakdown. - """ + """Capture token usage: codex emits it as a separate thread/tokenUsage/updated + notification (cumulative totals + last-turn breakdown), not on turn/completed.""" if not isinstance(note, dict) or note.get("method") != "thread/tokenUsage/updated": return - params = note.get("params") or {} - token_usage = params.get("tokenUsage") or {} + token_usage = (note.get("params") or {}).get("tokenUsage") or {} if not isinstance(token_usage, dict): return last = token_usage.get("last") - total = token_usage.get("total") if isinstance(last, dict): result.token_usage_last = dict(last) - if isinstance(total, dict): - result.token_usage_total = dict(total) window = token_usage.get("modelContextWindow") if isinstance(window, int) and window > 0: result.model_context_window = window def _apply_compaction_notification(result: TurnResult, note: dict) -> None: - """Capture Codex-native context compaction boundaries. - - Recent app-server builds expose compaction as a ContextCompaction item. - Older builds also emit the deprecated thread/compacted notification. Both - mean the underlying Codex thread history has been compacted. - """ + """Capture Codex-native compaction boundaries: a contextCompaction item + (recent builds) or the deprecated thread/compacted notification (older builds).""" if not isinstance(note, dict): return method = note.get("method") or "" params = note.get("params") or {} if not isinstance(params, dict): return - - if method == "thread/compacted": - result.compacted = True - result.thread_id = params.get("threadId") or result.thread_id - result.turn_id = params.get("turnId") or result.turn_id - return - - if method not in {"item/started", "item/completed"}: - return - - item = params.get("item") or {} - if not isinstance(item, dict) or item.get("type") != "contextCompaction": - return - + if method != "thread/compacted": + item = params.get("item") if method in {"item/started", "item/completed"} else None + if not isinstance(item, dict) or item.get("type") != "contextCompaction": + return result.compacted = True result.thread_id = params.get("threadId") or result.thread_id result.turn_id = params.get("turnId") or result.turn_id +# Hermes 'once'/'session'/'always'/'deny'(/'timeout') -> codex approval decisions +# (codex-rs app-server-protocol v2). Only the wire translation lives here; the +# approval mode/timeout resolution stays in tools/approval.py. "deny" and +# "timeout" both decline — codex has no "prompt expired" wire value. +_APPROVAL_CHOICE_TO_DECISION = {"once": "accept", "session": "acceptForSession", "always": "acceptForSession"} + + def _approval_choice_to_codex_decision(choice: str) -> str: - """Map Hermes approval choices onto codex's CommandExecutionApprovalDecision - / FileChangeApprovalDecision wire values. - - Hermes returns 'once', 'session', 'always', or 'deny'. - Codex expects 'accept', 'acceptForSession', 'decline', or 'cancel' - (verified against codex-rs/app-server-protocol/src/protocol/v2/item.rs - on codex 0.130.0). - - This mapping is Codex-protocol-semantic and intentionally lives here, - NOT in tools/approval.py: the Hermes approval mode/timeout resolution - and the choice itself come from the shared core (tools/approval.py); - only the wire-value translation is local. - """ - if choice in {"once",}: - return "accept" - if choice in {"session", "always"}: - return "acceptForSession" - # "deny" and "timeout" both map to decline — codex has no wire value for - # "prompt expired"; the Hermes-side messaging already distinguishes them. - return "decline" + """Map a Hermes approval choice onto codex's approval decision wire value.""" + return _APPROVAL_CHOICE_TO_DECISION.get(choice, "decline") def _has_turn_aborted_marker(text: str) -> bool: - """Return True if `text` contains any of the raw markers codex uses - to signal a turn was aborted without emitting `turn/completed`. - - Codex emits `` (and sometimes ``) as raw - text inside agentMessage items when an interrupt or upstream error - tears the turn down before the normal completion path fires. Mirrors - openclaw beta.8's terminal-marker fix so we don't burn the full turn - deadline waiting for a turn/completed that never comes. - """ - if not text: - return False - for marker in _TURN_ABORTED_MARKERS: - if marker in text: - return True - return False + """True if ``text`` carries a raw ```` marker (terminal without turn/completed).""" + return bool(text) and any(marker in text for marker in _TURN_ABORTED_MARKERS) def _get_hermes_version() -> str: diff --git a/agent/transports/codex_event_projector.py b/agent/transports/codex_event_projector.py index f375529a01..1bf331e77d 100644 --- a/agent/transports/codex_event_projector.py +++ b/agent/transports/codex_event_projector.py @@ -1,29 +1,14 @@ """Projects codex app-server events into Hermes' messages list. -The translator that lets Hermes' memory/skill review keep working under the -Codex runtime: it converts Codex `item/*` notifications into the standard -OpenAI-shaped `{role, content, tool_calls, tool_call_id}` entries that -`agent/curator.py` already knows how to read. - -Codex emits items with a discriminator field `type`: - - userMessage → {role: "user", content} - - agentMessage → {role: "assistant", content} - - reasoning → stashed in the assistant's "reasoning" field - - commandExecution → assistant tool_call(name="exec") + tool result - - fileChange → assistant tool_call(name="apply_patch") + tool result - - mcpToolCall → assistant tool_call(name=f"mcp.{server}.{tool}") + tool result - - dynamicToolCall → assistant tool_call(name=tool) + tool result - - plan/hookPrompt/collabAgentToolCall → recorded as opaque assistant notes - -Each item maps to AT MOST one assistant entry + one tool entry, preserving -Hermes' message-alternation invariants (system → user → assistant → user/tool -→ assistant → ...). Multiple Codex tool calls within one Codex turn produce -multiple consecutive (assistant, tool) pairs, which is the same shape Hermes -already produces for parallel tool calls. - -Counters tracked alongside projection: - - tool_iterations: ticks once per completed tool-shaped item. Used by - AIAgent._iters_since_skill (skill nudge gate, default threshold 10). +Converts Codex ``item/*`` notifications into OpenAI-shaped +``{role, content, tool_calls, tool_call_id}`` entries that memory/skill review +(agent/curator.py) already reads: + userMessage → user; agentMessage → assistant; reasoning → stashed onto the next + assistant entry; commandExecution / fileChange / mcpToolCall / dynamicToolCall → + assistant tool_call + tool result; anything else → opaque assistant note. +Each item yields AT MOST one assistant + one tool entry, preserving Hermes' +message-alternation invariants. ``is_tool_iteration`` ticks once per completed +tool-shaped item (AIAgent._iters_since_skill skill-nudge gate). """ from __future__ import annotations @@ -31,20 +16,13 @@ from __future__ import annotations import hashlib import json from dataclasses import dataclass, field -from typing import Any, Optional +from typing import Any, Callable, Optional def _deterministic_call_id(item_type: str, item_id: str) -> str: - """Stable id for tool_call message correlation. - - Uses the codex item id directly when present (already a uuid); falls back - to a content hash so replay produces the same id across sessions and - prefix caches stay valid. See AGENTS.md Pitfall #16 (deterministic IDs in - tool call history).""" - if item_id: - return f"codex_{item_type}_{item_id}" - digest = hashlib.sha256(f"{item_type}".encode()).hexdigest()[:16] - return f"codex_{item_type}_{digest}" + """Stable tool_call id: the codex item id when present, else a content hash so + replay yields the same id across sessions and prefix caches stay valid.""" + return f"codex_{item_type}_{item_id or hashlib.sha256(f'{item_type}'.encode()).hexdigest()[:16]}" def _format_tool_args(d: dict) -> str: @@ -52,14 +30,18 @@ def _format_tool_args(d: dict) -> str: return json.dumps(d, ensure_ascii=False, sort_keys=True) +def _dict_args(raw: Any) -> dict: + args = raw or {} + return args if isinstance(args, dict) else {"arguments": args} + + @dataclass class ProjectionResult: """Output of projecting one Codex item. - `messages` is a list because some Codex items produce two messages - (assistant tool_call + tool result). Empty list = item ignored (e.g. a - streaming `outputDelta` that doesn't materialize into messages until the - `item/completed` event).""" + ``messages`` may hold two entries (assistant tool_call + tool result); empty + means the item was ignored (e.g. a streaming delta before ``item/completed``). + """ messages: list[dict] = field(default_factory=list) is_tool_iteration: bool = False @@ -69,65 +51,56 @@ class ProjectionResult: class CodexEventProjector: """Stateful projector consuming Codex notifications in arrival order. - Owns the in-progress reasoning content (codex emits reasoning as separate - items but Hermes stashes it on the next assistant message).""" + Owns in-progress reasoning: codex emits it as separate items, Hermes stashes + it on the next assistant message. + """ def __init__(self) -> None: self._pending_reasoning: list[str] = [] def project(self, notification: dict) -> ProjectionResult: - """Project a single notification. Idempotent for non-completion events; - only `item/completed` and `turn/completed` materialize messages.""" + """Project one notification; only ``item/completed`` materializes messages. + + Streaming deltas are display-only, mirroring how Hermes writes the + assistant message only after the streaming completion event. + """ method = notification.get("method", "") params = notification.get("params", {}) or {} - - # We only materialize messages on `item/completed`. Streaming deltas - # (`item//outputDelta`, `item//delta`) are display-only and - # don't enter the messages list — same way Hermes already only writes - # the assistant message after the streaming completion event. if method != "item/completed": return ProjectionResult() - item = params.get("item") or {} item_type = item.get("type") or "" item_id = item.get("id") or "" - if item_type == "agentMessage": return self._project_agent_message(item) if item_type == "reasoning": self._pending_reasoning.extend(item.get("summary") or []) self._pending_reasoning.extend(item.get("content") or []) return ProjectionResult() - if item_type == "commandExecution": - return self._project_command(item, item_id) - if item_type == "fileChange": - return self._project_file_change(item, item_id) - if item_type == "mcpToolCall": - return self._project_mcp_tool_call(item, item_id) - if item_type == "dynamicToolCall": - return self._project_dynamic_tool_call(item, item_id) if item_type == "userMessage": return self._project_user_message(item) - - # Unknown / rare items (plan, hookPrompt, collabAgentToolCall, etc.) - # — record as opaque assistant note so memory review can still see - # *something* happened, but don't fabricate tool_call structure. + tool_projection = self._TOOL_PROJECTIONS.get(item_type) + if tool_projection is not None: + return self._project_tool_item(item, item_id, tool_projection) + # Unknown / rare items (plan, hookPrompt, collabAgentToolCall, ...): keep an + # opaque note so memory review sees *something* happened, without + # fabricating tool_call structure. return self._project_opaque(item, item_type) - # ---------- per-type projections ---------- - - def _project_agent_message(self, item: dict) -> ProjectionResult: - text = item.get("text") or "" - msg: dict[str, Any] = {"role": "assistant", "content": text} + def _assistant_message(self, content: Optional[str], **extra: Any) -> dict[str, Any]: + msg: dict[str, Any] = {"role": "assistant", "content": content, **extra} if self._pending_reasoning: msg["reasoning"] = "\n".join(self._pending_reasoning) self._pending_reasoning = [] - return ProjectionResult(messages=[msg], final_text=text) + return msg + + def _project_agent_message(self, item: dict) -> ProjectionResult: + text = item.get("text") or "" + return ProjectionResult(messages=[self._assistant_message(text)], final_text=text) def _project_user_message(self, item: dict) -> ProjectionResult: - # codex's userMessage content is a list of UserInput variants. For - # projection purposes we flatten any text fragments and ignore - # non-text parts (images, etc.) — Hermes' messages store text only. + # userMessage content is a list of UserInput variants; flatten text + # fragments and drop non-text parts (Hermes' messages store text only). text_parts: list[str] = [] for fragment in item.get("content") or []: if isinstance(fragment, dict): @@ -135,111 +108,54 @@ class CodexEventProjector: text_parts.append(fragment.get("text") or "") elif "text" in fragment: text_parts.append(str(fragment["text"])) - return ProjectionResult( - messages=[{"role": "user", "content": "\n".join(text_parts)}] - ) + return ProjectionResult(messages=[{"role": "user", "content": "\n".join(text_parts)}]) - def _project_command(self, item: dict, item_id: str) -> ProjectionResult: - call_id = _deterministic_call_id("exec", item_id) - args = { - "command": item.get("command") or "", - "cwd": item.get("cwd") or "", - } - assistant_msg = { - "role": "assistant", - "content": None, - "tool_calls": [ - { - "id": call_id, - "type": "function", - "function": { - "name": "exec_command", - "arguments": _format_tool_args(args), - }, - } - ], - } - if self._pending_reasoning: - assistant_msg["reasoning"] = "\n".join(self._pending_reasoning) - self._pending_reasoning = [] + def _project_tool_item( + self, + item: dict, + item_id: str, + spec: Callable[[dict], tuple[str, str, dict, str]], + ) -> ProjectionResult: + """Emit the (assistant tool_call, tool result) pair for a tool-shaped item. + + ``spec(item)`` returns ``(call_id_type, tool_name, args, tool_content)``. + """ + id_type, name, args, content = spec(item) + call_id = _deterministic_call_id(id_type, item_id) + assistant_msg = self._assistant_message( + None, + tool_calls=[{"id": call_id, "type": "function", "function": {"name": name, "arguments": _format_tool_args(args)}}], + ) + tool_msg = {"role": "tool", "tool_call_id": call_id, "content": content} + return ProjectionResult(messages=[assistant_msg, tool_msg], is_tool_iteration=True) + + @staticmethod + def _command_spec(item: dict) -> tuple[str, str, dict, str]: + args = {"command": item.get("command") or "", "cwd": item.get("cwd") or ""} output = item.get("aggregatedOutput") or "" exit_code = item.get("exitCode") if exit_code is not None and exit_code != 0: output = f"[exit {exit_code}]\n{output}" - tool_msg = { - "role": "tool", - "tool_call_id": call_id, - "content": output, - } - return ProjectionResult( - messages=[assistant_msg, tool_msg], is_tool_iteration=True - ) + return "exec", "exec_command", args, output - def _project_file_change(self, item: dict, item_id: str) -> ProjectionResult: - call_id = _deterministic_call_id("apply_patch", item_id) - # Reduce the codex changes array to a digest the agent loop will - # find readable. We record per-file change kinds (Add/Update/Delete) - # without inlining full file contents — those can be huge. - changes_summary = [] - for change in item.get("changes") or []: - kind = (change.get("kind") or {}).get("type") or "update" - path = change.get("path") or "" - changes_summary.append({"kind": kind, "path": path}) - args = {"changes": changes_summary} - assistant_msg = { - "role": "assistant", - "content": None, - "tool_calls": [ - { - "id": call_id, - "type": "function", - "function": { - "name": "apply_patch", - "arguments": _format_tool_args(args), - }, - } - ], - } - if self._pending_reasoning: - assistant_msg["reasoning"] = "\n".join(self._pending_reasoning) - self._pending_reasoning = [] + @staticmethod + def _file_change_spec(item: dict) -> tuple[str, str, dict, str]: + # Per-file change kinds only — full file contents can be huge. + changes_summary = [ + { + "kind": (change.get("kind") or {}).get("type") or "update", + "path": change.get("path") or "", + } + for change in item.get("changes") or [] + ] status = item.get("status") or "unknown" - n = len(changes_summary) - tool_msg = { - "role": "tool", - "tool_call_id": call_id, - "content": f"apply_patch status={status}, {n} change(s)", - } - return ProjectionResult( - messages=[assistant_msg, tool_msg], is_tool_iteration=True - ) + content = f"apply_patch status={status}, {len(changes_summary)} change(s)" + return "apply_patch", "apply_patch", {"changes": changes_summary}, content - def _project_mcp_tool_call(self, item: dict, item_id: str) -> ProjectionResult: + @staticmethod + def _mcp_tool_call_spec(item: dict) -> tuple[str, str, dict, str]: server = item.get("server") or "mcp" tool = item.get("tool") or "unknown" - # Mirror the native MCP tool-name convention (mcp__server__tool) so the - # deterministic call_id input stays consistent with registration names. - call_id = _deterministic_call_id(f"mcp__{server}__{tool}", item_id) - args = item.get("arguments") or {} - if not isinstance(args, dict): - args = {"arguments": args} - assistant_msg = { - "role": "assistant", - "content": None, - "tool_calls": [ - { - "id": call_id, - "type": "function", - "function": { - "name": f"mcp.{server}.{tool}", - "arguments": _format_tool_args(args), - }, - } - ], - } - if self._pending_reasoning: - assistant_msg["reasoning"] = "\n".join(self._pending_reasoning) - self._pending_reasoning = [] result = item.get("result") error = item.get("error") if error: @@ -248,67 +164,30 @@ class CodexEventProjector: content = json.dumps(result, ensure_ascii=False)[:4000] else: content = "" - tool_msg = { - "role": "tool", - "tool_call_id": call_id, - "content": content, - } - return ProjectionResult( - messages=[assistant_msg, tool_msg], is_tool_iteration=True - ) + # Mirror the native MCP name convention (mcp__server__tool) in the call id + # so it stays consistent with registration names. + return f"mcp__{server}__{tool}", f"mcp.{server}.{tool}", _dict_args(item.get("arguments")), content - def _project_dynamic_tool_call( - self, item: dict, item_id: str - ) -> ProjectionResult: + @staticmethod + def _dynamic_tool_call_spec(item: dict) -> tuple[str, str, dict, str]: tool = item.get("tool") or "unknown" - call_id = _deterministic_call_id(f"dyn_{tool}", item_id) - args = item.get("arguments") or {} - if not isinstance(args, dict): - args = {"arguments": args} - assistant_msg = { - "role": "assistant", - "content": None, - "tool_calls": [ - { - "id": call_id, - "type": "function", - "function": { - "name": tool, - "arguments": _format_tool_args(args), - }, - } - ], - } - if self._pending_reasoning: - assistant_msg["reasoning"] = "\n".join(self._pending_reasoning) - self._pending_reasoning = [] content_items = item.get("contentItems") or [] if isinstance(content_items, list) and content_items: content = json.dumps(content_items, ensure_ascii=False)[:4000] else: - success = item.get("success") - content = f"success={success}" - tool_msg = { - "role": "tool", - "tool_call_id": call_id, - "content": content, - } - return ProjectionResult( - messages=[assistant_msg, tool_msg], is_tool_iteration=True - ) + content = f"success={item.get('success')}" + return f"dyn_{tool}", tool, _dict_args(item.get("arguments")), content + + _TOOL_PROJECTIONS: dict[str, Callable[[dict], tuple[str, str, dict, str]]] = { + "commandExecution": _command_spec, + "fileChange": _file_change_spec, + "mcpToolCall": _mcp_tool_call_spec, + "dynamicToolCall": _dynamic_tool_call_spec, + } def _project_opaque(self, item: dict, item_type: str) -> ProjectionResult: - # Record the existence of the item without inventing tool_calls. - # Memory review will see this and may or may not save anything. try: payload = json.dumps(item, ensure_ascii=False)[:1500] except (TypeError, ValueError): payload = repr(item)[:1500] - return ProjectionResult( - messages=[ - { - "role": "assistant", - "content": f"[codex {item_type}] {payload}", - } - ] - ) + return ProjectionResult(messages=[{"role": "assistant", "content": f"[codex {item_type}] {payload}"}]) diff --git a/agent/transports/hermes_tools_mcp_server.py b/agent/transports/hermes_tools_mcp_server.py index e3fcec6258..a3801325e4 100644 --- a/agent/transports/hermes_tools_mcp_server.py +++ b/agent/transports/hermes_tools_mcp_server.py @@ -1,45 +1,10 @@ """Hermes-tools-as-MCP server for the codex_app_server runtime. -When the user runs `openai/*` turns through the codex app-server, codex -owns the loop and builds its own tool list. By default, that means -Hermes' richer tool surface — web search, browser automation, -delegate_task subagents, vision analysis, persistent memory, skills, -cross-session search, image generation, TTS — is unreachable. - -This module exposes a curated subset of those Hermes tools to the -spawned codex subprocess via stdio MCP. Codex registers it as a normal -MCP server (per `~/.codex/config.toml [mcp_servers.hermes-tools]`) and -the user gets full Hermes capability inside a Codex turn. - -Scope (what we expose): - - web_search, web_extract — Firecrawl, no codex equivalent - - browser_navigate / _click / _type / — Camofox/Browserbase automation - _snapshot / _scroll / _back / _press / - _get_images / _console / _vision - - vision_analyze — image inspection by vision model - - image_generate — image generation - - skill_view, skills_list — Hermes' skill library - - text_to_speech — TTS - - kanban_* (complete/block/comment/ — kanban worker + orchestrator - heartbeat/show/list/create/ handoff (stateless: read env var, - unblock/link) write ~/.hermes/kanban.db) - -What we DO NOT expose: - - terminal / shell — codex's own shell tool - - read_file / write_file / patch — codex's apply_patch + shell - - search_files / process — codex's shell - - clarify — codex's own UX - - delegate_task / memory / — `_AGENT_LOOP_TOOLS` in Hermes - session_search / todo (model_tools.py). They require - the running AIAgent context to - dispatch (mid-loop state), so a - stateless MCP callback can't - drive them. See the inline - comment on EXPOSED_TOOLS below. - -Run with: python -m agent.transports.hermes_tools_mcp_server -Spawned by: CodexAppServerSession.ensure_started() when the runtime is - active and config opts in. +Under the codex app-server, codex owns the loop and its own tool list, so +Hermes' richer surface (web search, browser, vision, image gen, skills, TTS, +kanban handoff) would be unreachable. This module exposes a curated subset over +stdio MCP; codex registers it via ``~/.codex/config.toml [mcp_servers.hermes-tools]``. +Run with: ``python -m agent.transports.hermes_tools_mcp_server``. """ from __future__ import annotations @@ -54,61 +19,31 @@ from typing import Any, Optional logger = logging.getLogger(__name__) # JSON Schema type -> Python type mapping for signature generation -_JSON_TO_PY = { - "string": str, - "integer": int, - "number": float, - "boolean": bool, - "array": list, - "object": dict, -} +_JSON_TO_PY = {"string": str, "integer": int, "number": float, "boolean": bool, "array": list, "object": dict} def _signature_from_schema(schema: dict | None) -> tuple[inspect.Signature, dict[str, type]]: - """Build a Python function signature and annotations from a JSON schema. - - Args: - schema: JSON Schema dict with "properties" and "required" keys. - - Returns: - (signature, annotations_dict) where signature has KEYWORD_ONLY params - and annotations maps param names to Python types. - """ + """Build a KEYWORD_ONLY signature + annotations dict from a JSON schema's + ``properties`` / ``required`` (optional params default to None).""" props = (schema or {}).get("properties") or {} required = set((schema or {}).get("required") or []) params, annots = [], {} - for pname, pspec in props.items(): if pname.startswith("_"): continue py = _JSON_TO_PY.get((pspec or {}).get("type"), Any) - ann, default = ( - (py, inspect.Parameter.empty) - if pname in required - else (Optional[py], None) - ) + ann, default = (py, inspect.Parameter.empty) if pname in required else (Optional[py], None) annots[pname] = ann - params.append( - inspect.Parameter( - pname, inspect.Parameter.KEYWORD_ONLY, annotation=ann, default=default - ) - ) - + params.append(inspect.Parameter(pname, inspect.Parameter.KEYWORD_ONLY, annotation=ann, default=default)) return inspect.Signature(params, return_annotation=str), annots -# Tools we expose. Each name MUST match a registered Hermes tool that -# `model_tools.handle_function_call()` can dispatch. -# -# What we deliberately DO NOT expose: -# - terminal / shell / read_file / write_file / patch / search_files / -# process — codex's built-ins cover these and approval routes through -# codex's own UI. -# - delegate_task / memory / session_search / todo — these are -# `_AGENT_LOOP_TOOLS` in Hermes (model_tools.py:493). They require -# the running AIAgent context to dispatch (mid-loop state), so a -# stateless MCP callback can't drive them. Hermes' default runtime -# keeps these working; the codex_app_server runtime cannot. +# Each name MUST match a registered Hermes tool that +# ``model_tools.handle_function_call()`` can dispatch. +# Deliberately NOT exposed: terminal/shell, read_file/write_file/patch, +# search_files/process, clarify — codex's built-ins cover them with codex's own +# approval UI; delegate_task/memory/session_search/todo — ``_AGENT_LOOP_TOOLS`` +# need the running AIAgent context, which a stateless MCP callback lacks. EXPOSED_TOOLS: tuple[str, ...] = ( "web_search", "web_extract", @@ -127,12 +62,8 @@ EXPOSED_TOOLS: tuple[str, ...] = ( "skill_view", "skills_list", "text_to_speech", - # Kanban worker handoff tools — gated on HERMES_KANBAN_TASK env var - # (set by the kanban dispatcher when spawning a worker). Without these - # in the callback, a worker spawned with openai_runtime=codex_app_server - # could do the work but couldn't report completion back to the kernel, - # making it hang until timeout. Stateless dispatch — they just read - # the env var and write to ~/.hermes/kanban.db. + # Kanban handoff tools: stateless (read HERMES_KANBAN_TASK, write kanban.db). + # Without them a codex-runtime worker can't report completion and hangs. "kanban_complete", "kanban_block", "kanban_request_review", @@ -141,10 +72,7 @@ EXPOSED_TOOLS: tuple[str, ...] = ( "kanban_heartbeat", "kanban_show", "kanban_list", - # NOTE: kanban_create / kanban_unblock / kanban_link are orchestrator- - # only — the kanban tool gates them on HERMES_KANBAN_TASK being unset. - # They're exposed here for orchestrator agents running on the codex - # runtime that need to dispatch new tasks. + # Orchestrator-only (the kanban tool gates them on HERMES_KANBAN_TASK unset). "kanban_create", "kanban_unblock", "kanban_link", @@ -152,23 +80,15 @@ EXPOSED_TOOLS: tuple[str, ...] = ( def _build_server() -> Any: - """Create the MCP server with Hermes tools attached. Lazy imports - so the module can be imported without the mcp package installed - (we degrade to a clear error only when actually run).""" + """Create the MCP server with Hermes tools attached (lazy imports so the module + imports without the mcp package; the clear error fires only when run).""" try: - # mcp 2.0 removed `mcp.server.fastmcp`; `mcp.server.MCPServer` is the - # same decorator/add_tool surface under the new name. + # mcp 2.0 renamed `mcp.server.fastmcp` to `mcp.server.MCPServer` (same surface). from mcp.server import MCPServer except ImportError as exc: # pragma: no cover - install hint - raise ImportError( - f"hermes-tools MCP server requires the 'mcp' package: {exc}" - ) from exc + raise ImportError(f"hermes-tools MCP server requires the 'mcp' package: {exc}") from exc - # Discover Hermes tools so dispatch works. - from model_tools import ( - get_tool_definitions, - handle_function_call, - ) + from model_tools import get_tool_definitions, handle_function_call mcp = MCPServer( "hermes-tools", @@ -181,72 +101,49 @@ def _build_server() -> Any: ), ) - # Pull authoritative Hermes tool schemas for the ones we expose, so - # MCP clients see the same parameter docs Hermes gives the model. + # Authoritative Hermes schemas so MCP clients see the same parameter docs the model does. all_defs = { td["function"]["name"]: td["function"] for td in (get_tool_definitions(quiet_mode=True) or []) if isinstance(td, dict) and td.get("type") == "function" } - exposed_count = 0 + def _make_handler(tool_name: str, schema: dict | None, description: str): + # The SDK derives the input schema from the callable's signature (no + # inputSchema parameter), so synthesize __signature__ from the Hermes JSON Schema. + sig, annots = _signature_from_schema(schema) + def _dispatch(**kwargs: Any) -> str: + try: + # Drop None so unset optionals aren't forwarded to the handler. + return handle_function_call(tool_name, {k: v for k, v in kwargs.items() if v is not None}) + except Exception as exc: + logger.exception("tool %s raised", tool_name) + return json.dumps({"error": str(exc), "tool": tool_name}) + + _dispatch.__name__ = tool_name + _dispatch.__doc__ = description + _dispatch.__signature__ = sig + _dispatch.__annotations__ = {**annots, "return": str} + return _dispatch + + exposed_count = 0 for name in EXPOSED_TOOLS: spec = all_defs.get(name) if spec is None: - logger.debug( - "skipping %s — not registered in this Hermes process", name - ) + logger.debug("skipping %s — not registered in this Hermes process", name) continue - description = spec.get("description") or f"Hermes {name} tool" params_schema = spec.get("parameters") or {"type": "object", "properties": {}} - - # The SDK wants a Python callable and derives the input schema from - # its signature — there is no inputSchema parameter on either the - # decorator or add_tool(). So build a closure that takes the arguments - # dict, dispatches via handle_function_call, returns the result - # string, and carries a __signature__ synthesized from the Hermes - # JSON Schema (see _signature_from_schema) for the SDK to read. - def _make_handler(tool_name: str, schema: dict | None): - sig, annots = _signature_from_schema(schema) - - def _dispatch(**kwargs: Any) -> str: - try: - # Filter out None values before dispatch so unset optionals - # aren't forwarded to the handler. - args = {k: v for k, v in kwargs.items() if v is not None} - return handle_function_call(tool_name, args or {}) - except Exception as exc: - logger.exception("tool %s raised", tool_name) - return json.dumps({"error": str(exc), "tool": tool_name}) - - _dispatch.__name__ = tool_name - _dispatch.__doc__ = description - _dispatch.__signature__ = sig - _dispatch.__annotations__ = {**annots, "return": str} - return _dispatch - + handler = _make_handler(name, params_schema, description) try: - mcp.add_tool( - _make_handler(name, params_schema), - name=name, - description=description, - ) + mcp.add_tool(handler, name=name, description=description) except TypeError: - # Older mcp SDK signature — fall back to decorator-style. The - # synthesized __signature__ on the handler still drives schema - # generation there. - handler = _make_handler(name, params_schema) - handler = mcp.tool(name=name, description=description)(handler) - + # Older mcp SDK: decorator-style registration; __signature__ still drives schema. + mcp.tool(name=name, description=description)(_make_handler(name, params_schema, description)) exposed_count += 1 - logger.info( - "hermes-tools MCP server registered %d/%d tools", - exposed_count, - len(EXPOSED_TOOLS), - ) + logger.info("hermes-tools MCP server registered %d/%d tools", exposed_count, len(EXPOSED_TOOLS)) return mcp @@ -254,15 +151,12 @@ def main(argv: Optional[list[str]] = None) -> int: """Entry point for `python -m agent.transports.hermes_tools_mcp_server`.""" argv = argv or sys.argv[1:] verbose = "--verbose" in argv or "-v" in argv - - log_level = logging.INFO if verbose else logging.WARNING logging.basicConfig( - level=log_level, + level=logging.INFO if verbose else logging.WARNING, stream=sys.stderr, # MCP uses stdio for protocol — logs MUST go to stderr format="%(asctime)s [%(levelname)s] %(name)s: %(message)s", ) - - # Quiet mode: keep Hermes' own banners off stdout (which is the MCP wire). + # Keep Hermes' own banners off stdout (the MCP wire). os.environ.setdefault("HERMES_QUIET", "1") os.environ.setdefault("HERMES_REDACT_SECRETS", "true") @@ -271,13 +165,10 @@ def main(argv: Optional[list[str]] = None) -> int: except ImportError as exc: sys.stderr.write(f"hermes-tools MCP server cannot start: {exc}\n") return 2 - - # MCPServer.run() defaults to stdio transport, which is what codex - # spawns us on. try: - server.run() + server.run() # defaults to stdio transport, which codex spawns us on except KeyboardInterrupt: - return 0 + pass except Exception as exc: logger.exception("hermes-tools MCP server crashed") sys.stderr.write(f"hermes-tools MCP server error: {exc}\n") diff --git a/agent/transports/types.py b/agent/transports/types.py index 8d9b82a49b..e572fdab5d 100644 --- a/agent/transports/types.py +++ b/agent/transports/types.py @@ -1,11 +1,8 @@ """Shared types for normalized provider responses. -These dataclasses define the canonical shape that all provider adapters -normalize responses to. The shared surface is intentionally minimal — -only fields that every downstream consumer reads are top-level. -Protocol-specific state goes in ``provider_data`` dicts (response-level -and per-tool-call) so that protocol-aware code paths can access it -without polluting the shared type. +Only fields every downstream consumer reads are top-level; protocol-specific +state lives in ``provider_data`` (response-level and per-tool-call) so +protocol-aware code can reach it without widening the shared type. """ from __future__ import annotations @@ -19,17 +16,11 @@ from typing import Any class ToolCall: """A normalized tool call from any provider. - ``id`` is the protocol's canonical identifier — what gets used in - ``tool_call_id`` / ``tool_use_id`` when constructing tool result - messages. May be ``None`` when the provider omits it; the agent - fills it via ``_deterministic_call_id()`` before storing in history. - - ``provider_data`` carries per-tool-call protocol metadata that only - protocol-aware code reads: - - * Codex: ``{"call_id": "call_XXX", "response_item_id": "fc_XXX"}`` - * Gemini: ``{"extra_content": {"google": {"thought_signature": "..."}}}`` - * Others: ``None`` + ``id`` is the protocol's canonical identifier (``tool_call_id`` / ``tool_use_id``); + may be ``None`` when the provider omits it — the agent fills it via + ``_deterministic_call_id()`` before storing history. + ``provider_data``: Codex ``{"call_id", "response_item_id"}``, Gemini + ``{"extra_content": {"google": {"thought_signature": ...}}}``, else ``None``. """ id: str | None @@ -37,43 +28,23 @@ class ToolCall: arguments: str # JSON string provider_data: dict[str, Any] | None = field(default=None, repr=False) - # ── Backward compatibility ────────────────────────────────── - # The agent loop reads tc.function.name / tc.function.arguments - # throughout run_agent.py (45+ sites). These properties let - # NormalizedResponse pass through without the _nr_to_assistant_message - # shim, while keeping ToolCall's canonical fields flat. + # Back-compat: run_agent reads tc.function.name / tc.function.arguments (45+ + # sites) and getattr()s the provider fields, so expose them as properties. @property def type(self) -> str: return "function" @property def function(self) -> ToolCall: - """Return self so tc.function.name / tc.function.arguments work.""" return self - @property - def call_id(self) -> str | None: - """Codex call_id from provider_data, accessed via getattr by _build_assistant_message.""" - return (self.provider_data or {}).get("call_id") + def _pd(self, key: str) -> Any: + return (self.provider_data or {}).get(key) - @property - def response_item_id(self) -> str | None: - """Codex response_item_id from provider_data.""" - return (self.provider_data or {}).get("response_item_id") - - @property - def extra_content(self) -> dict[str, Any] | None: - """Gemini extra_content (thought_signature) from provider_data. - - Gemini 3 thinking models attach ``extra_content`` with a - ``thought_signature`` to each tool call. This signature must be - replayed on subsequent API calls — without it the API rejects the - request with HTTP 400. The chat_completions transport stores this - in ``provider_data["extra_content"]``; this property exposes it so - ``_build_assistant_message`` can ``getattr(tc, "extra_content")`` - uniformly. - """ - return (self.provider_data or {}).get("extra_content") + call_id = property(lambda self: self._pd("call_id")) + response_item_id = property(lambda self: self._pd("response_item_id")) + # Gemini thought_signature; must be replayed on later calls or the API returns HTTP 400. + extra_content = property(lambda self: self._pd("extra_content")) @dataclass @@ -85,20 +56,22 @@ class Usage: total_tokens: int = 0 cached_tokens: int = 0 + @classmethod + def from_openai(cls, u: Any) -> Usage: + """Build from an OpenAI-shaped usage object, treating missing/None counts as 0.""" + return cls( + prompt_tokens=getattr(u, "prompt_tokens", 0) or 0, + completion_tokens=getattr(u, "completion_tokens", 0) or 0, + total_tokens=getattr(u, "total_tokens", 0) or 0, + ) + @dataclass class NormalizedResponse: """Normalized API response from any provider. - Shared fields are truly cross-provider — every caller can rely on - them without branching on api_mode. Protocol-specific state goes in - ``provider_data`` so that only protocol-aware code paths read it. - - Response-level ``provider_data`` examples: - - * Anthropic: ``{"reasoning_details": [...]}`` - * Codex: ``{"codex_reasoning_items": [...], "codex_message_items": [...]}`` - * Others: ``None`` + Response-level ``provider_data``: Anthropic ``{"reasoning_details": [...]}``, + Codex ``{"codex_reasoning_items": [...], "codex_message_items": [...]}``, else ``None``. """ content: str | None @@ -108,73 +81,27 @@ class NormalizedResponse: usage: Usage | None = None provider_data: dict[str, Any] | None = field(default=None, repr=False) - # ── Backward compatibility ────────────────────────────────── - # The shim _nr_to_assistant_message() mapped these from provider_data. - # These properties let NormalizedResponse pass through directly. - @property - def reasoning_content(self) -> str | None: - pd = self.provider_data or {} - return pd.get("reasoning_content") + # Back-compat accessors so NormalizedResponse passes through where the old + # _nr_to_assistant_message() shim mapped these from provider_data. + def _pd(self, key: str) -> Any: + return (self.provider_data or {}).get(key) - @property - def reasoning_details(self): - pd = self.provider_data or {} - return pd.get("reasoning_details") - - @property - def anthropic_content_blocks(self): - """Verbatim, order-preserving Anthropic content blocks for a turn. - - Present only when an Anthropic turn interleaves signed thinking with - tool_use — the one shape the parallel reasoning_details + tool_calls - lists reconstruct in the wrong order, invalidating thinking-block - signatures on replay. See agent/transports/anthropic.py. - """ - pd = self.provider_data or {} - return pd.get("anthropic_content_blocks") - - @property - def bedrock_content_blocks(self): - """Verbatim, order-preserving Bedrock Converse content blocks.""" - pd = self.provider_data or {} - return pd.get("bedrock_content_blocks") - - @property - def codex_reasoning_items(self): - pd = self.provider_data or {} - return pd.get("codex_reasoning_items") - - @property - def codex_message_items(self): - pd = self.provider_data or {} - return pd.get("codex_message_items") + reasoning_content = property(lambda self: self._pd("reasoning_content")) + reasoning_details = property(lambda self: self._pd("reasoning_details")) + # Order-preserving Anthropic blocks, present only when a turn interleaves signed + # thinking with tool_use (replay order invalidates signatures otherwise). + anthropic_content_blocks = property(lambda self: self._pd("anthropic_content_blocks")) + bedrock_content_blocks = property(lambda self: self._pd("bedrock_content_blocks")) # order-preserving Converse blocks + codex_reasoning_items = property(lambda self: self._pd("codex_reasoning_items")) + codex_message_items = property(lambda self: self._pd("codex_message_items")) -# --------------------------------------------------------------------------- -# Factory helpers -# --------------------------------------------------------------------------- - - -def build_tool_call( - id: str | None, - name: str, - arguments: Any, - **provider_fields: Any, -) -> ToolCall: - """Build a ``ToolCall``, auto-serialising *arguments* if it's a dict. - - Any extra keyword arguments are collected into ``provider_data``. - """ +def build_tool_call(id: str | None, name: str, arguments: Any, **provider_fields: Any) -> ToolCall: + """Build a ``ToolCall``; dict *arguments* are JSON-serialised, extra kwargs become ``provider_data``.""" args_str = json.dumps(arguments) if isinstance(arguments, dict) else str(arguments) - pd = dict(provider_fields) if provider_fields else None - return ToolCall(id=id, name=name, arguments=args_str, provider_data=pd) + return ToolCall(id=id, name=name, arguments=args_str, provider_data=dict(provider_fields) if provider_fields else None) def map_finish_reason(reason: str | None, mapping: dict[str, str]) -> str: - """Translate a provider-specific stop reason to the normalised set. - - Falls back to ``"stop"`` for unknown or ``None`` reasons. - """ - if reason is None: - return "stop" - return mapping.get(reason, "stop") + """Translate a provider stop reason via *mapping*; unknown or ``None`` -> ``"stop"``.""" + return "stop" if reason is None else mapping.get(reason, "stop")