From 267cdbecf369adf47ab267b76ff2caa5210f8050 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 11:21:11 -0700 Subject: [PATCH] refactor(agent/adapters): simplify anthropic adapter, message convert, endpoints (-857 LOC) Table-driven model capability checks and beta-header assembly, shared _cache_control_of/_block_type/_image_block_from_data_url helpers in the message converter, extracted _apply_claude_code_identity/_base_client_kwargs. convert_messages_to_anthropic / convert_tools_to_anthropic and request kwargs verified byte-identical against merge-base. --- agent/anthropic_adapter.py | 1332 ++++++++++----------------- agent/anthropic_endpoints.py | 223 ++--- agent/anthropic_message_convert.py | 1334 ++++++++++------------------ 3 files changed, 1016 insertions(+), 1873 deletions(-) diff --git a/agent/anthropic_adapter.py b/agent/anthropic_adapter.py index e03c07b9b3..3903d26895 100644 --- a/agent/anthropic_adapter.py +++ b/agent/anthropic_adapter.py @@ -1,116 +1,56 @@ """Anthropic Messages API adapter for Hermes Agent. Translates between Hermes's internal OpenAI-style message format and -Anthropic's Messages API. Follows the same pattern as the codex_responses -adapter — all provider-specific logic is isolated here. +Anthropic's Messages API; all provider-specific logic is isolated here. Auth supports: - - Regular API keys (sk-ant-api*) → x-api-key header - - OAuth setup-tokens (sk-ant-oat*) → Bearer auth + beta header - - Claude Code credentials (~/.claude.json or ~/.claude/.credentials.json) → Bearer auth + - Regular API keys (sk-ant-api*) -> x-api-key header + - OAuth setup-tokens (sk-ant-oat*) -> Bearer auth + beta header + - Claude Code credentials (~/.claude.json or ~/.claude/.credentials.json) -> Bearer auth """ -import copy -import json import logging -import os -import platform +import math import re -import secrets -import stat import subprocess -from pathlib import Path -from urllib.parse import urlparse +from pathlib import Path # noqa: F401 (tests patch ``anthropic_adapter.Path.home``) +from typing import Any, Dict, List, Optional -from hermes_constants import get_hermes_home -from typing import Any, Dict, List, Optional, Tuple -from utils import base_url_host_matches, base_url_hostname, normalize_proxy_env_vars -from agent.secret_scope import get_secret as _get_secret +from utils import normalize_proxy_env_vars -# This module keeps client construction and the Messages API call itself. The -# three surfaces it used to inline now live next to it: -# +# This module keeps client construction and the Messages API call itself; the +# three surfaces it used to inline live next to it and are re-exported below so +# long-standing ``from agent.anthropic_adapter import ...`` imports keep resolving: # agent/anthropic_endpoints.py base-URL/endpoint-family predicates # agent/anthropic_message_convert.py OpenAI -> Anthropic payload conversion # agent/anthropic_credentials.py credential sources, OAuth, refresh commit -# -# All three are re-exported below so long standing -# ``from agent.anthropic_adapter import resolve_anthropic_token`` (or -# ``convert_messages_to_anthropic``, ...) imports keep resolving. from agent.anthropic_endpoints import ( # noqa: F401 - _KIMI_FAMILY_EXACT_SLUGS, - _KIMI_FAMILY_MODEL_PREFIXES, - _base_url_needs_context_1m_beta, - _is_azure_anthropic_endpoint, - _is_deepseek_anthropic_endpoint, - _is_kimi_coding_endpoint, - _is_kimi_family_endpoint, - _is_minimax_anthropic_endpoint, - _is_nous_portal_endpoint, - _is_opencode_endpoint, - _is_third_party_anthropic_endpoint, - _model_name_is_kimi_family, - _normalize_base_url_text, - _requires_bearer_auth, + _KIMI_FAMILY_EXACT_SLUGS, _KIMI_FAMILY_MODEL_PREFIXES, _base_url_needs_context_1m_beta, + _is_azure_anthropic_endpoint, _is_deepseek_anthropic_endpoint, _is_kimi_coding_endpoint, + _is_kimi_family_endpoint, _is_minimax_anthropic_endpoint, _is_nous_portal_endpoint, + _is_opencode_endpoint, _is_third_party_anthropic_endpoint, _model_name_is_kimi_family, + _normalize_base_url_text, _requires_bearer_auth ) from agent.anthropic_message_convert import ( # noqa: F401 - _EMPTY_TEXT_PLACEHOLDER, - _apply_assistant_cache_control_to_last_cacheable_block, - _content_parts_to_anthropic_blocks, - _convert_assistant_message, - _convert_content_part_to_anthropic, - _convert_content_to_anthropic, - _convert_tool_message_to_result, - _convert_user_message, - _ensure_leading_user_turn, - _evict_old_screenshots, - _extract_preserved_thinking_blocks, - _fix_blank_text_blocks_in_list, - _image_source_from_openai_url, - _is_bedrock_model_id, - _manage_thinking_signatures, - _merge_consecutive_roles, - _normalize_tool_input_schema, - _safe_text, - _sanitize_replay_block, - _sanitize_tool_id, - _scrub_blank_text_blocks, - _strip_orphaned_tool_blocks, - _to_plain_data, - convert_messages_to_anthropic, - convert_tools_to_anthropic, - normalize_model_name, + _EMPTY_TEXT_PLACEHOLDER, _apply_assistant_cache_control_to_last_cacheable_block, + _content_parts_to_anthropic_blocks, _convert_assistant_message, _convert_content_part_to_anthropic, + _convert_content_to_anthropic, _convert_tool_message_to_result, _convert_user_message, + _ensure_leading_user_turn, _evict_old_screenshots, _extract_preserved_thinking_blocks, + _fix_blank_text_blocks_in_list, _image_source_from_openai_url, _is_bedrock_model_id, + _manage_thinking_signatures, _merge_consecutive_roles, _normalize_tool_input_schema, _safe_text, + _sanitize_replay_block, _sanitize_tool_id, _scrub_blank_text_blocks, _strip_orphaned_tool_blocks, + _to_plain_data, convert_messages_to_anthropic, convert_tools_to_anthropic, normalize_model_name ) from agent.anthropic_credentials import ( # noqa: F401 - _OAUTH_CLIENT_ID, - _OAUTH_REDIRECT_URI, - _OAUTH_SCOPES, - _OAUTH_TOKEN_URL, - _OAUTH_TOKEN_URLS, - _OAUTH_TOKEN_USER_AGENT, - CredentialPersistError, - _generate_pkce, - _get_hermes_oauth_file, - _getenv, - _is_oauth_token, - _prefer_refreshable_claude_code_token, - _read_claude_code_credentials_from_file, - _read_claude_code_credentials_from_keychain, - _refresh_oauth_token, - _resolve_anthropic_pool_token, - _resolve_claude_code_token_from_credentials, - _write_claude_code_credentials, - _write_hermes_oauth_credentials, - claude_code_credentials_path, - is_claude_code_token_valid, - is_rotation_consumed_uncommitted, - mark_rotation_consumed_uncommitted, - read_claude_code_credentials, - read_hermes_oauth_credentials, - refresh_anthropic_oauth_pure, - resolve_anthropic_token, - run_hermes_oauth_login_pure, - run_oauth_setup_token, + _OAUTH_CLIENT_ID, _OAUTH_REDIRECT_URI, _OAUTH_SCOPES, _OAUTH_TOKEN_URL, _OAUTH_TOKEN_URLS, + _OAUTH_TOKEN_USER_AGENT, CredentialPersistError, _generate_pkce, _get_hermes_oauth_file, _getenv, + _is_oauth_token, _prefer_refreshable_claude_code_token, _read_claude_code_credentials_from_file, + _read_claude_code_credentials_from_keychain, _refresh_oauth_token, _resolve_anthropic_pool_token, + _resolve_claude_code_token_from_credentials, _write_claude_code_credentials, + _write_hermes_oauth_credentials, claude_code_credentials_path, is_claude_code_token_valid, + is_rotation_consumed_uncommitted, mark_rotation_consumed_uncommitted, read_claude_code_credentials, + read_hermes_oauth_credentials, refresh_anthropic_oauth_pure, resolve_anthropic_token, + run_hermes_oauth_login_pure, run_oauth_setup_token ) try: @@ -121,14 +61,10 @@ except Exception: _HERMES_VERSION = "0.0.0" - -# NOTE: `import anthropic` is deliberately NOT at module top — the SDK pulls -# ~220 ms of imports (anthropic.types, anthropic.lib.tools._beta_runner, etc.) -# and the 3 usage sites (build_anthropic_client, build_anthropic_bedrock_client, -# read_claude_code_credentials_from_keychain) are all on cold user-triggered -# paths. Access via the `_get_anthropic_sdk()` accessor below, which caches -# the module after the first call and returns None on ImportError. -_anthropic_sdk: Any = ... # sentinel — None means "tried and missing" +# ``import anthropic`` is deliberately NOT at module top: the SDK costs ~220 ms +# of imports and every usage site is a cold user-triggered path. ``...`` is the +# "not yet tried" sentinel; None means tried and missing. +_anthropic_sdk: Any = ... def _get_anthropic_sdk(): @@ -138,10 +74,7 @@ def _get_anthropic_sdk(): try: from tools.lazy_deps import ensure as _lazy_ensure _lazy_ensure("provider.anthropic", prompt=False) - except ImportError: - pass - except Exception: - # FeatureUnavailable — fall through to ImportError handling below + except Exception: # ImportError or FeatureUnavailable — fall through to the import below pass try: import anthropic as _sdk @@ -150,17 +83,25 @@ def _get_anthropic_sdk(): _anthropic_sdk = None return _anthropic_sdk + +def _require_sdk(purpose: str, verb: str = "Install it with"): + """``_get_anthropic_sdk()`` or ImportError naming the feature that needs it.""" + sdk = _get_anthropic_sdk() + if sdk is None: + raise ImportError( + f"The 'anthropic' package is required for {purpose}. " + f"{verb}: pip install 'anthropic>=0.39.0'" + ) + return sdk + + logger = logging.getLogger(__name__) THINKING_BUDGET = {"xhigh": 32000, "high": 16000, "medium": 8000, "low": 4000} -# Hermes effort → Anthropic adaptive-thinking effort (output_config.effort). -# Anthropic exposes 5 levels on 4.7+: low, medium, high, xhigh, max. -# Opus/Sonnet 4.6 only expose 4 levels: low, medium, high, max — no xhigh. -# We preserve xhigh as xhigh on 4.7+ (the recommended default for coding/ -# agentic work) and downgrade it to max on pre-4.7 adaptive models (which -# is the strongest level they accept). "minimal" is a legacy alias that -# maps to low on every model. See: -# https://platform.claude.com/docs/en/about-claude/models/migration-guide +# Hermes effort -> Anthropic adaptive-thinking effort (output_config.effort). +# 4.7+ exposes low/medium/high/xhigh/max; Opus/Sonnet 4.6 have no xhigh, so +# callers downgrade xhigh->max there (see _supports_xhigh_effort). "minimal" is +# a legacy alias for low on every model. ADAPTIVE_EFFORT_MAP = { "ultra": "max", "max": "max", @@ -172,26 +113,17 @@ ADAPTIVE_EFFORT_MAP = { } # ── Anthropic thinking-mode classification ──────────────────────────── -# Claude 4.6 replaced budget-based extended thinking with *adaptive* thinking, -# and 4.7 additionally forbids the manual ``thinking`` block entirely and drops -# temperature/top_p/top_k. Newer Claude releases (4.8, and named models like -# claude-fable-5) follow the same modern contract — but they share no common -# version substring, so an allowlist of version numbers ("4.6", "4.7", …) goes -# stale the moment a model ships without a recognized number and silently -# routes it down the legacy manual-thinking path. -# -# Instead we DEFAULT unknown Claude models to the modern contract and keep an -# explicit *legacy* list of the older Claude families that still require manual -# thinking. This mirrors _get_anthropic_max_output's "default to newest" design -# (future models are unlikely to regress to the older contract), so each new -# Claude release works without a code change. -# -# Non-Claude Anthropic-Messages models (minimax, qwen3, GLM, …) are NOT Claude, -# so they fall through to the legacy path automatically — exactly what those -# manual-thinking endpoints need. +# Claude 4.6 replaced budget-based extended thinking with *adaptive* thinking; +# 4.7 additionally forbids the manual ``thinking`` block and drops +# temperature/top_p/top_k. Newer releases (4.8, named models like claude-fable-5) +# share no common version substring, so an allowlist of "modern" versions would +# go stale and silently route a new model down the legacy path. We therefore +# DEFAULT unknown Claude models to the modern contract and keep explicit +# *legacy* lists (mirroring _get_anthropic_max_output's default-to-newest). +# Non-Claude Anthropic-Messages models (minimax, qwen3, GLM, ...) are not Claude +# and fall through to the legacy manual-thinking path, which is what they need. -# Older Claude families that DON'T support adaptive thinking (manual thinking -# with budget_tokens only). Substring-matched against the model name. +# Older Claude families that need manual thinking (budget_tokens only). _LEGACY_MANUAL_THINKING_CLAUDE_SUBSTRINGS = ( "claude-3", # 3, 3.5, 3.7 "claude-opus-4-0", "claude-opus-4.0", "claude-opus-4-1", "claude-opus-4.1", @@ -202,118 +134,89 @@ _LEGACY_MANUAL_THINKING_CLAUDE_SUBSTRINGS = ( "claude-haiku-4-5", "claude-haiku-4.5", ) -# Older Claude families that DON'T accept the "xhigh" effort level (4.6 only -# supports low/medium/high/max). xhigh arrived with Opus 4.7. Adaptive models -# not in this list (4.7, 4.8, fable, future) accept xhigh. +# Adaptive families that reject the "xhigh" effort (arrived with Opus 4.7) and +# still accept sampling params. _NO_XHIGH_CLAUDE_SUBSTRINGS = ( "claude-opus-4-6", "claude-opus-4.6", "claude-sonnet-4-6", "claude-sonnet-4.6", ) -# Adaptive Claude families that REJECT a thinking disable — thinking is -# mandatory and ``thinking: {"type": "disabled"}`` answers HTTP 400. The Portal -# catalog flags the same families with ``reasoning.mandatory``. -# -# Unlike the two lists above, the failure here is asymmetric: a missing entry -# 400s the turn, while a spurious one only leaves thinking on. When in doubt, -# add the family. +# Adaptive families where thinking is mandatory: ``thinking: {"type": +# "disabled"}`` answers HTTP 400 (Portal flags them ``reasoning.mandatory``). +# The failure is asymmetric — a missing entry 400s the turn, a spurious one +# only leaves thinking on — so when in doubt, add the family. _MANDATORY_THINKING_CLAUDE_SUBSTRINGS = ( "claude-fable", ) +_FAST_MODE_SUPPORTED_SUBSTRINGS = ("opus-4-8", "opus-4.8", "opus-5") + def _is_claude_model(model: str | None) -> bool: return "claude" in (model or "").lower() -_FAST_MODE_SUPPORTED_SUBSTRINGS = ("opus-4-8", "opus-4.8", "opus-5") +def _model_matches(model: str, substrings) -> bool: + """Case-insensitive substring match of ``model`` against a family list.""" + m = model.lower() + return any(v in m for v in substrings) + # ── Max output token limits per Anthropic model ─────────────────────── -# Source: Anthropic docs + Cline model catalog. Anthropic's API requires -# max_tokens as a mandatory field. Previously we hardcoded 16384, which -# starves thinking-enabled models (thinking tokens count toward the limit). +# Anthropic requires max_tokens; a fixed 16384 starved thinking-enabled models +# (thinking tokens count toward the limit). Source: Anthropic docs + Cline catalog. _ANTHROPIC_OUTPUT_LIMITS = { - # Mythos-class named models (claude-fable-5, …) — 1M context, reasoning - "claude-fable": 128_000, - # Claude Sonnet 5 + "claude-fable": 128_000, # Mythos-class named models — 1M context, reasoning "claude-sonnet-5": 128_000, - # Claude 4.8 "claude-opus-4-8": 128_000, - # Claude 4.7 "claude-opus-4-7": 128_000, - # Claude 4.6 "claude-opus-4-6": 128_000, "claude-sonnet-4-6": 64_000, - # Claude 4.5 "claude-opus-4-5": 64_000, "claude-sonnet-4-5": 64_000, "claude-haiku-4-5": 64_000, - # Claude 4 "claude-opus-4": 32_000, "claude-sonnet-4": 64_000, - # Claude 3.7 "claude-3-7-sonnet": 128_000, - # Claude 3.5 "claude-3-5-sonnet": 8_192, "claude-3-5-haiku": 8_192, - # Claude 3 "claude-3-opus": 4_096, "claude-3-sonnet": 4_096, "claude-3-haiku": 4_096, - # Third-party Anthropic-compatible providers - "minimax": 131_072, - # Qwen models via DashScope Anthropic-compatible endpoint - # DashScope enforces max_tokens ∈ [1, 65536] - "qwen3": 65_536, + "minimax": 131_072, # third-party Anthropic-compatible + "qwen3": 65_536, # DashScope enforces max_tokens in [1, 65536] } -# For any model not in the table, assume the highest current limit. -# Future Anthropic models are unlikely to have *less* output capacity. +# Unknown models get the highest current limit: future models are unlikely to +# have *less* output capacity. _ANTHROPIC_DEFAULT_OUTPUT_LIMIT = 128_000 def _get_anthropic_max_output(model: str) -> int: - """Look up the max output token limit for an Anthropic model. - - Uses substring matching against _ANTHROPIC_OUTPUT_LIMITS so date-stamped - model IDs (claude-sonnet-4-5-20250929) and variant suffixes (:1m, :fast) - resolve correctly. Longest-prefix match wins to avoid e.g. "claude-3-5" - matching before "claude-3-5-sonnet". - - Normalizes dots to hyphens so that model names like - ``anthropic/claude-opus-4.6`` match the ``claude-opus-4-6`` table key. + """Max output tokens for ``model`` via longest substring match against + ``_ANTHROPIC_OUTPUT_LIMITS`` (so date-stamped ids and ``:1m``/``:fast`` + suffixes resolve, and ``claude-3-5-sonnet`` beats ``claude-3-5``). Dots are + normalized to hyphens so ``claude-opus-4.6`` matches ``claude-opus-4-6``. """ m = model.lower().replace(".", "-") - best_key = "" - best_val = _ANTHROPIC_DEFAULT_OUTPUT_LIMIT - for key, val in _ANTHROPIC_OUTPUT_LIMITS.items(): - if key in m and len(key) > len(best_key): - best_key = key - best_val = val - return best_val + best_key = max((key for key in _ANTHROPIC_OUTPUT_LIMITS if key in m), key=len, default=None) + return _ANTHROPIC_OUTPUT_LIMITS[best_key] if best_key else _ANTHROPIC_DEFAULT_OUTPUT_LIMIT def _resolve_positive_anthropic_max_tokens(value) -> Optional[int]: - """Return ``value`` floored to a positive int, or ``None`` if it is not a - finite positive number. Ported from openclaw/openclaw#66664. + """``value`` floored to a positive int, or None when it is not a finite + positive number. - Anthropic's Messages API rejects ``max_tokens`` values that are 0, - negative, non-integer, or non-finite with HTTP 400. Python's ``or`` - idiom (``max_tokens or fallback``) correctly catches ``0`` but lets - negative ints and fractional floats (``-1``, ``0.5``) through to the - API, producing a user-visible failure instead of a local error. + Anthropic 400s on max_tokens that are 0, negative, fractional or non-finite; + the ``max_tokens or fallback`` idiom catches 0 but lets ``-1``/``0.5`` + through. Booleans are excluded explicitly (they subclass int). """ - # Booleans are a subclass of int — exclude explicitly so ``True`` doesn't - # silently become 1 and ``False`` doesn't become 0. - if isinstance(value, bool): - return None - if not isinstance(value, (int, float)): + if isinstance(value, bool) or not isinstance(value, (int, float)): return None try: - import math if not math.isfinite(value): return None - except Exception: + except Exception: # e.g. OverflowError for ints too large for float return None floored = int(value) # truncates toward zero for floats return floored if floored > 0 else None @@ -324,19 +227,9 @@ def _resolve_anthropic_messages_max_tokens( model: str, context_length: Optional[int] = None, ) -> int: - """Resolve the ``max_tokens`` budget for an Anthropic Messages call. - - Prefers ``requested`` when it is a positive finite number; otherwise - falls back to the model's output ceiling. Raises ``ValueError`` if no - positive budget can be resolved (should not happen with current model - table defaults, but guards against a future regression where - ``_get_anthropic_max_output`` could return ``0``). - - Separately, callers apply a context-window clamp — this resolver does - not, to keep the positive-value contract independent of endpoint - specifics. - - Ported from openclaw/openclaw#66664 (resolveAnthropicMessagesMaxTokens). + """``requested`` when it is a positive finite number, else the model's output + ceiling. Raises ValueError if neither is positive. The context-window clamp + is the caller's job so the positive-value contract stays endpoint-agnostic. """ resolved = _resolve_positive_anthropic_max_tokens(requested) if resolved is not None: @@ -351,177 +244,101 @@ def _resolve_anthropic_messages_max_tokens( def _supports_adaptive_thinking(model: str) -> bool: - """Return True for Claude models that use adaptive thinking (4.6+). - - Defaults *unknown* Claude models to adaptive (the modern contract) and - only returns False for the explicit legacy list of older Claude families - that require manual budget-based thinking. Non-Claude Anthropic-Messages - models (minimax, qwen3, …) return False so they keep the manual path. - - Kimi / Moonshot models are the exception: their Anthropic-compatible - endpoints implement the adaptive contract (``thinking.type="adaptive"`` - + ``output_config.effort``, including ``xhigh`` and ``display``). + """True for Claude models using adaptive thinking (4.6+): unknown Claude + models default to adaptive, the explicit legacy list stays manual, and + non-Claude models return False — except Kimi/Moonshot, whose Anthropic- + compatible endpoints implement the adaptive contract (incl. xhigh/display). """ if _model_name_is_kimi_family(model): return True if not _is_claude_model(model): return False - m = model.lower() - return not any(v in m for v in _LEGACY_MANUAL_THINKING_CLAUDE_SUBSTRINGS) + return not _model_matches(model, _LEGACY_MANUAL_THINKING_CLAUDE_SUBSTRINGS) def _supports_xhigh_effort(model: str) -> bool: - """Return True for models that accept the 'xhigh' adaptive effort level. - - Opus 4.7 introduced xhigh as a distinct level between high and max. - Pre-4.7 adaptive models (Opus/Sonnet 4.6) only accept low/medium/high/max - and reject xhigh with an HTTP 400. Callers should downgrade xhigh→max - when this returns False. - - Defaults unknown adaptive Claude models to accepting xhigh (4.7+ contract); - only the 4.6 family and legacy manual-thinking models are excluded. - """ - if not _supports_adaptive_thinking(model): - return False - m = model.lower() - return not any(v in m for v in _NO_XHIGH_CLAUDE_SUBSTRINGS) + """True for models accepting the 'xhigh' effort (Opus 4.7+). Opus/Sonnet 4.6 + 400 on it — callers downgrade xhigh->max when this returns False.""" + return _supports_adaptive_thinking(model) and not _model_matches(model, _NO_XHIGH_CLAUDE_SUBSTRINGS) def _accepts_thinking_disable(model: str) -> bool: - """Return True when *model* accepts an explicit thinking disable. + """True when ``model`` accepts an explicit ``thinking: {"type": "disabled"}``. - Adaptive Claude models default to thinking ON, so "thinking off" only - takes effect if we actively send ``thinking: {"type": "disabled"}`` — - omitting the parameter leaves the upstream default in place and the model - thinks anyway. Reasoning-mandatory families reject the disable outright - with an HTTP 400, so they keep the omit-everything behavior. - - Legacy manual-thinking Claude models are excluded because they need no - disable: thinking is opt-in there via ``budget_tokens``, so not sending - the block already means off. - - Scoped to Claude deliberately. Kimi/Moonshot endpoints also speak the - adaptive contract, but their documented disable behavior is omission - (#13848) and they are not part of this bug; sending them a new parameter - on the strength of Claude's contract would be a guess. + Adaptive Claude thinks by default, so "off" only works if the disable is + sent; mandatory-thinking families 400 on it and keep the omit behavior. + Legacy manual-thinking models are opt-in via budget_tokens, so omission is + already off. Scoped to Claude: Kimi's documented disable is omission, and + sending it a new parameter on the strength of Claude's contract is a guess. """ - if not _is_claude_model(model): - return False - if not _supports_adaptive_thinking(model): - return False - m = model.lower() - return not any(v in m for v in _MANDATORY_THINKING_CLAUDE_SUBSTRINGS) + return ( + _is_claude_model(model) + and _supports_adaptive_thinking(model) + and not _model_matches(model, _MANDATORY_THINKING_CLAUDE_SUBSTRINGS) + ) def _forbids_sampling_params(model: str) -> bool: - """Return True for models that 400 on any non-default temperature/top_p/top_k. - - Opus 4.7 introduced this restriction; later Claude releases follow it. - Defaults unknown Claude models to forbidding sampling params (the modern - contract). The 4.6 family still accepts them, and the legacy manual-thinking - families (4.5 and older) accept them too, so both are excluded. Non-Claude - models are unaffected. Callers should omit these fields entirely rather than - passing zero/default values (the API rejects anything non-null). - """ - if not _is_claude_model(model): - return False - m = model.lower() - # 4.6 family is adaptive but still accepts sampling params. - if any(v in m for v in _NO_XHIGH_CLAUDE_SUBSTRINGS): - return False - return not any(v in m for v in _LEGACY_MANUAL_THINKING_CLAUDE_SUBSTRINGS) + """True for models that 400 on any non-default temperature/top_p/top_k + (Opus 4.7 and later; unknown Claude defaults to forbidding). The 4.6 family + and the legacy manual-thinking families still accept them. Callers omit the + fields entirely — the API rejects anything non-null, even defaults.""" + return _is_claude_model(model) and not _model_matches( + model, _NO_XHIGH_CLAUDE_SUBSTRINGS + _LEGACY_MANUAL_THINKING_CLAUDE_SUBSTRINGS + ) def _supports_fast_mode(model: str) -> bool: - """Return True for models that accept the ``speed: "fast"`` request param. + """True for models accepting ``speed: "fast"`` (Opus 4.8 / Opus 5, Claude API only). - Per the Anthropic fast-mode docs (research preview), the ``speed`` param - is supported on Opus 4.8 and Opus 5 — Claude API only. The matrix has - changed with nearly every Opus release, in both directions: - - - Opus 4.6 HAD fast mode at launch and LOST it (2026-06-29): requests - with ``speed: "fast"`` do not error — they silently run at standard - speed and bill standard rates (``usage.speed: "standard"``). Keeping - 4.6 in this allowlist would show users a fast toggle that does - nothing. - - Opus 4.7 never had it and hard-400s on the parameter. - - Dedicated ``…-fast`` model ids (e.g. OpenRouter's - ``claude-opus-4.8-fast``) select fast inference via the model field - itself and must NOT also receive the speed parameter. - - Keep this an explicit allowlist rather than a version-floor check so a - model that drops fast mode again fails closed (standard speed) instead - of silently 400'ing. + Explicit allowlist, not a version floor: the matrix has flipped both ways. + Opus 4.6 had fast mode and lost it (requests silently run and bill at + standard speed, so listing it would show a toggle that does nothing); Opus + 4.7 hard-400s on the param. Dedicated ``...-fast`` ids select fast inference + via the model field and must NOT also receive the speed parameter. """ - if "-fast" in model: - return False - return any(v in model for v in _FAST_MODE_SUPPORTED_SUBSTRINGS) + return "-fast" not in model and any(v in model for v in _FAST_MODE_SUPPORTED_SUBSTRINGS) -# Beta headers for enhanced features that are safe on ordinary/native Anthropic -# requests. As of Opus 4.7 (2026-04-16), these are GA on Claude 4.6+ — the -# beta headers are still accepted (harmless no-op) but not required. Kept -# here so older Claude (4.5, 4.1) + compatible endpoints that still gate on -# the headers continue to get the enhanced features. -# -# Do NOT include ``context-1m-2025-08-07`` here. Anthropic returns HTTP 400 -# ("long context beta is not yet available for this subscription") for -# accounts without the long-context beta, which breaks normal short auxiliary -# calls like title generation/session summarization. -# -# ``context-1m-2025-08-07`` is still required to unlock the 1M context window -# on Claude Opus 4.6/4.7 and Sonnet 4.6 when served via AWS Bedrock or Azure -# AI Foundry. Add it only for those endpoint-specific paths below. +# Beta headers safe on ordinary/native Anthropic requests. GA on Claude 4.6+ +# (harmless no-op there) but older Claude and compatible endpoints still gate +# on them. Do NOT add ``context-1m-2025-08-07``: accounts without the +# long-context beta get HTTP 400 ("long context beta is not yet available for +# this subscription"), breaking short auxiliary calls. Bedrock/Azure still need +# it for 1M context and opt in on their own paths. _COMMON_BETAS = [ "interleaved-thinking-2025-05-14", "fine-grained-tool-streaming-2025-05-14", ] -# MiniMax's Anthropic-compatible endpoints fail tool-use requests when -# the fine-grained tool streaming beta is present. Omit it so tool calls -# fall back to the provider's default response path. +# MiniMax's Anthropic-compatible endpoints fail tool-use requests when this beta +# is present. _TOOL_STREAMING_BETA = "fine-grained-tool-streaming-2025-05-14" -# 1M context beta. Native Anthropic does not get this by default because some -# subscriptions reject it, but Bedrock/Azure still need it for 1M context. _CONTEXT_1M_BETA = "context-1m-2025-08-07" - -# Fast mode beta — enables the ``speed: "fast"`` request parameter for -# significantly higher output token throughput on Opus 4.6 (~2.5x). -# See https://platform.claude.com/docs/en/build-with-claude/fast-mode +# Enables the ``speed: "fast"`` request parameter. _FAST_MODE_BETA = "fast-mode-2026-02-01" - -# Additional beta headers required for OAuth/subscription auth. -# Matches what Claude Code (and pi-ai / OpenCode) send. +# Required for OAuth/subscription auth; matches Claude Code / pi-ai / OpenCode. _OAUTH_ONLY_BETAS = [ "claude-code-20250219", "oauth-2025-04-20", ] -# Claude Code identity — required for OAuth requests to be routed correctly. -# Without these, Anthropic's infrastructure intermittently 500s OAuth traffic. -# The version must stay reasonably current — Anthropic rejects OAuth requests -# when the spoofed user-agent version is too far behind the actual release. +# Claude Code identity — OAuth requests without it intermittently 500. Anthropic +# rejects OAuth requests whose user-agent version is too far behind the actual +# release, so the installed version is detected and this fallback kept current. _CLAUDE_CODE_VERSION_FALLBACK = "2.1.74" _claude_code_version_cache: Optional[str] = None def _detect_claude_code_version() -> str: - """Detect the installed Claude Code version, fall back to a static constant. - - Anthropic's OAuth infrastructure validates the user-agent version and may - reject requests with a version that's too old. Detecting dynamically means - users who keep Claude Code updated never hit stale-version 400s. - """ - import subprocess as _sp - + """Installed Claude Code version (``claude --version``), else the static fallback.""" for cmd in ("claude", "claude-code"): try: - result = _sp.run( + result = subprocess.run( [cmd, "--version"], capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5, ) if result.returncode == 0 and result.stdout.strip(): - # Output is like "2.1.74 (Claude Code)" or just "2.1.74" - version = result.stdout.strip().split()[0] + version = result.stdout.strip().split()[0] # "2.1.74 (Claude Code)" or "2.1.74" if version and version[0].isdigit(): return version except Exception: @@ -529,20 +346,23 @@ def _detect_claude_code_version() -> str: return _CLAUDE_CODE_VERSION_FALLBACK +def _get_claude_code_version() -> str: + """Lazily detect the installed Claude Code version when OAuth headers need it.""" + global _claude_code_version_cache + if _claude_code_version_cache is None: + _claude_code_version_cache = _detect_claude_code_version() + return _claude_code_version_cache + + _CLAUDE_CODE_SYSTEM_PREFIX = "You are Claude Code, Anthropic's official CLI for Claude." _MCP_TOOL_PREFIX = "mcp__" # Anthropic's OAuth billing classifier fingerprints certain Hermes tool -# schemas/prose as a third-party app and reroutes the request to the metered -# extra-usage lane, surfacing as HTTP 400 "You're out of extra usage" on a -# valid subscription token (#65365). Deterministic live A/B repros (issue -# #65365 comments, replayed with the anthropic-ratelimit-unified-* response -# headers as a lane oracle) isolated two independent triggers: -# - the ``session_search`` tool schema/name/prose, alone -# - the ``memory`` tool schema/name, alone -# Both are aliased to neutral names on the OAuth wire only. normalize_response -# reverses the mapping before dispatch, so tool behavior and API-key requests -# are unchanged. +# schemas/prose as a third-party app and reroutes to the metered extra-usage +# lane (HTTP 400 "You're out of extra usage" on a valid subscription). Live A/B +# repros isolated two independent triggers — the ``session_search`` tool +# (schema/name/prose) and the ``memory`` tool (schema/name) — so both are +# aliased on the OAuth wire only; normalize_response reverses the mapping. _OAUTH_TOOL_NAME_ALIASES = { "session_search": "chat_history_lookup", "memory": "context_notes", @@ -551,24 +371,16 @@ _OAUTH_TOOL_NAME_REVERSE_ALIASES = { wire_name: name for name, wire_name in _OAUTH_TOOL_NAME_ALIASES.items() } -# Aliases that are ALSO safe to substitute in free-form prose (system prompt -# text, tool descriptions). Only unambiguous snake_case tool tokens qualify: -# "memory" is ordinary English throughout the system prompt ("persistent -# memory across sessions", "OS, CPU, memory, disk") and inside the memory -# tool's own parameter docs (the ``target`` enum the model must still emit -# verbatim), so rewriting it in prose would corrupt guidance the model has -# to follow. Renaming a tool is a different operation from rewriting the -# vocabulary that describes it — a model that follows unaliased "memory" -# prose and calls ``memory`` still dispatches correctly: normalize_response -# resolves the bare name through the tool registry regardless. +# Aliases ALSO safe to substitute in free-form prose (system prompt, tool +# descriptions). "memory" is ordinary English throughout the prompt and inside +# the memory tool's own parameter docs (an enum the model must emit verbatim), +# so rewriting it would corrupt guidance; a model that calls bare ``memory`` +# still dispatches, since normalize_response resolves it through the registry. _OAUTH_PROSE_ALIAS_NAMES = frozenset({"session_search"}) -# Word-boundary matchers so a prose substitution can't corrupt a longer -# identifier that merely contains the token (project AGENTS.md / memory -# snapshots can carry arbitrary text, e.g. a path like -# ``tools/session_search_tool.py`` must not become -# ``tools/chat_history_lookup_tool.py``). ``\b`` treats ``_`` as a word -# char, so only the standalone token matches. +# Word-boundary matchers so a longer identifier containing the token (e.g. +# ``tools/session_search_tool.py`` in AGENTS.md) is left alone; ``\b`` treats +# ``_`` as a word char. _OAUTH_PROSE_ALIAS_PATTERNS = tuple( (re.compile(rf"\b{re.escape(name)}\b"), _OAUTH_TOOL_NAME_ALIASES[name]) for name in sorted(_OAUTH_PROSE_ALIAS_NAMES) @@ -582,49 +394,73 @@ def _apply_oauth_prose_aliases(text: str) -> str: return text -def _get_claude_code_version() -> str: - """Lazily detect the installed Claude Code version when OAuth headers need it.""" - global _claude_code_version_cache - if _claude_code_version_cache is None: - _claude_code_version_cache = _detect_claude_code_version() - return _claude_code_version_cache - - - - - def _common_betas_for_base_url( base_url: str | None, *, drop_context_1m_beta: bool = False, ) -> list[str]: - """Return the beta headers that are safe for the configured endpoint. + """Beta headers safe for the configured endpoint. - MiniMax's Anthropic-compatible endpoints (Bearer-auth) reject requests - that include Anthropic's ``fine-grained-tool-streaming`` beta — every - tool-use message triggers a connection error. They also reject the - 1M-context beta. Azure AI Foundry's Anthropic endpoint also uses - Bearer auth but keeps both betas (it needs the 1M beta for 1M context). - - The ``context-1m-2025-08-07`` beta is not sent to native Anthropic by - default because some subscriptions reject it. Add it only for endpoint - families that still require it for 1M context, currently Microsoft Foundry. - Bedrock uses its own client helper below and opts in explicitly. - - ``drop_context_1m_beta=True`` strips the 1M-context beta from any path that - would otherwise include it after a subscription/endpoint rejects the beta. + MiniMax (Bearer-auth) rejects both the fine-grained-tool-streaming beta + (every tool-use message errors) and the 1M-context beta. Azure AI Foundry + also uses Bearer auth but keeps both — it needs the 1M beta for 1M context, + which native Anthropic does not get by default (some subscriptions reject + it; Bedrock opts in via its own client helper). ``drop_context_1m_beta`` + strips the 1M beta after a subscription/endpoint rejected it. """ betas = list(_COMMON_BETAS) if _base_url_needs_context_1m_beta(base_url) and not drop_context_1m_beta: betas.append(_CONTEXT_1M_BETA) if _is_minimax_anthropic_endpoint(base_url): - _stripped = {_TOOL_STREAMING_BETA, _CONTEXT_1M_BETA} - return [b for b in betas if b not in _stripped] - if drop_context_1m_beta: - return [b for b in betas if b != _CONTEXT_1M_BETA] + return [b for b in betas if b not in (_TOOL_STREAMING_BETA, _CONTEXT_1M_BETA)] return betas +def _beta_header(betas: list) -> Dict[str, str]: + """``{"anthropic-beta": ...}`` when there are betas, else ``{}``.""" + return {"anthropic-beta": ",".join(betas)} if betas else {} + + +_ATTRIBUTION_HEADERS = { + "HTTP-Referer": "https://hermes-agent.nousresearch.com", + "X-Title": "Hermes Agent", +} + + +def _attribution_headers() -> Dict[str, str]: + """Same client-attribution set sent to OpenRouter / Vercel AI Gateway / Fireworks.""" + return {**_ATTRIBUTION_HEADERS, "User-Agent": f"HermesAgent/{_HERMES_VERSION}"} + + +def _client_timeout(timeout): + """httpx.Timeout with the caller's read timeout (default 900s) and a 10s connect.""" + from httpx import Timeout + + read = timeout if (isinstance(timeout, (int, float)) and timeout > 0) else 900.0 + return Timeout(timeout=float(read), connect=10.0) + + +def _base_client_kwargs(base_url, timeout) -> tuple[str, Dict[str, Any]]: + """Shared SDK constructor kwargs; returns ``(normalized_base_url, kwargs)``. + + Retry is delegated to hermes's outer loop (``max_retries=0``): the SDK + default of 2 uses its own backoff that ignores Retry-After and double- + retries inside our loop, burning request slots against a bucket that won't + refill for minutes. Any trailing ``/v1`` is stripped because the SDK appends + ``/v1/messages``. Azure endpoints need an ``api-version`` query param; it + goes through ``default_query`` so the base_url is not corrupted into + ``/anthropic?api-version=.../v1/messages``. + """ + kwargs: Dict[str, Any] = {"timeout": _client_timeout(timeout), "max_retries": 0} + normalized = _normalize_base_url_text(base_url) + if normalized: + normalized = re.sub(r"/v1/?$", "", normalized.rstrip("/")) + kwargs["base_url"] = normalized + if _is_azure_anthropic_endpoint(normalized) and "api-version" not in normalized: + kwargs["default_query"] = {"api-version": "2025-04-15"} + return normalized, kwargs + + def _build_anthropic_client_with_bearer_hook( token_provider, base_url: str = None, @@ -634,71 +470,27 @@ def _build_anthropic_client_with_bearer_hook( ): """Anthropic-on-Foundry Entra ID variant of :func:`build_anthropic_client`. - Anthropic SDK 0.86.0 stores ``api_key`` / ``auth_token`` as static - strings; there is no callable-token contract. To get per-request - bearer refresh (Microsoft's documented Foundry pattern), we hand - the SDK a custom ``httpx.Client`` whose request event hook mints a - fresh JWT from the Entra credential chain and rewrites - ``Authorization: Bearer `` on every outbound request. The SDK - ignores its own auth logic when ``http_client`` is provided (the - hook strips any pre-set Authorization). - - The placeholder ``auth_token`` is required because the SDK raises - ``AnthropicError`` at construction if neither ``api_key`` nor - ``auth_token`` is set — but the hook overrides it per-request so - the placeholder value never reaches Azure. + The SDK stores ``api_key``/``auth_token`` as static strings, so per-request + bearer refresh (Microsoft's documented Foundry pattern) is done with a custom + ``httpx.Client`` whose request hook mints a fresh JWT and rewrites + ``Authorization`` on every request; the SDK skips its own auth when + ``http_client`` is given. A placeholder ``auth_token`` is still required at + construction — the hook overrides it, and the sentinel makes any accidental + leak diagnosable in logs. """ - _anthropic_sdk = _get_anthropic_sdk() - if _anthropic_sdk is None: - raise ImportError( - "The 'anthropic' package is required for Azure Foundry Anthropic-style " - "endpoints with Entra ID auth. Install with: pip install 'anthropic>=0.39.0'" - ) - + sdk = _require_sdk("Azure Foundry Anthropic-style endpoints with Entra ID auth", verb="Install with") normalize_proxy_env_vars() - from httpx import Timeout from agent.azure_identity_adapter import build_bearer_http_client - _read_timeout = timeout if (isinstance(timeout, (int, float)) and timeout > 0) else 900.0 - timeout_obj = Timeout(timeout=float(_read_timeout), connect=10.0) + normalized_base_url, kwargs = _base_client_kwargs(base_url, timeout) + kwargs["http_client"] = build_bearer_http_client(token_provider, timeout=kwargs["timeout"]) + kwargs["auth_token"] = "entra-id-bearer-via-http-hook" + headers = _beta_header(_common_betas_for_base_url(normalized_base_url, drop_context_1m_beta=drop_context_1m_beta)) + if headers: + kwargs["default_headers"] = headers - # Strip any trailing /v1 — the Anthropic SDK appends /v1/messages. - normalized_base_url = _normalize_base_url_text(base_url) - if normalized_base_url: - import re as _re - normalized_base_url = _re.sub(r"/v1/?$", "", normalized_base_url.rstrip("/")) - - http_client = build_bearer_http_client(token_provider, timeout=timeout_obj) - - kwargs = { - "timeout": timeout_obj, - "http_client": http_client, - # Delegate retry to hermes's outer loop (honors Retry-After); the SDK - # default max_retries=2 ignores it and double-retries. (#26293) - "max_retries": 0, - # The SDK requires *something* for api_key/auth_token. Our - # event hook overrides Authorization per request so this value - # is never sent. The sentinel string makes accidental leaks - # diagnosable in logs. - "auth_token": "entra-id-bearer-via-http-hook", - } - - if normalized_base_url: - if _is_azure_anthropic_endpoint(normalized_base_url) and "api-version" not in normalized_base_url: - kwargs["base_url"] = normalized_base_url - kwargs["default_query"] = {"api-version": "2025-04-15"} - else: - kwargs["base_url"] = normalized_base_url - - common_betas = _common_betas_for_base_url( - normalized_base_url, - drop_context_1m_beta=drop_context_1m_beta, - ) - if common_betas: - kwargs["default_headers"] = {"anthropic-beta": ",".join(common_betas)} - - client = _anthropic_sdk.Anthropic(**kwargs) + client = sdk.Anthropic(**kwargs) # Same env-inference trap as build_anthropic_client: auth_token-only # construction would otherwise also send ANTHROPIC_API_KEY as X-Api-Key. client.api_key = None @@ -714,40 +506,16 @@ def build_anthropic_client( ): """Create an Anthropic client, auto-detecting setup-tokens vs API keys. - ``api_key`` accepts either: - - * a static ``str`` — the historical contract for all key-based and - OAuth flows. - * a ``Callable[[], str]`` — an Entra ID bearer token provider from - :mod:`agent.azure_identity_adapter`. The Anthropic SDK itself - requires a static string, so when given a callable we construct - a custom ``httpx.Client`` with a request event hook that mints a - fresh JWT per outbound request and rewrites the ``Authorization`` - header. The SDK never sees the callable directly. - - If *timeout* is provided it overrides the default 900s read timeout. The - connect timeout stays at 10s. Callers pass this from the per-provider / - per-model ``request_timeout_seconds`` config so Anthropic-native and - Anthropic-compatible providers respect the same knob as OpenAI-wire - providers. - - ``drop_context_1m_beta=True`` strips ``context-1m-2025-08-07`` from the - client-level ``anthropic-beta`` header. Used by the reactive OAuth retry - path in ``run_agent.py`` when a subscription rejects the beta; leave at - its default on fresh clients so 1M-capable subscriptions keep the - capability. - - Returns an anthropic.Anthropic instance. + ``api_key`` is a static ``str`` (all key-based and OAuth flows) or a + ``Callable[[], str]`` Entra ID bearer provider, which is routed through + :func:`_build_anthropic_client_with_bearer_hook`. ``timeout`` overrides the + 900s read timeout (connect stays 10s) from the per-provider/per-model + ``request_timeout_seconds`` config. ``drop_context_1m_beta`` strips + ``context-1m-2025-08-07`` from the client-level beta header — used by the + reactive OAuth retry in run_agent when a subscription rejects it; fresh + clients keep the default so 1M-capable subscriptions keep the capability. """ - _anthropic_sdk = _get_anthropic_sdk() - if _anthropic_sdk is None: - raise ImportError( - "The 'anthropic' package is required for the Anthropic provider. " - "Install it with: pip install 'anthropic>=0.39.0'" - ) - - # Callable api_key → Entra ID bearer provider path. Delegated to a - # helper so the existing static-key code below stays unchanged. + sdk = _require_sdk("the Anthropic provider") if callable(api_key) and not isinstance(api_key, str): return _build_anthropic_client_with_bearer_hook( api_key, base_url, timeout, @@ -755,151 +523,200 @@ def build_anthropic_client( ) normalize_proxy_env_vars() - - from httpx import Timeout - - normalized_base_url = _normalize_base_url_text(base_url) - if normalized_base_url: - import re as _re - normalized_base_url = _re.sub(r"/v1/?$", "", normalized_base_url.rstrip("/")) - _read_timeout = timeout if (isinstance(timeout, (int, float)) and timeout > 0) else 900.0 - kwargs = { - "timeout": Timeout(timeout=float(_read_timeout), connect=10.0), - # Delegate all rate-limit / 5xx retry to hermes's outer conversation - # loop, which honors Retry-After. The SDK default (max_retries=2) uses - # its own 1-2s backoff that ignores Retry-After and double-retries - # inside our loop — burning request slots against a bucket that won't - # refill for minutes. (#26293) - "max_retries": 0, - } - if normalized_base_url: - # Azure Anthropic endpoints require an ``api-version`` query parameter. - # Pass it via default_query so the SDK appends it to every request URL - # without corrupting the base_url (appending it directly produces - # malformed paths like /anthropic?api-version=.../v1/messages). - if _is_azure_anthropic_endpoint(normalized_base_url) and "api-version" not in normalized_base_url: - kwargs["base_url"] = normalized_base_url.rstrip("/") - kwargs["default_query"] = {"api-version": "2025-04-15"} - else: - kwargs["base_url"] = normalized_base_url - common_betas = _common_betas_for_base_url( - normalized_base_url, - drop_context_1m_beta=drop_context_1m_beta, - ) + normalized_base_url, kwargs = _base_client_kwargs(base_url, timeout) + if "default_query" in kwargs: # historical: this path also strips a stray trailing slash on Azure + kwargs["base_url"] = normalized_base_url.rstrip("/") + common_betas = _common_betas_for_base_url(normalized_base_url, drop_context_1m_beta=drop_context_1m_beta) if _is_kimi_coding_endpoint(base_url): - # Kimi's /coding endpoint requires a non-empty User-Agent to be - # recognized as a valid Coding Agent. Originally we sent - # ``claude-code/0.1.0`` (the minimum that avoided a 403), but the Kimi - # team asked us to identify ourselves properly so they can attribute - # traffic correctly. Send the same attribution header set we send to - # OpenRouter, Vercel AI Gateway, and Fireworks: - # HTTP-Referer + X-Title + HermesAgent User-Agent. + # Kimi's /coding endpoint 403s without a User-Agent; the Kimi team asked + # for proper attribution instead of the ``claude-code/0.1.0`` minimum. kwargs["api_key"] = api_key - kwargs["default_headers"] = { - "HTTP-Referer": "https://hermes-agent.nousresearch.com", - "X-Title": "Hermes Agent", - "User-Agent": f"HermesAgent/{_HERMES_VERSION}", - **( {"anthropic-beta": ",".join(common_betas)} if common_betas else {} ) - } + headers = {**_attribution_headers(), **_beta_header(common_betas)} elif _requires_bearer_auth(normalized_base_url): - # Some Anthropic-compatible providers (e.g. MiniMax) expect the API key in - # Authorization: Bearer *** for regular API keys. Route those endpoints - # through auth_token so the SDK sends Bearer auth instead of x-api-key. - # Check this before OAuth token shape detection because MiniMax secrets do - # not use Anthropic's sk-ant-api prefix and would otherwise be misread as - # Anthropic OAuth/setup tokens. + # MiniMax & co. want the key in Authorization: Bearer. Checked before the + # OAuth shape test: their secrets lack the sk-ant-api prefix and would + # otherwise be misread as Anthropic OAuth/setup tokens. kwargs["auth_token"] = api_key - if common_betas: - kwargs["default_headers"] = {"anthropic-beta": ",".join(common_betas)} + headers = _beta_header(common_betas) elif _is_third_party_anthropic_endpoint(base_url): - # Third-party proxies (Microsoft Foundry, AWS Bedrock, etc.) use their - # own API keys with x-api-key auth. Skip OAuth detection — their keys - # don't follow Anthropic's sk-ant-* prefix convention and would be - # misclassified as OAuth tokens. + # Third-party proxies use their own x-api-key keys; skip OAuth detection + # (their keys don't follow the sk-ant-* convention). kwargs["api_key"] = api_key - if common_betas: - kwargs["default_headers"] = {"anthropic-beta": ",".join(common_betas)} + headers = _beta_header(common_betas) elif _is_oauth_token(api_key): - # OAuth access token / setup-token → Bearer auth + Claude Code identity. - # Anthropic routes OAuth requests based on user-agent and headers; - # without Claude Code's fingerprint, requests get intermittent 500s. - all_betas = common_betas + _OAUTH_ONLY_BETAS + # OAuth/setup-token -> Bearer auth + Claude Code identity. Anthropic + # routes OAuth by user-agent/headers; without the fingerprint, 500s. kwargs["auth_token"] = api_key - kwargs["default_headers"] = { - "anthropic-beta": ",".join(all_betas), + headers = { + **_beta_header(common_betas + _OAUTH_ONLY_BETAS), "user-agent": f"claude-code/{_get_claude_code_version()} (external, cli)", "x-app": "cli", } else: - # Regular API key → x-api-key header + common betas kwargs["api_key"] = api_key - if common_betas: - kwargs["default_headers"] = {"anthropic-beta": ",".join(common_betas)} + headers = _beta_header(common_betas) if _is_opencode_endpoint(base_url): - # OpenCode identifies clients by request headers, like OpenRouter does. - # The OpenAI-wire paths pick these up from profile.default_headers - # (plugins/model-providers/opencode-zen), but the Anthropic Messages - # route builds its client right here and never sees the profile. Merge - # the same set on top of whatever auth branch ran above. - headers = dict(kwargs.get("default_headers") or {}) - headers.setdefault("HTTP-Referer", "https://hermes-agent.nousresearch.com") - headers.setdefault("X-Title", "Hermes Agent") - headers.setdefault("User-Agent", f"HermesAgent/{_HERMES_VERSION}") + # OpenCode identifies clients by request headers (like OpenRouter). The + # OpenAI-wire paths get these from profile.default_headers, but this + # route builds its client here and never sees the profile. + for k, v in _attribution_headers().items(): + headers.setdefault(k, v) + if headers: kwargs["default_headers"] = headers - client = _anthropic_sdk.Anthropic(**kwargs) - # Bearer-only construction leaves ``api_key`` unset, so the SDK fills it - # from ``ANTHROPIC_API_KEY`` (Hermes loads that into the process env from - # ``~/.hermes/.env``). The result is dual auth — - # ``X-Api-Key: sk-ant-…`` *and* ``Authorization: Bearer `` — - # on every Portal / MiniMax / OAuth Messages request. Clear the env-filled - # key whenever we intentionally authenticated via auth_token alone. + client = sdk.Anthropic(**kwargs) + # Bearer-only construction leaves ``api_key`` unset, so the SDK fills it from + # ANTHROPIC_API_KEY (loaded from ~/.hermes/.env) and sends dual auth — + # X-Api-Key *and* Authorization: Bearer — on every Portal/MiniMax/OAuth + # request. Clear it whenever we intentionally authenticated via auth_token. if "auth_token" in kwargs and "api_key" not in kwargs: client.api_key = None return client def build_anthropic_bedrock_client(region: str): - """Create an AnthropicBedrock client for Bedrock Claude models. + """AnthropicBedrock client for Bedrock Claude models (boto3 default credential chain). - Uses the Anthropic SDK's native Bedrock adapter, which provides full - Claude feature parity: prompt caching, thinking budgets, adaptive - thinking, fast mode — features not available via the Converse API. - - Attaches the common Anthropic beta headers as client-level defaults so - that Bedrock-hosted Claude models get the same enhanced features as - native Anthropic. The ``context-1m-2025-08-07`` beta in particular - unlocks the 1M context window for Opus 4.6/4.7 on Bedrock — without - it, Bedrock caps these models at 200K even though the Anthropic API - serves them with 1M natively. - - Auth uses the boto3 default credential chain (IAM roles, SSO, env vars). + The SDK's native Bedrock adapter gives full Claude feature parity (prompt + caching, thinking budgets, adaptive thinking, fast mode) that Converse + lacks. The common betas plus ``context-1m-2025-08-07`` are attached: without + the latter Bedrock caps Opus 4.6/4.7 at 200K instead of 1M. """ - _anthropic_sdk = _get_anthropic_sdk() - if _anthropic_sdk is None: - raise ImportError( - "The 'anthropic' package is required for the Bedrock provider. " - "Install it with: pip install 'anthropic>=0.39.0'" - ) - if not hasattr(_anthropic_sdk, "AnthropicBedrock"): + sdk = _require_sdk("the Bedrock provider") + if not hasattr(sdk, "AnthropicBedrock"): raise ImportError( "anthropic.AnthropicBedrock not available. " "Upgrade with: pip install 'anthropic>=0.39.0'" ) - from httpx import Timeout - - return _anthropic_sdk.AnthropicBedrock( + return sdk.AnthropicBedrock( aws_region=region, - timeout=Timeout(timeout=900.0, connect=10.0), - # Delegate retry to hermes's outer loop (honors Retry-After); the SDK - # default max_retries=2 ignores it and double-retries. (#26293) - max_retries=0, - default_headers={"anthropic-beta": ",".join([*_COMMON_BETAS, _CONTEXT_1M_BETA])}, + timeout=_client_timeout(None), + max_retries=0, # retry belongs to hermes's outer loop (honors Retry-After) + default_headers=_beta_header([*_COMMON_BETAS, _CONTEXT_1M_BETA]), ) +def _normalize_to_mcp_wire(name: str) -> str: + """OAuth wire form of a tool name (no aliasing): ``mcp__<...>``. + + Anthropic's OAuth billing classifier treats a single-underscore ``mcp_`` + tool name as a third-party-app fingerprint (HTTP 400 "Third-party apps now + draw from extra usage"); ``mcp__foo`` is accepted. Both bare Hermes tools + (``read_file``) and native MCP tools registered as ``mcp__`` + must land on the double-underscore form — the latter was the gap a bare + prefix swap left open. normalize_response reverses both via registry lookup. + """ + if name.startswith("mcp__"): + return name # already correct, don't double-prefix + if name.startswith("mcp_"): + return "mcp__" + name[len("mcp_"):] + return _MCP_TOOL_PREFIX + name + + +def _oauth_wire_namer(anthropic_tools: List[Dict[str, Any]]): + """Return ``name -> OAuth wire name`` for this request's tool set. + + An alias must never collide with a wire name owned by a non-alias tool: two + identical tool names in one request is a hard 400, strictly worse than the + bug being fixed. Mirrors normalize_response's "registered tool wins" so + outbound and inbound agree on who owns a contested name. + """ + claimed = { + _normalize_to_mcp_wire(tool["name"]) + for tool in (anthropic_tools or []) + if isinstance(tool.get("name"), str) and tool["name"] not in _OAUTH_TOOL_NAME_ALIASES + } + + def to_wire(name: str) -> str: + if name in _OAUTH_TOOL_NAME_ALIASES: + aliased = _OAUTH_TOOL_NAME_ALIASES[name] + if _MCP_TOOL_PREFIX + aliased not in claimed: + name = aliased + return _normalize_to_mcp_wire(name) + + return to_wire + + +_OAUTH_SYSTEM_REPLACEMENTS = ( + ("Hermes Agent", "Claude Code"), + ("Hermes agent", "Claude Code"), + ("hermes-agent", "claude-code"), + ("Nous Research", "Anthropic"), +) + + +def _apply_claude_code_identity(system, anthropic_tools, anthropic_messages, to_wire): + """OAuth transforms: Claude Code system prefix, product-name sanitizing (avoids + server-side content filters), tool/description aliasing, and the same tool + renames on replayed tool_use blocks so history matches ``tools[]``. Returns + the new ``system``; tools and messages are mutated in place. + """ + cc_block = {"type": "text", "text": _CLAUDE_CODE_SYSTEM_PREFIX} + if isinstance(system, list): + system = [cc_block] + system + elif isinstance(system, str) and system: + system = [cc_block, {"type": "text", "text": system}] + else: + system = [cc_block] + for block in system: + if isinstance(block, dict) and block.get("type") == "text": + text = block.get("text", "") + for old, new in _OAUTH_SYSTEM_REPLACEMENTS: + text = text.replace(old, new) + block["text"] = _apply_oauth_prose_aliases(text) + + for tool in anthropic_tools or []: + if "name" in tool: + tool["name"] = to_wire(tool["name"]) + description = tool.get("description") + if isinstance(description, str): + tool["description"] = _apply_oauth_prose_aliases(description) # prose-safe aliases only + + for msg in anthropic_messages: + content = msg.get("content") + if isinstance(content, list): + for block in content: + if isinstance(block, dict) and block.get("type") == "tool_use" and "name" in block: + block["name"] = to_wire(block["name"]) # tool_result pairs by id, not name + return system + + +def _thinking_kwargs(reasoning_config: Dict[str, Any], model: str, effective_max_tokens: int) -> Dict[str, Any]: + """Map ``reasoning_config`` to Anthropic thinking kwargs. + + Adaptive models (Claude 4.6+, Kimi/Moonshot — the replay-validation 400s + that once motivated dropping the param for Kimi no longer occur) get + ``thinking.type=adaptive`` + ``output_config.effort``; older models and + manual-only compat endpoints (MiniMax) get budget_tokens. Haiku has no + extended thinking. On 4.7+ ``thinking.display`` defaults to "omitted", + hiding the reasoning Hermes shows in its CLI, so "summarized" is requested + to keep the activity feed populated. + """ + if reasoning_config.get("enabled") is False: + # Adaptive models think by DEFAULT, so omitting the parameter is not a + # disable — the user silently keeps paying. Mandatory-thinking models + # 400 on the disable, so they keep the omission: a silently-ignored + # disable beats a dead turn. + return {"thinking": {"type": "disabled"}} if _accepts_thinking_disable(model) else {} + if "haiku" in model.lower(): + return {} + effort = str(reasoning_config.get("effort", "medium")).lower() + budget = THINKING_BUDGET.get(effort, 8000) + if _supports_adaptive_thinking(model): + adaptive_effort = ADAPTIVE_EFFORT_MAP.get(effort, "medium") + if adaptive_effort == "xhigh" and not _supports_xhigh_effort(model): + adaptive_effort = "max" + return { + "thinking": {"type": "adaptive", "display": "summarized"}, + "output_config": {"effort": adaptive_effort}, + } + return { + "thinking": {"type": "enabled", "budget_tokens": budget}, + "temperature": 1, # required when thinking is enabled on older models + "max_tokens": max(effective_max_tokens, budget + 4096), + } def build_anthropic_kwargs( @@ -918,41 +735,19 @@ def build_anthropic_kwargs( ) -> Dict[str, Any]: """Build kwargs for anthropic.messages.create(). - Naming note — two distinct concepts, easily confused: - max_tokens = OUTPUT token cap for a single response. - Anthropic's API calls this "max_tokens" but it only - limits the *output*. Anthropic's own native SDK - renamed it "max_output_tokens" for clarity. - context_length = TOTAL context window (input tokens + output tokens). - The API enforces: input_tokens + max_tokens ≤ context_length. - Stored on the ContextCompressor; reduced on overflow errors. + Two easily confused concepts: ``max_tokens`` is the OUTPUT cap for one + response (Anthropic's name for it; their native SDK says max_output_tokens); + ``context_length`` is the TOTAL window (input + output), enforced as + ``input_tokens + max_tokens <= context_length``. ``max_tokens=None`` uses the + model's native output ceiling; if that exceeds ``context_length`` (small + local endpoints) it is clamped to ``context_length - 1``. The clamp ignores + prompt size — callers must catch "max_tokens too large given prompt" and + retry smaller (parse_available_output_tokens_from_error). - When *max_tokens* is None the model's native output ceiling is used - (e.g. 128K for Opus 4.6, 64K for Sonnet 4.6). - - When *context_length* is provided and the model's native output ceiling - exceeds it (e.g. a local endpoint with an 8K window), the output cap is - clamped to context_length − 1. This only kicks in for unusually small - context windows; for full-size models the native output cap is always - smaller than the context window so no clamping happens. - NOTE: this clamping does not account for prompt size — if the prompt is - large, Anthropic may still reject the request. The caller must detect - "max_tokens too large given prompt" errors and retry with a smaller cap - (see parse_available_output_tokens_from_error + _ephemeral_max_output_tokens). - - When *is_oauth* is True, applies Claude Code compatibility transforms: - system prompt prefix, tool name prefixing, and prompt sanitization. - - When *preserve_dots* is True, model name dots are not converted to hyphens - (for Alibaba/DashScope anthropic-compatible endpoints: qwen3.5-plus). - - When *base_url* points to a third-party Anthropic-compatible endpoint, - thinking block signatures are stripped (they are Anthropic-proprietary). - - When *fast_mode* is True, adds ``extra_body["speed"] = "fast"`` and the - fast-mode beta header for ~2.5x faster output throughput on Opus 4.8 / - Opus 5. Currently only supported on native Anthropic endpoints (not - third-party compatible ones). + ``is_oauth`` applies Claude Code compatibility transforms; ``preserve_dots`` + keeps model-name dots (DashScope: qwen3.5-plus); a third-party ``base_url`` + strips thinking signatures; ``fast_mode`` adds ``extra_body.speed="fast"`` + plus the fast-mode beta on native Anthropic only. """ system, anthropic_messages = convert_messages_to_anthropic( messages, base_url=base_url, model=model @@ -960,241 +755,68 @@ def build_anthropic_kwargs( anthropic_tools = convert_tools_to_anthropic(tools) if tools else [] # Nous Portal routes on its own catalog ids (``anthropic/claude-opus-4.8``); - # normalizing to the bare Anthropic slug would make the model unresolvable - # there. Skipping the call preserves the prefix AND the dots, so - # ``preserve_dots`` stays irrelevant for Portal. + # normalizing would make the model unresolvable there (prefix AND dots kept). if not _is_nous_portal_endpoint(base_url): model = normalize_model_name(model, preserve_dots=preserve_dots) - # effective_max_tokens = output cap for this call (≠ total context window) - # Use the resolver helper so non-positive values (negative ints, - # fractional floats, NaN, non-numeric) fail locally with a clear error - # rather than 400-ing at the Anthropic API. See openclaw/openclaw#66664. + # Non-positive/non-finite values fail locally instead of 400-ing upstream. effective_max_tokens = _resolve_anthropic_messages_max_tokens( max_tokens, model, context_length=context_length ) - - # Clamp output cap to fit inside the total context window. - # Only matters for small custom endpoints where context_length < native - # output ceiling. For standard Anthropic models context_length (e.g. - # 200K) is always larger than the output ceiling (e.g. 128K), so this - # branch is not taken. if context_length and effective_max_tokens > context_length: effective_max_tokens = max(context_length - 1, 1) - # ── OAuth: Claude Code identity ────────────────────────────────── - if is_oauth: - # 1. Prepend Claude Code system prompt identity - cc_block = {"type": "text", "text": _CLAUDE_CODE_SYSTEM_PREFIX} - if isinstance(system, list): - system = [cc_block] + system - elif isinstance(system, str) and system: - system = [cc_block, {"type": "text", "text": system}] - else: - system = [cc_block] - - # 2. Sanitize system prompt — replace product name references - # to avoid Anthropic's server-side content filters. - for block in system: - if isinstance(block, dict) and block.get("type") == "text": - text = block.get("text", "") - text = text.replace("Hermes Agent", "Claude Code") - text = text.replace("Hermes agent", "Claude Code") - text = text.replace("hermes-agent", "claude-code") - text = text.replace("Nous Research", "Anthropic") - text = _apply_oauth_prose_aliases(text) - block["text"] = text - - # 3. Normalize tool names so NOTHING goes on the OAuth wire with a - # single-underscore ``mcp_`` prefix. Anthropic's subscription/OAuth - # billing classifier treats a single-underscore ``mcp_`` tool name as - # a third-party-app fingerprint and rejects the request with HTTP 400 - # "Third-party apps now draw from extra usage, not plan limits" - # (verified empirically: a single ``mcp_foo`` tool flips a request - # from plan-billing to the extra-usage lane; ``mcp__foo`` is accepted). - # - # Two cases, both must land on the double-underscore ``mcp__`` form: - # a) bare Hermes-native tools (``read_file``) -> ``mcp__read_file`` - # b) native MCP server tools registered under their full - # single-underscore ``mcp__`` name - # (``mcp_linear_get_issue``) -> ``mcp__linear_get_issue`` - # Case (b) is the gap that the bare ``mcp_``->``mcp__`` constant swap - # left open: those tools were *skipped* and stayed single-underscore, - # so any session with an MCP server configured still tripped the - # classifier. normalize_response reverses both forms via registry - # lookup so the dispatcher still sees the original name. GH-25255. - # Wire names owned by tools that are NOT alias sources. An alias must - # never collide with one: two identical tool names in a single - # request is a hard 400 from Anthropic, strictly worse than the bug - # being fixed. Mirrors the "registered tool wins" precedence in - # normalize_response so outbound and inbound agree on who owns a - # contested name. - def _normalize_to_mcp_wire(name: str) -> str: - """OAuth wire form of a tool name (no aliasing): mcp__<...>.""" - if name.startswith("mcp__"): - return name # already correct, don't double-prefix - if name.startswith("mcp_"): - # single-underscore native MCP tool -> promote to double - return "mcp__" + name[len("mcp_"):] - return _MCP_TOOL_PREFIX + name # bare name -> mcp__ - - _claimed_wire_names = { - _normalize_to_mcp_wire(tool["name"]) - for tool in (anthropic_tools or []) - if isinstance(tool.get("name"), str) - and tool["name"] not in _OAUTH_TOOL_NAME_ALIASES - } - - def _to_oauth_wire_name(name: str) -> str: - if name in _OAUTH_TOOL_NAME_ALIASES: - aliased = _OAUTH_TOOL_NAME_ALIASES[name] - if _MCP_TOOL_PREFIX + aliased not in _claimed_wire_names: - name = aliased - return _normalize_to_mcp_wire(name) - - if anthropic_tools: - for tool in anthropic_tools: - if "name" in tool: - tool["name"] = _to_oauth_wire_name(tool["name"]) - description = tool.get("description") - if isinstance(description, str): - # Prose-safe aliases only — see _OAUTH_PROSE_ALIAS_NAMES. - tool["description"] = _apply_oauth_prose_aliases(description) - - # 4. Apply the same normalization to tool names in message history - # (tool_use blocks) so replayed turns match the wire names above. - for msg in anthropic_messages: - content = msg.get("content") - if isinstance(content, list): - for block in content: - if isinstance(block, dict): - if block.get("type") == "tool_use" and "name" in block: - block["name"] = _to_oauth_wire_name(block["name"]) - elif block.get("type") == "tool_result" and "tool_use_id" in block: - pass # tool_result uses ID, not name + to_wire = _oauth_wire_namer(anthropic_tools) if is_oauth else None + if to_wire: + system = _apply_claude_code_identity(system, anthropic_tools, anthropic_messages, to_wire) kwargs: Dict[str, Any] = { "model": model, "messages": anthropic_messages, "max_tokens": effective_max_tokens, } - if system: kwargs["system"] = system if anthropic_tools: kwargs["tools"] = anthropic_tools - # Map OpenAI tool_choice to Anthropic format if tool_choice == "auto" or tool_choice is None: kwargs["tool_choice"] = {"type": "auto"} elif tool_choice == "required": kwargs["tool_choice"] = {"type": "any"} elif tool_choice == "none": - # Anthropic has no tool_choice "none" — omit tools entirely to prevent use - kwargs.pop("tools", None) + kwargs.pop("tools", None) # no Anthropic "none" — omit tools to prevent use elif isinstance(tool_choice, str): - # Specific tool name. Under OAuth every tool on the wire is - # mcp__-prefixed and/or alias-renamed (see _to_oauth_wire_name - # above) — route the forced name through the same normalizer so - # tool_choice always matches the corresponding tools[] entry. - # Left un-normalized, a forced ``session_search``/``memory`` - # choice would (a) still carry the literal trigger string onto - # the wire, defeating the alias, and (b) reference a tool name - # that no longer exists in ``tools[]``, which Anthropic rejects. - wire_tool_choice = tool_choice - if is_oauth: - wire_tool_choice = _to_oauth_wire_name(tool_choice) - kwargs["tool_choice"] = {"type": "tool", "name": wire_tool_choice} + # Under OAuth every tools[] entry is mcp__-prefixed/aliased; the forced + # name must go through the same normalizer or it (a) leaks the literal + # trigger string and (b) names a tool that no longer exists -> 400. + kwargs["tool_choice"] = {"type": "tool", "name": to_wire(tool_choice) if to_wire else tool_choice} - # Map reasoning_config to Anthropic's thinking parameter. - # Claude 4.6+ models use adaptive thinking + output_config.effort. - # Older models use manual thinking with budget_tokens. - # MiniMax Anthropic-compat endpoints support thinking (manual mode only, - # not adaptive). Haiku does NOT support extended thinking — skip entirely. - # - # Kimi / Moonshot models also use adaptive thinking: their - # Anthropic-compatible endpoints (api.moonshot.cn/anthropic, - # api.kimi.com/coding) accept ``thinking.type="adaptive"`` + - # ``output_config.effort``, and the replay-validation 400s that - # originally motivated dropping the parameter (#13848) no longer - # occur. (Kimi on chat_completions enables thinking via extra_body - # in the ChatCompletionsTransport — see #13503.) - # - # On 4.7+ the `thinking.display` field defaults to "omitted", which - # silently hides reasoning text that Hermes surfaces in its CLI. We - # request "summarized" so the reasoning blocks stay populated — matching - # 4.6 behavior and preserving the activity-feed UX during long tool runs. if reasoning_config and isinstance(reasoning_config, dict): - if reasoning_config.get("enabled") is False: - # "Thinking off". Adaptive models think by DEFAULT, so omitting the - # parameter is not a disable — it silently leaves thinking on and - # the user keeps paying for it. Send the disable explicitly. - # Mandatory-thinking models reject it with a 400, so they keep the - # omission: a silently-ignored disable beats a dead turn. - if _accepts_thinking_disable(model): - kwargs["thinking"] = {"type": "disabled"} - elif "haiku" not in model.lower(): - effort = str(reasoning_config.get("effort", "medium")).lower() - budget = THINKING_BUDGET.get(effort, 8000) - if _supports_adaptive_thinking(model): - kwargs["thinking"] = { - "type": "adaptive", - "display": "summarized", - } - adaptive_effort = ADAPTIVE_EFFORT_MAP.get(effort, "medium") - # Downgrade xhigh→max on models that don't list xhigh as a - # supported level (Opus/Sonnet 4.6). Opus 4.7+ keeps xhigh. - if adaptive_effort == "xhigh" and not _supports_xhigh_effort(model): - adaptive_effort = "max" - kwargs["output_config"] = { - "effort": adaptive_effort, - } - else: - kwargs["thinking"] = {"type": "enabled", "budget_tokens": budget} - # Anthropic requires temperature=1 when thinking is enabled on older models - kwargs["temperature"] = 1 - kwargs["max_tokens"] = max(effective_max_tokens, budget + 4096) + kwargs.update(_thinking_kwargs(reasoning_config, model, effective_max_tokens)) - # ── Strip sampling params on 4.7+ ───────────────────────────────── - # Opus 4.7 rejects any non-default temperature/top_p/top_k with a 400. - # Callers (auxiliary_client, etc.) may set these for older models; - # drop them here as a safety net so upstream 4.6 → 4.7 migrations - # don't require coordinated edits everywhere. + # Safety net so upstream 4.6 -> 4.7 migrations don't need coordinated edits + # everywhere callers (auxiliary_client, ...) set sampling params. if _forbids_sampling_params(model): for _sampling_key in ("temperature", "top_p", "top_k"): kwargs.pop(_sampling_key, None) - # ── Fast mode (Opus 4.8 / Opus 5) ──────────────────────────────── - # Adds extra_body.speed="fast" + the fast-mode beta header for ~2.5x - # output speed. Per Anthropic docs the speed param is supported on - # Opus 4.8 and Opus 5 (research preview); Opus 4.7 400s on it and - # Opus 4.6 silently ignores it (standard speed, standard billing). - # Only for native Anthropic endpoints — third-party providers would - # reject the unknown beta header and speed parameter, and Anthropic - # itself scopes fast mode to the Claude API (not Bedrock/Vertex/ - # Foundry). - if ( - fast_mode - and not _is_third_party_anthropic_endpoint(base_url) - and _supports_fast_mode(model) - ): + # Fast mode: native Anthropic only — third-party providers reject the + # unknown beta/param and Anthropic scopes it to the Claude API (not + # Bedrock/Vertex/Foundry). Per-request extra_headers OVERRIDE the + # client-level anthropic-beta header, so rebuild the full beta list. + if fast_mode and not _is_third_party_anthropic_endpoint(base_url) and _supports_fast_mode(model): kwargs.setdefault("extra_body", {})["speed"] = "fast" - # Build extra_headers with ALL applicable betas (the per-request - # extra_headers override the client-level anthropic-beta header). - betas = list(_common_betas_for_base_url( - base_url, - drop_context_1m_beta=drop_context_1m_beta, - )) + betas = list(_common_betas_for_base_url(base_url, drop_context_1m_beta=drop_context_1m_beta)) if is_oauth: betas.extend(_OAUTH_ONLY_BETAS) betas.append(_FAST_MODE_BETA) - kwargs["extra_headers"] = {"anthropic-beta": ",".join(betas)} + kwargs["extra_headers"] = _beta_header(betas) return kwargs -# Keys that belong exclusively to the OpenAI Responses / Codex API shape. -# The Anthropic Messages SDK (``messages.create()`` / ``messages.stream()``) -# raises ``TypeError: ... got an unexpected keyword argument`` on any of them. +# Keys exclusive to the OpenAI Responses / Codex shape; the Messages SDK raises +# ``TypeError: ... unexpected keyword argument`` on any of them. _RESPONSES_ONLY_KWARGS = frozenset( {"instructions", "input", "store", "parallel_tool_calls"} ) @@ -1203,24 +825,18 @@ _RESPONSES_ONLY_KWARGS = frozenset( def sanitize_anthropic_kwargs(api_kwargs: Any, *, log_prefix: str = "") -> Any: """Drop Responses-API-only keys before an Anthropic Messages SDK call. - Defensive boundary guard for #31673: under rare api_mode-flip races - (e.g. a concurrent auxiliary call mutating a shared agent between the - kwargs build and the stream dispatch), a Responses-shaped payload - carrying ``instructions=`` can reach ``messages.stream()`` / - ``messages.create()``. The Anthropic SDK rejects it with a - non-retryable ``TypeError`` that nukes the whole turn and propagates - the entire fallback chain. - - Mutates ``api_kwargs`` in place and returns it. When a foreign key is - present we log a WARNING so the underlying race stays visible in the - wild instead of being silently papered over. + Boundary guard for api_mode-flip races (a concurrent auxiliary call mutating + a shared agent between kwargs build and dispatch): a Responses-shaped payload + reaching ``messages.stream()`` dies with a non-retryable TypeError that takes + the whole turn and fallback chain with it. Mutates and returns + ``api_kwargs``; logs a WARNING so the race stays visible. """ if not isinstance(api_kwargs, dict): return api_kwargs leaked = _RESPONSES_ONLY_KWARGS.intersection(api_kwargs) if leaked: for _key in leaked: - api_kwargs.pop(_key, None) + del api_kwargs[_key] logger.warning( "%sStripped Responses-only kwarg(s) %s from an Anthropic Messages " "call (api_mode flip race — see #31673). The call will proceed; " @@ -1233,7 +849,7 @@ def sanitize_anthropic_kwargs(api_kwargs: Any, *, log_prefix: str = "") -> Any: def _is_stream_unavailable_error(exc: Exception) -> bool: - """Return True when an Anthropic stream call should fall back to create().""" + """True when an Anthropic stream call should fall back to create().""" err_lower = str(exc).lower() if "stream" in err_lower and "not supported" in err_lower: return True @@ -1255,26 +871,17 @@ def create_anthropic_message( ) -> Any: """Create an Anthropic message, aggregating via stream when available. - Some Anthropic-compatible gateways are SSE-only: they ignore non-streaming - requests and return ``text/event-stream`` even for ``messages.create()``. - The SDK can surface that as raw text, so callers that expect a Message then - crash on ``.content``. Prefer ``messages.stream().get_final_message()`` to - match the main turn path, falling back to ``create()`` only for providers - that explicitly do not support streaming, such as restricted Bedrock roles. + Some Anthropic-compatible gateways are SSE-only and answer ``create()`` with + ``text/event-stream``, which the SDK surfaces as raw text (callers then + crash on ``.content``). So prefer ``messages.stream().get_final_message()`` + like the main turn path, falling back to ``create()`` only for providers + that explicitly don't support streaming (restricted Bedrock roles). - ``on_stream_event``: optional callable invoked once per streamed event - (best-effort, exceptions swallowed). Lets callers report forward progress - to liveness watchdogs — e.g. the auxiliary compression path ticking its - progress hook so a slow-but-generating summary model isn't treated as - hung. Only fires on the streaming path; the ``create()`` fallback has no - events to report. - - ``on_response``: optional callable invoked once with the underlying httpx - response before the message is aggregated (best-effort, exceptions - swallowed). Response *headers* carry out-of-band provider state that the - parsed ``Message`` drops — Nous Portal's ``x-nous-credits-*`` balance family - in particular. Only fires on the streaming path, which is the one the main - turn loop takes. + Both callbacks are best-effort (exceptions swallowed) and fire only on the + streaming path. ``on_stream_event(event)`` lets liveness watchdogs see + forward progress (e.g. a slow compression summary is not "hung"); + ``on_response(httpx_response)`` exposes headers the parsed Message drops + (Nous Portal's ``x-nous-credits-*`` balance family). """ sanitize_anthropic_kwargs(api_kwargs, log_prefix=log_prefix) @@ -1289,29 +896,20 @@ def create_anthropic_message( try: on_response(getattr(stream, "response", None)) except Exception: - logger.debug( - "%son_response callback failed", - log_prefix, exc_info=True, - ) + logger.debug("%son_response callback failed", log_prefix, exc_info=True) if callable(on_stream_event): - # Consume the event stream manually so each event can - # tick the caller's progress callback; get_final_message - # then returns the accumulated snapshot. + # Consume manually so each event ticks the progress callback; + # get_final_message then returns the accumulated snapshot. for _event in stream: try: on_stream_event(_event) except TimeoutError: - # The callback is the caller's deadline seam - # (#99692: the host waiting on this summary has - # already given up). Abandon the stream — the - # ``with`` closes it — instead of streaming an - # answer nobody will read. + # The callback is the caller's deadline seam: the host + # has given up, so abandon the stream (``with`` closes + # it) instead of streaming an answer nobody will read. raise except Exception: - logger.debug( - "%son_stream_event callback failed", - log_prefix, exc_info=True, - ) + logger.debug("%son_stream_event callback failed", log_prefix, exc_info=True) return stream.get_final_message() except TimeoutError: raise diff --git a/agent/anthropic_endpoints.py b/agent/anthropic_endpoints.py index 3cf3e10c62..a9052567d2 100644 --- a/agent/anthropic_endpoints.py +++ b/agent/anthropic_endpoints.py @@ -1,71 +1,58 @@ """Endpoint-family detection for Anthropic-compatible base URLs. -Hermes talks to a dozen services that speak the Anthropic Messages API but -differ in auth style, accepted beta headers, and request quirks: MiniMax, -Kimi/Moonshot, DeepSeek, OpenCode, Azure AI Foundry, the Nous portal, Bedrock. -Every one of those differences is decided by inspecting the configured base -URL, so the predicates live together here instead of being scattered through -client construction and message conversion. +A dozen services speak the Anthropic Messages API but differ in auth style, +accepted beta headers, and request quirks (MiniMax, Kimi/Moonshot, DeepSeek, +OpenCode, Azure AI Foundry, Nous Portal, Bedrock). Every such difference is +decided from the configured base URL, so the predicates live together here. -Pure functions over a base-URL string - no I/O, no SDK, no credentials - which -is what lets both ``agent/anthropic_adapter.py`` and -``agent/anthropic_message_convert.py`` depend on this module without a cycle. - -``agent.anthropic_adapter`` re-exports every name below. +Pure functions over a base-URL string - no I/O, no SDK, no credentials - so +both ``agent/anthropic_adapter.py`` and ``agent/anthropic_message_convert.py`` +can depend on this module without a cycle. ``agent.anthropic_adapter`` +re-exports every name below. """ from urllib.parse import urlparse from utils import base_url_host_matches, base_url_hostname +_MINIMAX_ANTHROPIC_PREFIXES = ("https://api.minimax.io/anthropic", "https://api.minimaxi.com/anthropic") + def _normalize_base_url_text(base_url) -> str: - """Normalize SDK/base transport URL values to a plain string for inspection. - - Some client objects expose ``base_url`` as an ``httpx.URL`` instead of a raw - string. Provider/auth detection should accept either shape. - """ + """Coerce a base URL (str or ``httpx.URL``) to a stripped string; "" when falsy.""" if not base_url: return "" return str(base_url).strip() -def _is_third_party_anthropic_endpoint(base_url: str | None) -> bool: - """Return True for non-Anthropic endpoints using the Anthropic Messages API. +def _normalized_lower(base_url) -> str: + """``_normalize_base_url_text`` + rstrip("/") + lower(), the shape most predicates match on.""" + return _normalize_base_url_text(base_url).rstrip("/").lower() - Third-party proxies (Microsoft Foundry, AWS Bedrock, self-hosted) authenticate - with their own API keys via x-api-key, not Anthropic OAuth tokens. OAuth - detection should be skipped for these endpoints. + +def _is_third_party_anthropic_endpoint(base_url: str | None) -> bool: + """True for any non-anthropic.com endpoint (own API keys via x-api-key; skip OAuth detection). + + No base_url means the direct Anthropic API. """ - normalized = _normalize_base_url_text(base_url) - if not normalized: - return False # No base_url = direct Anthropic API - normalized = normalized.rstrip("/").lower() - if "anthropic.com" in normalized: - return False # Direct Anthropic API — OAuth applies - return True # Any other endpoint is a third-party proxy + normalized = _normalized_lower(base_url) + return bool(normalized) and "anthropic.com" not in normalized def _is_kimi_coding_endpoint(base_url: str | None) -> bool: - """Return True for Kimi's /coding endpoint that requires claude-code UA.""" - normalized = _normalize_base_url_text(base_url) - if not normalized: - return False - return normalized.rstrip("/").lower().startswith("https://api.kimi.com/coding") + """True for Kimi's /coding endpoint, which requires a claude-code User-Agent.""" + return _normalized_lower(base_url).startswith("https://api.kimi.com/coding") def _is_opencode_endpoint(base_url: str | None) -> bool: - """Return True for OpenCode's Zen/Go relay (opencode.ai).""" + """True for OpenCode's Zen/Go relay (opencode.ai).""" return base_url_host_matches(base_url or "", "opencode.ai") -# Model-name prefixes that identify the Kimi / Moonshot family. Covers -# - official slugs: ``kimi-k2.5``, ``kimi_thinking``, ``moonshot-v1-8k`` -# - common release lines: ``k1.5-...``, ``k2-thinking``, ``k25-...``, ``k2.5-...``, -# and the bare Coding Plan slug ``k3`` (plus ``k3.x``/``k3-...`` variants) -# Matched case-insensitively against the post-``normalize_model_name`` form, -# so a caller's ``provider/vendor/model`` slug is handled the same as a -# bare name. +# Model-name prefixes identifying the Kimi / Moonshot family: official slugs +# (``kimi-k2.5``, ``kimi_thinking``, ``moonshot-v1-8k``) and release lines +# (``k1.5-…``, ``k2-thinking``, ``k25-…``, ``k3.x``/``k3-…``). Matched +# case-insensitively after stripping any ``vendor/`` prefix. _KIMI_FAMILY_MODEL_PREFIXES = ( "kimi-", "kimi_", "moonshot-", "moonshot_", @@ -75,9 +62,8 @@ _KIMI_FAMILY_MODEL_PREFIXES = ( "k3.", "k3-", ) -# Bare release slugs with no separator suffix (Kimi Coding Plan serves K3 -# as the exact slug ``k3``). Kept exact-match so unrelated model names that -# merely start with the same characters don't get misclassified. +# Bare release slugs with no separator suffix (Kimi Coding Plan serves K3 as +# exactly ``k3``). Exact-match so unrelated names sharing the prefix don't match. _KIMI_FAMILY_EXACT_SLUGS = frozenset({"k3"}) @@ -87,84 +73,48 @@ def _model_name_is_kimi_family(model: str | None) -> bool: m = model.strip().lower() if not m: return False - # Strip vendor prefix (e.g. ``moonshotai/kimi-k2.5`` → ``kimi-k2.5``) - if "/" in m: + if "/" in m: # ``moonshotai/kimi-k2.5`` -> ``kimi-k2.5`` m = m.rsplit("/", 1)[-1] - if m in _KIMI_FAMILY_EXACT_SLUGS: - return True - return m.startswith(_KIMI_FAMILY_MODEL_PREFIXES) + return m in _KIMI_FAMILY_EXACT_SLUGS or m.startswith(_KIMI_FAMILY_MODEL_PREFIXES) def _is_kimi_family_endpoint(base_url: str | None, model: str | None = None) -> bool: - """Return True for any Kimi / Moonshot Anthropic-Messages-speaking endpoint. + """True for any Kimi / Moonshot Anthropic-Messages endpoint. - Broader than ``_is_kimi_coding_endpoint`` — matches: - - - Kimi's official ``/coding`` URL (legacy check, preserved) - - Any ``api.kimi.com`` / ``moonshot.ai`` / ``moonshot.cn`` host - - Custom or proxied endpoints whose *model* name is in the Kimi / Moonshot - family (``kimi-*``, ``moonshot-*``, ``k1.*``, ``k2.*``, …). Users with - ``api_mode: anthropic_messages`` on a private gateway fronting Kimi - fall into this branch — the upstream still enforces Kimi's thinking - semantics (reasoning_content required on every replayed tool-call - message) regardless of the gateway's hostname. - - Used to decide whether to drop Anthropic's ``thinking`` kwarg and to - preserve unsigned reasoning_content-derived thinking blocks on replay. - See hermes-agent#13848, #17057. + Broader than ``_is_kimi_coding_endpoint``: also matches any api.kimi.com / + moonshot.ai / moonshot.cn host, and any endpoint (e.g. a private gateway) + whose *model* is in the Kimi family — the upstream still enforces Kimi's + thinking semantics regardless of hostname. Decides whether unsigned + reasoning_content-derived thinking blocks are preserved on replay. """ if _is_kimi_coding_endpoint(base_url): return True - for _domain in ("api.kimi.com", "moonshot.ai", "moonshot.cn"): - if base_url_host_matches(base_url or "", _domain): - return True - if _model_name_is_kimi_family(model): + if any(base_url_host_matches(base_url or "", d) for d in ("api.kimi.com", "moonshot.ai", "moonshot.cn")): return True - return False + return _model_name_is_kimi_family(model) def _is_deepseek_anthropic_endpoint(base_url: str | None) -> bool: - """Return True for DeepSeek's Anthropic-compatible endpoint. + """True for DeepSeek's ``/anthropic`` route. - DeepSeek's ``/anthropic`` route speaks the Anthropic Messages protocol - but, when thinking mode is enabled, requires the ``thinking`` blocks - from prior assistant turns to round-trip on subsequent requests — the - generic third-party path strips them and triggers HTTP 400:: - - The content[].thinking in the thinking mode must be passed back - to the API. - - Per DeepSeek's published compatibility matrix the blocks are unsigned - (no Anthropic-proprietary signature, no ``redacted_thinking`` support), - so this endpoint is handled with the same strip-signed / keep-unsigned - policy used for Kimi's ``/coding`` endpoint. The match is pinned to - the ``/anthropic`` path so the OpenAI-compatible ``api.deepseek.com`` - base URL (which never reaches this adapter) is not misclassified. - See hermes-agent#16748. + In thinking mode DeepSeek requires prior-turn ``thinking`` blocks to round-trip + ("The content[].thinking in the thinking mode must be passed back to the API"), + while the generic third-party path strips them. Its blocks are unsigned, so it + gets the same strip-signed / keep-unsigned policy as Kimi. Pinned to the + ``/anthropic`` path so the OpenAI-compatible base URL is not misclassified. """ if not base_url_host_matches(base_url or "", "api.deepseek.com"): return False - normalized = _normalize_base_url_text(base_url) - if not normalized: - return False - return "/anthropic" in normalized.rstrip("/").lower() + return "/anthropic" in _normalized_lower(base_url) def _is_nous_portal_endpoint(base_url: str | None) -> bool: - """Return True for Nous Portal's Anthropic Messages route. + """True for Nous Portal's Anthropic Messages route (Bearer JWT, verbatim catalog + ids, native thinking-signature replay). - Portal serves its ``anthropic/*`` catalog natively at - ``https://inference-api.nousresearch.com/v1/messages``. Portal-specific - behaviours key off this: Bearer JWT auth, verbatim catalog model ids, - and native thinking-signature replay. - - Trusted hosts only: - - 1. Prod hostname ``inference-api.nousresearch.com`` - 2. The operator-set ``NOUS_INFERENCE_BASE_URL`` hostname (staging/preview) - - Lookalikes such as ``inference-api.nousresearch.com.attacker.test`` are - rejected (hostname match, not substring). + Trusted hosts only: prod ``inference-api.nousresearch.com`` or the operator-set + ``NOUS_INFERENCE_BASE_URL`` host (exact hostname equality, so neither lookalike + domains nor sibling hosts of the override match). """ if base_url_host_matches(base_url or "", "inference-api.nousresearch.com"): return True @@ -176,83 +126,54 @@ def _is_nous_portal_endpoint(base_url: str | None) -> bool: return False if not override: return False - # Exact host equality (not subdomain) so the env override can't broaden - # into sibling hosts the operator did not set. override_host = base_url_hostname(override) return bool(override_host) and base_url_hostname(base_url or "") == override_host def _requires_bearer_auth(base_url: str | None) -> bool: - """Return True for Anthropic-compatible providers that require Bearer auth. + """True for Anthropic-compatible providers that need ``Authorization: Bearer`` + instead of ``x-api-key``: MiniMax, Azure AI Foundry, Palantir Foundry's LLM + proxy, CommandCode, and Nous Portal. - Some third-party /anthropic endpoints implement Anthropic's Messages API but - require Authorization: Bearer instead of Anthropic's native x-api-key header. - MiniMax's global and China Anthropic-compatible endpoints, Azure AI - Foundry's Anthropic-style endpoint, Palantir Foundry's LLM proxy, and Nous - Portal's Messages route follow this pattern. + Palantir/CommandCode use hostname matching (not substring) so e.g. + ``evil.com/palantirfoundry`` paths don't trigger Bearer auth. """ if _is_nous_portal_endpoint(base_url): return True - normalized = _normalize_base_url_text(base_url) + normalized = _normalized_lower(base_url) if not normalized: return False - normalized = normalized.rstrip("/").lower() return ( - normalized.startswith(("https://api.minimax.io/anthropic", "https://api.minimaxi.com/anthropic")) + normalized.startswith(_MINIMAX_ANTHROPIC_PREFIXES) or "azure.com" in normalized - # Palantir Foundry LLM proxy (.palantirfoundry.com/api/v2/llm/proxy/anthropic) - # rejects x-api-key with 401 and requires Authorization: Bearer. - # Hostname match (not substring) so e.g. evil.com/palantirfoundry - # paths don't trigger Bearer auth. or base_url_host_matches(normalized, "palantirfoundry.com") - # CommandCode's /provider/v1/messages endpoint uses Bearer auth, - # not Anthropic's native x-api-key header. Hostname match for the - # same reason as above. or base_url_host_matches(normalized, "api.commandcode.ai") ) def _base_url_needs_context_1m_beta(base_url: str | None) -> bool: - """Return True for endpoints that still gate 1M context behind a beta.""" - normalized = _normalize_base_url_text(base_url).lower() - if not normalized: - return False - return "azure.com" in normalized + """True for endpoints that still gate 1M context behind a beta (Azure).""" + return "azure.com" in _normalize_base_url_text(base_url).lower() def _is_minimax_anthropic_endpoint(base_url: str | None) -> bool: - """Return True for MiniMax's Anthropic-compatible endpoints. - - MiniMax rejects the fine-grained-tool-streaming and context-1m betas; - those need to be stripped even though MiniMax also uses Bearer auth. - """ - normalized = _normalize_base_url_text(base_url) - if not normalized: - return False - normalized = normalized.rstrip("/").lower() - return normalized.startswith( - ("https://api.minimax.io/anthropic", "https://api.minimaxi.com/anthropic") - ) + """True for MiniMax's Anthropic-compatible endpoints, which reject the + fine-grained-tool-streaming and context-1m betas (stripped even though MiniMax + also uses Bearer auth).""" + return _normalized_lower(base_url).startswith(_MINIMAX_ANTHROPIC_PREFIXES) def _is_azure_anthropic_endpoint(base_url: str | None) -> bool: - """Return True for Azure-hosted Anthropic Messages endpoints. + """True for Azure-hosted Anthropic Messages endpoints serving ``/anthropic``: + modern Foundry (``*.services.ai.azure.*``) and legacy Azure OpenAI + (``*.openai.azure.*``) hosts. Opts them into ``api-version`` query plumbing. - Covers both the modern Foundry host family (``*.services.ai.azure.*``) - and the legacy Azure OpenAI host family (``*.openai.azure.*``) when - serving Anthropic's ``/anthropic`` route. Used to opt-in those hosts - to the ``api-version`` query-param plumbing required by Azure. - - Intentionally avoids a finite allow-list of TLD suffixes so it works - across sovereign / private Azure clouds. + Deliberately no finite TLD allow-list, so sovereign/private clouds work. """ normalized = _normalize_base_url_text(base_url) if not normalized: return False parsed = urlparse(normalized) - host = (parsed.hostname or "").lower().rstrip(".") - path = (parsed.path or "").lower() - host_padded = f".{host}." - is_foundry_host = ".services.ai.azure." in host_padded - is_legacy_azoai_host = ".openai.azure." in host_padded - return (is_foundry_host or is_legacy_azoai_host) and "/anthropic" in path + host_padded = f".{(parsed.hostname or '').lower().rstrip('.')}." + is_azure_host = ".services.ai.azure." in host_padded or ".openai.azure." in host_padded + return is_azure_host and "/anthropic" in (parsed.path or "").lower() diff --git a/agent/anthropic_message_convert.py b/agent/anthropic_message_convert.py index 4ef3e887cb..f3bca75ff6 100644 --- a/agent/anthropic_message_convert.py +++ b/agent/anthropic_message_convert.py @@ -6,20 +6,15 @@ signatures, tool_use/tool_result pairing, cache_control placement, screenshot eviction, blank-block scrubbing). Split out of ``agent/anthropic_adapter.py`` so the adapter keeps client -construction and the API call itself, while the payload-shaping rules - by far -the largest and most fiddly part - have their own home. The endpoint-family -predicates a few of these rules branch on come from +construction and the API call itself. Endpoint predicates come from ``agent/anthropic_endpoints.py``, so this module never imports the adapter and -there is no import cycle. - -``agent.anthropic_adapter`` re-exports every name below, so existing -``from agent.anthropic_adapter import convert_messages_to_anthropic`` imports -keep working. +there is no cycle. ``agent.anthropic_adapter`` re-exports every name below. """ import copy import json import logging +import re from typing import Any, Dict, List, Optional, Tuple from agent.anthropic_endpoints import ( @@ -31,98 +26,100 @@ from agent.anthropic_endpoints import ( logger = logging.getLogger(__name__) +_THINKING_TYPES = frozenset(("thinking", "redacted_thinking")) +_CACHEABLE_TYPES = frozenset(("text", "tool_use")) +_EMPTY_TEXT_PLACEHOLDER = "(empty)" +_BEDROCK_REGION_PREFIXES = ( + "global.", "us.", "eu.", "apac.", "ap.", "au.", "jp.", "ca.", "sa.", "me.", "af.", +) + # --------------------------------------------------------------------------- -# Message / tool / response format conversion +# Small shared predicates +# --------------------------------------------------------------------------- + + +def _block_type(b: Any) -> Any: + """``type`` of a dict block, None for non-dicts.""" + return b.get("type") if isinstance(b, dict) else None + + +def _has_block_type(blocks: List[Any], types) -> bool: + return any(_block_type(b) in types for b in blocks) + + +def _is_blank_text_block(b: Any) -> bool: + """A text block whose ``text`` is not a non-whitespace string (None/int/blank all count). + + Anthropic rejects such blocks with HTTP 400 ("text content blocks must contain + non-whitespace text"); checking isinstance first keeps a non-string from + reaching ``.strip()``. + """ + if _block_type(b) != "text": + return False + text = b.get("text") + return not (isinstance(text, str) and text.strip()) + + +def _cache_control_of(b: Any) -> Optional[Dict[str, Any]]: + cc = b.get("cache_control") if isinstance(b, dict) else None + return cc if isinstance(cc, dict) else None + + +def _text_block(text: str) -> Dict[str, str]: + return {"type": "text", "text": text} + + +def _parse_tool_args(raw: Any) -> Any: + """JSON-decode a tool_call ``arguments`` string; non-strings pass through, bad JSON -> {}.""" + try: + return json.loads(raw) if isinstance(raw, str) else raw + except (json.JSONDecodeError, ValueError): + return {} + + +# --------------------------------------------------------------------------- +# Model / tool conversion # --------------------------------------------------------------------------- def _is_bedrock_model_id(model: str) -> bool: - """Detect AWS Bedrock model IDs that use dots as namespace separators. - - Bedrock model IDs come in two forms: - - Bare: ``anthropic.claude-opus-4-7`` - - Regional (inference profiles): ``us.anthropic.claude-sonnet-4-5-v1:0`` - - In both cases the dots separate namespace components, not version - numbers, and must be preserved verbatim for the Bedrock API. - """ - lower = model.lower() - # Regional inference-profile prefixes - if any(lower.startswith(p) for p in ( - "global.", "us.", "eu.", "apac.", "ap.", "au.", "jp.", - "ca.", "sa.", "me.", "af.", - )): - return True - # Bare Bedrock model IDs: provider.model-family - if lower.startswith("anthropic."): - return True - return False + """Bedrock ids (``anthropic.claude-opus-4-7``, ``us.anthropic.claude-*``) use dots + as namespace separators that must be preserved verbatim.""" + return model.lower().startswith(_BEDROCK_REGION_PREFIXES + ("anthropic.",)) def normalize_model_name(model: str, preserve_dots: bool = False) -> str: """Normalize a model name for the Anthropic API. - - Strips 'anthropic/' prefix (OpenRouter format, case-insensitive) - - Converts dots to hyphens in version numbers (OpenRouter uses dots, - Anthropic uses hyphens: claude-opus-4.6 → claude-opus-4-6), unless - preserve_dots is True (e.g. for Alibaba/DashScope: qwen3.5-plus). - - Preserves Bedrock model IDs (``anthropic.claude-opus-4-7``) and - regional inference profiles (``us.anthropic.claude-*``) whose dots - are namespace separators, not version separators. + Strips the ``anthropic/`` prefix (case-insensitive) and, unless + ``preserve_dots`` (DashScope: ``qwen3.5-plus``), converts version dots to + hyphens for Claude models only (``claude-opus-4.6`` -> ``claude-opus-4-6``). + Bedrock ids keep their namespace dots; non-Anthropic names (``gpt-5.4``) + keep dots as part of their canonical form. """ - lower = model.lower() - if lower.startswith("anthropic/"): + if model.lower().startswith("anthropic/"): model = model[len("anthropic/"):] - if not preserve_dots: - # Bedrock model IDs use dots as namespace separators - # (e.g. "anthropic.claude-opus-4-7", "us.anthropic.claude-*"). - # These must not be converted to hyphens. See issue #12295. - if _is_bedrock_model_id(model): - return model - # Only convert dots to hyphens for Anthropic/Claude models. - # Non-Anthropic models (gpt-5.4, gemini-2.5, etc.) use dots - # as part of their canonical names. See issue #17171. - _lower = model.lower() - if _lower.startswith("claude-") or _lower.startswith("anthropic/"): - model = model.replace(".", "-") + if not preserve_dots and not _is_bedrock_model_id(model) and model.lower().startswith(("claude-", "anthropic/")): + model = model.replace(".", "-") return model def _sanitize_tool_id(tool_id: str) -> str: - """Sanitize a tool call ID for the Anthropic API. - - Anthropic requires IDs matching [a-zA-Z0-9_-]. Replace invalid - characters with underscores and ensure non-empty. - """ - import re + """Anthropic requires ids matching [a-zA-Z0-9_-]; replace the rest, never empty.""" if not tool_id: return "tool_0" - sanitized = re.sub(r"[^a-zA-Z0-9_-]", "_", tool_id) - return sanitized or "tool_0" + return re.sub(r"[^a-zA-Z0-9_-]", "_", tool_id) or "tool_0" def _normalize_tool_input_schema(schema: Any) -> Dict[str, Any]: - """Normalize tool schemas before sending them to Anthropic. + """Normalize a tool schema for Anthropic's validator. - Anthropic's tool schema validator rejects nullable unions such as - ``anyOf: [{"type": "string"}, {"type": "null"}]`` that Pydantic/MCP - commonly emits for optional fields. Tool optionality is represented by - the parent ``required`` array, so we delegate to the shared - ``strip_nullable_unions`` helper to collapse nullable unions to the - non-null branch while preserving metadata like description/default. - - ``keep_nullable_hint=False`` because the Anthropic validator does not - recognize the OpenAPI-style ``nullable: true`` extension and strict - schema-to-grammar converters may reject unknown keywords. - - Top-level ``oneOf``/``allOf``/``anyOf`` are also stripped here: the - Anthropic API rejects union keywords at the schema root with a generic - HTTP 400. Several upstream and plugin tools ship schemas with one of - these keywords at the top level (commonly for Pydantic discriminated - unions). If we land here with those keywords still present after - nullable-union stripping, drop them and fall back to a plain object - schema so the tool still validates at the Anthropic boundary. + Collapses nullable unions (``anyOf: [{type: string}, {type: null}]``, which + Pydantic/MCP emit for optional fields) to the non-null branch — optionality is + already expressed by ``required``. ``keep_nullable_hint=False`` because the + OpenAPI ``nullable`` keyword is not recognized. Top-level oneOf/allOf/anyOf are + rejected with a generic 400, so they are dropped in favour of a plain object. """ if not schema: return {"type": "object", "properties": {}} @@ -132,19 +129,21 @@ def _normalize_tool_input_schema(schema: Any) -> Dict[str, Any]: normalized = strip_nullable_unions(schema, keep_nullable_hint=False) if not isinstance(normalized, dict): return {"type": "object", "properties": {}} - # Strip top-level union keywords that Anthropic's validator rejects. banned = {"oneOf", "allOf", "anyOf"} if banned & normalized.keys(): normalized = {k: v for k, v in normalized.items() if k not in banned} - if "type" not in normalized: - normalized["type"] = "object" + normalized.setdefault("type", "object") if normalized.get("type") == "object" and not isinstance(normalized.get("properties"), dict): normalized = {**normalized, "properties": {}} return normalized def convert_tools_to_anthropic(tools: List[Dict]) -> List[Dict]: - """Convert OpenAI tool definitions to Anthropic format.""" + """Convert OpenAI tool definitions to Anthropic format. + + Duplicate names are dropped with a warning (Anthropic hard-400s on them); + ``cache_control`` on the OpenAI tool dict is forwarded. + """ if not tools: return [] result = [] @@ -152,14 +151,9 @@ def convert_tools_to_anthropic(tools: List[Dict]) -> List[Dict]: for t in tools: fn = t.get("function", {}) name = fn.get("name", "") - # Defensive dedup: Anthropic rejects requests with duplicate tool - # names. Upstream injection paths already dedup, but this guard - # converts a hard API failure into a warning. See: #18478 if name and name in seen_names: logger.warning( - "convert_tools_to_anthropic: duplicate tool name '%s' " - "— dropping second occurrence", - name, + "convert_tools_to_anthropic: duplicate tool name '%s' — dropping second occurrence", name ) continue if name: @@ -167,63 +161,48 @@ def convert_tools_to_anthropic(tools: List[Dict]) -> List[Dict]: anthropic_tool: Dict[str, Any] = { "name": name, "description": fn.get("description", ""), - "input_schema": _normalize_tool_input_schema( - fn.get("parameters", {"type": "object", "properties": {}}) - ), + "input_schema": _normalize_tool_input_schema(fn.get("parameters") or {}), } - # Forward cache_control marker when present on the OpenAI-format - # tool dict. Anthropic's tools array supports cache_control on the - # last tool to cache the entire schema cross-session. - cache_control = t.get("cache_control") - if isinstance(cache_control, dict): + cache_control = _cache_control_of(t) + if cache_control is not None: anthropic_tool["cache_control"] = dict(cache_control) result.append(anthropic_tool) return result -def _image_source_from_openai_url(url: str) -> Dict[str, str]: - """Convert an OpenAI-style image URL/data URL into Anthropic image source.""" - url = str(url or "").strip() - if not url: - return {"type": "url", "url": ""} +# --------------------------------------------------------------------------- +# Content-part conversion +# --------------------------------------------------------------------------- + +def _image_source_from_openai_url(url: str) -> Dict[str, str]: + """OpenAI image URL / data URL -> Anthropic image ``source``.""" + url = str(url or "").strip() if url.startswith("data:"): header, _, data = url.partition(",") - media_type = "image/jpeg" - if header.startswith("data:"): - mime_part = header[len("data:"):].split(";", 1)[0].strip() - if mime_part.startswith("image/"): - media_type = mime_part - return { - "type": "base64", - "media_type": media_type, - "data": data, - } - + mime_part = header[len("data:"):].split(";", 1)[0].strip() + media_type = mime_part if mime_part.startswith("image/") else "image/jpeg" + return {"type": "base64", "media_type": media_type, "data": data} return {"type": "url", "url": url} def _convert_content_part_to_anthropic(part: Any) -> Optional[Dict[str, Any]]: - """Convert a single OpenAI-style content part to Anthropic format.""" + """Convert one OpenAI-style content part to an Anthropic block (None -> dropped).""" if part is None: return None if isinstance(part, str): - return {"type": "text", "text": part} + return _text_block(part) if not isinstance(part, dict): - return {"type": "text", "text": str(part)} + return _text_block(str(part)) ptype = part.get("type") - - if ptype == "input_text": - block: Dict[str, Any] = {"type": "text", "text": part.get("text", "")} - elif ptype == "text": - # A stored Anthropic text block. Rebuild from whitelisted fields only — - # SDK response text blocks carry output-only siblings (parsed_output, - # citations=None) that the Messages INPUT schema rejects with HTTP 400 - # "Extra inputs are not permitted". Do NOT dict(part) it verbatim. - block = {"type": "text", "text": part.get("text", "")} + if ptype in ("input_text", "text"): + # Rebuild from whitelisted fields only: stored SDK text blocks carry + # output-only siblings (parsed_output, citations=None) that the INPUT + # schema rejects with 400 "Extra inputs are not permitted". + block: Dict[str, Any] = _text_block(part.get("text", "")) cits = part.get("citations") - if isinstance(cits, list) and cits: + if ptype == "text" and isinstance(cits, list) and cits: block["citations"] = cits elif ptype in {"image_url", "input_image"}: image_value = part.get("image_url", {}) @@ -232,136 +211,93 @@ def _convert_content_part_to_anthropic(part: Any) -> Optional[Dict[str, Any]]: else: block = dict(part) - if isinstance(part.get("cache_control"), dict) and "cache_control" not in block: - block["cache_control"] = dict(part["cache_control"]) + cache_control = _cache_control_of(part) + if cache_control is not None: + block.setdefault("cache_control", dict(cache_control)) return block def _to_plain_data(value: Any, *, _depth: int = 0, _path: Optional[set] = None) -> Any: - """Recursively convert SDK objects to plain Python data structures. + """Recursively convert SDK objects to plain Python data. - Guards against circular references (``_path`` tracks ``id()`` of objects - on the *current* recursion path) and runaway depth (capped at 20 levels). - Uses path-based tracking so shared (but non-cyclic) objects referenced by - multiple siblings are converted correctly rather than being stringified. + ``_path`` tracks ids on the *current* recursion path (so shared but + non-cyclic objects convert normally while true cycles stringify); depth is + capped at 20. """ - _MAX_DEPTH = 20 - if _depth > _MAX_DEPTH: + if _depth > 20: return str(value) - if _path is None: _path = set() - obj_id = id(value) if obj_id in _path: return str(value) + def rec(v): + return _to_plain_data(v, _depth=_depth + 1, _path=_path) + + _path.add(obj_id) if hasattr(value, "model_dump"): - _path.add(obj_id) try: - # warnings=False: content blocks from the streaming accumulator - # (ParsedTextBlock et al.) trip pydantic's serializer-mismatch - # UserWarning against the generic Message union; the dump itself - # is correct, and the warning leaks to the user's terminal. + # warnings=False: streaming-accumulator blocks trip pydantic's + # serializer-mismatch UserWarning, which otherwise leaks to the terminal. dumped = value.model_dump(warnings=False) - except TypeError: - # Duck-typed model_dump without pydantic's signature. + except TypeError: # duck-typed model_dump without pydantic's signature dumped = value.model_dump() - result = _to_plain_data(dumped, _depth=_depth + 1, _path=_path) - _path.discard(obj_id) - return result - if isinstance(value, dict): - _path.add(obj_id) - result = {k: _to_plain_data(v, _depth=_depth + 1, _path=_path) for k, v in value.items()} - _path.discard(obj_id) - return result - if isinstance(value, (list, tuple)): - _path.add(obj_id) - result = [_to_plain_data(v, _depth=_depth + 1, _path=_path) for v in value] - _path.discard(obj_id) - return result - if hasattr(value, "__dict__"): - _path.add(obj_id) - result = { - k: _to_plain_data(v, _depth=_depth + 1, _path=_path) - for k, v in vars(value).items() - if not k.startswith("_") - } - _path.discard(obj_id) - return result - return value + result = rec(dumped) + elif isinstance(value, dict): + result = {k: rec(v) for k, v in value.items()} + elif isinstance(value, (list, tuple)): + result = [rec(v) for v in value] + elif hasattr(value, "__dict__"): + result = {k: rec(v) for k, v in vars(value).items() if not k.startswith("_")} + else: + result = value + _path.discard(obj_id) + return result def _extract_preserved_thinking_blocks(message: Dict[str, Any]) -> List[Dict[str, Any]]: - """Return Anthropic thinking blocks previously preserved on the message.""" + """Deep-copied thinking/redacted_thinking blocks from ``reasoning_details``.""" raw_details = message.get("reasoning_details") if not isinstance(raw_details, list): return [] - - preserved: List[Dict[str, Any]] = [] - for detail in raw_details: - if not isinstance(detail, dict): - continue - block_type = str(detail.get("type", "") or "").strip().lower() - if block_type not in {"thinking", "redacted_thinking"}: - continue - preserved.append(copy.deepcopy(detail)) - return preserved + return [ + copy.deepcopy(d) + for d in raw_details + if isinstance(d, dict) and str(d.get("type", "") or "").strip().lower() in _THINKING_TYPES + ] def _convert_content_to_anthropic(content: Any) -> Any: - """Convert OpenAI-style multimodal content arrays to Anthropic blocks.""" + """Convert an OpenAI multimodal content list to Anthropic blocks (non-lists pass through).""" if not isinstance(content, list): return content - - converted = [] - for part in content: - block = _convert_content_part_to_anthropic(part) - if block is not None: - converted.append(block) - return converted + return [b for b in map(_convert_content_part_to_anthropic, content) if b is not None] def _content_parts_to_anthropic_blocks(parts: Any) -> List[Dict[str, Any]]: - """Convert OpenAI-style tool-message content parts → Anthropic tool_result inner blocks. - - Used for multimodal tool results (e.g. computer_use screenshots). Each - part is normalized via `_convert_content_part_to_anthropic`, then - filtered to the block types Anthropic tool_result accepts (text + image). - """ + """Tool-message content parts -> tool_result inner blocks (text + image only, + the types Anthropic accepts there). Used for multimodal tool results.""" if not isinstance(parts, list): return [] out: List[Dict[str, Any]] = [] - for part in parts: - block = _convert_content_part_to_anthropic(part) + for block in map(_convert_content_part_to_anthropic, parts): if not block: continue - btype = block.get("type") - if btype == "text": - text_val = block.get("text") - if isinstance(text_val, str) and text_val: - out.append({"type": "text", "text": text_val}) - elif btype == "image": - src = block.get("source") - if isinstance(src, dict) and src: - out.append({"type": "image", "source": src}) + btype, text_val, src = block.get("type"), block.get("text"), block.get("source") + if btype == "text" and isinstance(text_val, str) and text_val: + out.append(_text_block(text_val)) + elif btype == "image" and isinstance(src, dict) and src: + out.append({"type": "image", "source": src}) return out -_EMPTY_TEXT_PLACEHOLDER = "(empty)" - - def _safe_text(text: Any) -> str: - """Return ``text`` if it's non-whitespace, else a non-whitespace placeholder. + """``text`` if non-whitespace, else the placeholder. - The Anthropic Messages API rejects requests where a text content block is - empty or whitespace-only (HTTP 400 "text content blocks must contain - non-whitespace text"). When such a block gets stored in session history — - e.g. produced by context compression — it is replayed verbatim on every - subsequent turn, permanently wedging the session. Coercing to a - non-whitespace placeholder is self-healing: the next API call recovers. - - Mirrors ``bedrock_adapter._safe_text`` (#9486); ref #69512. + A blank text block stored in history (e.g. by compression) is replayed on every + turn and wedges the session with HTTP 400; the placeholder is self-healing. + Mirrors ``bedrock_adapter._safe_text`` (kept separate on purpose). """ if text is None: return _EMPTY_TEXT_PLACEHOLDER @@ -370,67 +306,76 @@ def _safe_text(text: Any) -> str: return text if text.strip() else _EMPTY_TEXT_PLACEHOLDER +# --------------------------------------------------------------------------- +# Replay-block sanitizing (per-type whitelist) +# --------------------------------------------------------------------------- + + +def _replay_text(b: Dict[str, Any]) -> Optional[Dict[str, Any]]: + # Drop blank blocks rather than coerce in place: the caller relocates any + # cache_control and falls back to a placeholder only when nothing survives, + # so "(empty)" never sits as model-visible noise next to real blocks. + if _is_blank_text_block(b): + return None + out: Dict[str, Any] = _text_block(b["text"]) + cits = b.get("citations") # input-valid ONLY as a non-empty list + if isinstance(cits, list) and cits: + out["citations"] = cits + if _cache_control_of(b) is not None: + out["cache_control"] = b["cache_control"] + return out + + +def _replay_thinking(b: Dict[str, Any]) -> Dict[str, Any]: + out = {"type": "thinking", "thinking": b.get("thinking", "")} + if b.get("signature"): + out["signature"] = b["signature"] + return out + + +def _replay_redacted_thinking(b: Dict[str, Any]) -> Optional[Dict[str, Any]]: + return {"type": "redacted_thinking", "data": b["data"]} if b.get("data") else None + + +def _replay_tool_use(b: Dict[str, Any]) -> Dict[str, Any]: + out = { + "type": "tool_use", + "id": _sanitize_tool_id(b.get("id", "")), + "name": b.get("name", ""), + "input": b.get("input", {}), + } + if _cache_control_of(b) is not None: + out["cache_control"] = b["cache_control"] + return out + + +def _replay_image(b: Dict[str, Any]) -> Optional[Dict[str, Any]]: + src = b.get("source") + return {"type": "image", "source": src} if isinstance(src, dict) else None + + +_REPLAY_SANITIZERS = { + "text": _replay_text, + "thinking": _replay_thinking, + "redacted_thinking": _replay_redacted_thinking, + "tool_use": _replay_tool_use, + "image": _replay_image, +} + + def _sanitize_replay_block(b: Dict[str, Any]) -> Optional[Dict[str, Any]]: - """Strip output-only fields from a stored Anthropic content block so it is - valid as REQUEST input on replay. + """Whitelist a stored Anthropic block so it is valid as REQUEST input. - The SDK response objects carry output-only attributes that the Messages - *input* schema forbids ("Extra inputs are not permitted"): text blocks get - ``parsed_output``/``citations`` (when null), tool_use blocks get ``caller``, - etc. ``normalize_response`` captured blocks verbatim via ``_to_plain_data``, - so these leak back as input on the next turn → HTTP 400. - - Whitelist per type (NOT a blacklist) so future SDK output-only fields can't - reintroduce the bug. Returns a clean block, or None to drop it. + SDK response blocks carry output-only fields the INPUT schema forbids + ("Extra inputs are not permitted": ``parsed_output``, ``caller``, + ``citations=None``), and ``_to_plain_data`` captured them verbatim. Whitelist + per type (not blacklist) so future SDK fields can't reintroduce the bug; + unknown types are dropped. Returns a clean block or None. """ if not isinstance(b, dict): return None - btype = b.get("type") - if btype == "text": - text_val = b.get("text", "") - # Bedrock and strict Anthropic-compatible endpoints reject text - # blocks where "text" is empty or whitespace-only (#69512). Drop the - # blank block (the caller relocates any cache_control it carried and - # falls back to a non-whitespace placeholder when nothing survives) - # rather than coercing in place — a coerced "(empty)" block would be - # model-visible noise next to surviving thinking/tool_use blocks. - # Type-safe: captured blocks can carry text=None from an invalid - # upstream payload, which a bare .strip() would crash on. - if not isinstance(text_val, str) or not text_val.strip(): - return None - out: Dict[str, Any] = {"type": "text", "text": text_val} - # citations is input-valid ONLY when it's a non-empty list; the SDK - # emits citations=None on responses, which the input schema rejects. - cits = b.get("citations") - if isinstance(cits, list) and cits: - out["citations"] = cits - if isinstance(b.get("cache_control"), dict): - out["cache_control"] = b["cache_control"] - return out - if btype == "thinking": - out = {"type": "thinking", "thinking": b.get("thinking", "")} - if b.get("signature"): - out["signature"] = b["signature"] - return out - if btype == "redacted_thinking": - # Only valid with its data payload; drop if missing. - return {"type": "redacted_thinking", "data": b["data"]} if b.get("data") else None - if btype == "tool_use": - out = { - "type": "tool_use", - "id": _sanitize_tool_id(b.get("id", "")), - "name": b.get("name", ""), - "input": b.get("input", {}), - } - if isinstance(b.get("cache_control"), dict): - out["cache_control"] = b["cache_control"] - return out - if btype == "image": - src = b.get("source") - return {"type": "image", "source": src} if isinstance(src, dict) else None - # Unknown/unsupported block type on the input path — drop rather than risk - # another "Extra inputs are not permitted". - return None + sanitizer = _REPLAY_SANITIZERS.get(b.get("type")) + return sanitizer(b) if sanitizer else None def _apply_assistant_cache_control_to_last_cacheable_block( @@ -440,607 +385,347 @@ def _apply_assistant_cache_control_to_last_cacheable_block( if not isinstance(cache_control, dict): return for block in reversed(blocks): - if isinstance(block, dict) and block.get("type") in {"text", "tool_use"}: + if _block_type(block) in _CACHEABLE_TYPES: block.setdefault("cache_control", dict(cache_control)) break -def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]: - """Convert an assistant message to Anthropic content blocks. +# --------------------------------------------------------------------------- +# Per-message conversion +# --------------------------------------------------------------------------- - Handles thinking blocks, regular content, tool calls, and - reasoning_content injection for Kimi/DeepSeek endpoints. + +def _replay_ordered_blocks(m: Dict[str, Any], ordered_blocks: List[Any]) -> Optional[List[Dict[str, Any]]]: + """Interleaved-thinking replay: rebuild the assistant turn from the verbatim + block list normalize_response stored (only for turns interleaving SIGNED + thinking with tool_use). Preserves block ORDER; returns None if nothing survives. + + tool_use ``input`` is re-sourced from ``tool_calls`` (redacted at storage time) + rather than the captured block (raw API response, NOT redacted), so a secret + the model inlined into a tool call never goes back on the wire. """ + redacted_input_by_id = { + _sanitize_tool_id(tc.get("id", "")): _parse_tool_args((tc.get("function", {}) or {}).get("arguments", "{}")) + for tc in m.get("tool_calls", []) or [] + if isinstance(tc, dict) + } + replayed: List[Dict[str, Any]] = [] + relocated_cc = None + dropped_blank_text = False + for b in ordered_blocks: + clean = _sanitize_replay_block(b) + if clean is None: + if _block_type(b) == "text": + dropped_blank_text = True + if _cache_control_of(b) is not None: # relocate a dropped block's breakpoint + relocated_cc = b["cache_control"] + continue + if clean.get("type") == "tool_use": + redacted = redacted_input_by_id.get(clean.get("id", "")) + if redacted is not None: + clean["input"] = redacted + replayed.append(clean) + # Nothing cacheable survived (e.g. signed thinking + blank text): emit the + # placeholder so the turn stays schema-valid and a relocated marker has a carrier. + if not _has_block_type(replayed, _CACHEABLE_TYPES) and (dropped_blank_text or relocated_cc is not None): + replayed.append(_text_block(_EMPTY_TEXT_PLACEHOLDER)) + if not replayed: + return None + _apply_assistant_cache_control_to_last_cacheable_block(replayed, relocated_cc) + _apply_assistant_cache_control_to_last_cacheable_block(replayed, m.get("cache_control")) + # prompt_caching marks an assistant turn with text by writing cache_control + # INTO ``content`` (not top-level). This path never reads ``content``, so + # carry that marker over or the breakpoint is burned rather than relocated. + msg_content = m.get("content") + if isinstance(msg_content, list): + inline_cc = next((cc for cc in map(_cache_control_of, msg_content) if cc is not None), None) + _apply_assistant_cache_control_to_last_cacheable_block(replayed, inline_cc) + return replayed + + +def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]: + """Assistant message -> Anthropic content blocks (thinking, text, tool_use, + Kimi/DeepSeek reasoning_content injection).""" content = m.get("content", "") - # Anthropic interleaved-thinking fast path: when this turn carries a - # verbatim, order-preserving block list (set by normalize_response only - # for turns that interleave SIGNED thinking with tool_use), replay it. - # Each block is run through _sanitize_replay_block to strip output-only - # SDK fields (parsed_output, caller, citations=None, …) that the Messages - # INPUT schema forbids — replaying them verbatim caused HTTP 400 "Extra - # inputs are not permitted" (text.parsed_output). Block ORDER is preserved - # (the reason this channel exists); only forbidden sibling fields are - # dropped, leaving thinking signatures and tool_use id/name/input intact. ordered_blocks = m.get("anthropic_content_blocks") if isinstance(ordered_blocks, list) and ordered_blocks: - # Re-source each tool_use input from the stored tool_calls map rather - # than the captured block. The ordered-blocks list captures tool_use - # input from the RAW API response (normalize_response), which is NOT - # credential-redacted; tool_calls[].function.arguments IS redacted at - # storage time (build_assistant_message, #19798). Replaying the raw - # block input would resurrect a secret the model inlined into a tool - # call (e.g. terminal(command="curl -H 'Authorization: Bearer sk-...'") - # onto the wire, even though the same value is redacted everywhere else - # in history. Keying by sanitized tool id preserves interleave order - # (the reason this channel exists) while swapping in the redacted - # input. Adapted from #36071 (replay-time tool-input re-sourcing). - redacted_input_by_id: Dict[str, Any] = {} - for tc in m.get("tool_calls", []) or []: - if not isinstance(tc, dict): - continue - fn = tc.get("function", {}) or {} - raw_args = fn.get("arguments", "{}") - try: - parsed_args = json.loads(raw_args) if isinstance(raw_args, str) else raw_args - except (json.JSONDecodeError, ValueError): - parsed_args = {} - redacted_input_by_id[_sanitize_tool_id(tc.get("id", ""))] = parsed_args - replayed: List[Dict[str, Any]] = [] - _relocated_replay_cache_control = None - _dropped_blank_text = False - for b in ordered_blocks: - clean = _sanitize_replay_block(b) - if clean is None: - if isinstance(b, dict) and b.get("type") == "text": - _dropped_blank_text = True - if isinstance(b, dict) and isinstance(b.get("cache_control"), dict): - # A dropped blank text block can still carry the cache - # breakpoint marker -- relocate it rather than losing it. - _relocated_replay_cache_control = b["cache_control"] - continue - if clean.get("type") == "tool_use": - # Override raw (un-redacted) input with the redacted copy when - # we have one for this id; fall back to the sanitized block - # input only if the tool_call is missing (shape mismatch). - redacted = redacted_input_by_id.get(clean.get("id", "")) - if redacted is not None: - clean["input"] = redacted - replayed.append(clean) - # When every text block was blank and nothing cacheable survived - # (e.g. signed thinking + a blank text block, or a SOLE blank - # cache-marked block), emit the non-whitespace placeholder so the - # replayed message stays schema-valid (#69512) and a relocated cache - # marker still has a carrier instead of being silently lost. - _has_cacheable_replay = any( - isinstance(b, dict) and b.get("type") in {"text", "tool_use"} - for b in replayed - ) - if not _has_cacheable_replay and ( - _dropped_blank_text or _relocated_replay_cache_control is not None - ): - replayed.append({"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER}) + replayed = _replay_ordered_blocks(m, ordered_blocks) if replayed: - if _relocated_replay_cache_control is not None: - _apply_assistant_cache_control_to_last_cacheable_block( - replayed, _relocated_replay_cache_control - ) - _apply_assistant_cache_control_to_last_cacheable_block( - replayed, m.get("cache_control") - ) - # apply_anthropic_cache_control marks an assistant turn with - # non-empty text by writing cache_control INTO ``content`` (see - # _apply_cache_marker's list branch), not at the top level. This - # branch rebuilds the message from ordered_blocks and never reads - # ``content``, so that marker would be dropped -- and because - # _can_carry_marker already counted this message as a carrier, the - # breakpoint is burned rather than relocated. #56195 covered the - # complementary shape (blank content -> top-level marker); this is - # the interleaved thinking + preamble-text + tool_use shape. - _inline_cc = None - _msg_content = m.get("content") - if isinstance(_msg_content, list): - for _blk in _msg_content: - if isinstance(_blk, dict) and isinstance( - _blk.get("cache_control"), dict - ): - _inline_cc = _blk["cache_control"] - break - if _inline_cc is not None: - _apply_assistant_cache_control_to_last_cacheable_block( - replayed, _inline_cc - ) return {"role": "assistant", "content": replayed} blocks = _extract_preserved_thinking_blocks(m) - # Cache markers dropped along with a blank block are relocated onto the - # last surviving cacheable block below (via - # _apply_assistant_cache_control_to_last_cacheable_block), rather than - # lost -- prompt_caching.py's _apply_cache_marker() sets cache_control - # directly on content[-1] for list content, so if that last part happens - # to be blank text, dropping it silently would lose the breakpoint. - _relocated_cache_control = None - if content: - if isinstance(content, list): - converted_content = _convert_content_to_anthropic(content) - if isinstance(converted_content, list): - # Bedrock and strict Anthropic-compatible endpoints reject - # text blocks where "text" is empty or whitespace-only. The - # ordered-replay path enforces the same invariant via - # _sanitize_replay_block(). Type-safe against ANY invalid - # "text" value from an upstream payload -- None, or a - # truthy non-string like an int -- not just None: checking - # isinstance() first (rather than `blk.get("text") or ""`) - # means a non-string value is treated as blank/invalid - # instead of reaching .strip() and raising AttributeError. - for blk in converted_content: - _blk_text = blk.get("text") if isinstance(blk, dict) else None - if ( - isinstance(blk, dict) - and blk.get("type") == "text" - and (not isinstance(_blk_text, str) or not _blk_text.strip()) - ): - if isinstance(blk.get("cache_control"), dict): - _relocated_cache_control = blk["cache_control"] - continue - blocks.append(blk) - else: - # Scalar (non-list) content: a whitespace-only string is the - # same invalid-payload case as an empty list block -- drop it - # rather than emitting a blank text block. - text_str = str(content) - if text_str.strip(): - blocks.append({"type": "text", "text": text_str}) + # Blank text blocks are dropped; a cache marker riding on one is relocated + # onto the last surviving cacheable block (prompt_caching sets cache_control + # on content[-1], which may be exactly the blank block). + relocated_cc = None + if isinstance(content, list): + for blk in _convert_content_to_anthropic(content): + if _is_blank_text_block(blk): + if _cache_control_of(blk) is not None: + relocated_cc = blk["cache_control"] + continue + blocks.append(blk) + elif content and str(content).strip(): + blocks.append(_text_block(str(content))) for tc in m.get("tool_calls", []): if not tc or not isinstance(tc, dict): continue fn = tc.get("function", {}) - args = fn.get("arguments", "{}") - try: - parsed_args = json.loads(args) if isinstance(args, str) else args - except (json.JSONDecodeError, ValueError): - parsed_args = {} blocks.append({ "type": "tool_use", "id": _sanitize_tool_id(tc.get("id", "")), "name": fn.get("name", ""), - "input": parsed_args, + "input": _parse_tool_args(fn.get("arguments", "{}")), }) - # Kimi's /coding endpoint (Anthropic protocol) requires assistant - # tool-call messages to carry reasoning_content when thinking is - # enabled server-side. Preserve it as a thinking block so Kimi - # can validate the message history. See hermes-agent#13848. - # - # Accept empty string "" — _copy_reasoning_content_for_api() - # injects "" as a tier-3 fallback for Kimi tool-call messages - # that had no reasoning. Kimi requires the field to exist, even - # if empty. - # - # Prepend (not append): Anthropic protocol requires thinking - # blocks before text and tool_use blocks. - # - # Guard: only add when reasoning_details didn't already contribute - # thinking blocks. On native Anthropic, reasoning_details produces - # signed thinking blocks — adding another unsigned one from - # reasoning_content would create a duplicate (same text) that gets - # downgraded to a spurious text block on the last assistant message. + # Kimi's /coding endpoint requires reasoning_content on replayed tool-call + # turns — even "" (injected as a fallback upstream). Prepend, since thinking + # must precede text/tool_use. Skip when reasoning_details already supplied + # (signed) thinking blocks: a duplicate unsigned one would be downgraded to a + # spurious text block on the last assistant message. reasoning_content = m.get("reasoning_content") - _already_has_thinking = any( - isinstance(b, dict) and b.get("type") in {"thinking", "redacted_thinking"} - for b in blocks - ) - if isinstance(reasoning_content, str) and not _already_has_thinking: + if isinstance(reasoning_content, str) and not _has_block_type(blocks, _THINKING_TYPES): blocks.insert(0, {"type": "thinking", "thinking": reasoning_content}) - # Anthropic rejects empty assistant content. IMPORTANT: fall back only - # to the placeholder, never to the raw `content` variable -- `content` - # is the UNFILTERED original message content, and can itself be exactly - # the blank/whitespace-only payload the filtering above just removed - # (a sole blank text block, or scalar whitespace with no tool_calls). - # `blocks or content` there would silently restore the invalid provider - # payload this function exists to prevent (#69512). - effective = blocks if blocks else [{"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER}] - # Applied here (after the empty-fallback resolution) rather than - # earlier against `blocks` directly, so a cache_control relocated from - # a dropped blank block that was the ONLY block still lands on the - # (empty) placeholder instead of being silently lost when blocks was - # empty at the point the marker would otherwise have been applied. - if _relocated_cache_control is not None: - _apply_assistant_cache_control_to_last_cacheable_block( - effective, _relocated_cache_control - ) - _apply_assistant_cache_control_to_last_cacheable_block( - effective, m.get("cache_control") - ) + # Empty assistant content is rejected. Fall back ONLY to the placeholder, + # never to raw ``content`` — that is the unfiltered blank payload just removed. + effective = blocks or [_text_block(_EMPTY_TEXT_PLACEHOLDER)] + # Applied after the fallback so a marker from a sole dropped blank block + # lands on the placeholder instead of being lost. + _apply_assistant_cache_control_to_last_cacheable_block(effective, relocated_cc) + _apply_assistant_cache_control_to_last_cacheable_block(effective, m.get("cache_control")) return {"role": "assistant", "content": effective} -def _convert_tool_message_to_result( - result: List[Dict[str, Any]], m: Dict[str, Any] -) -> None: - """Convert a tool message to an Anthropic tool_result, merging consecutive - results into one user message. - - Mutates ``result`` in place — either appends a new user message or extends - the trailing user message's tool_result list. - """ +def _tool_result_content(m: Dict[str, Any]) -> Any: + """Resolve a tool message's content into tool_result content (blocks or string).""" content = m.get("content", "") multimodal_blocks: Optional[List[Dict[str, Any]]] = None if isinstance(content, dict) and content.get("_multimodal"): - multimodal_blocks = _content_parts_to_anthropic_blocks( - content.get("content") or [] - ) - # Fallback text if the conversion produced nothing usable. + multimodal_blocks = _content_parts_to_anthropic_blocks(content.get("content") or []) if not multimodal_blocks and content.get("text_summary"): - multimodal_blocks = [ - {"type": "text", "text": str(content["text_summary"])} - ] + multimodal_blocks = [_text_block(str(content["text_summary"]))] elif isinstance(content, list): converted = _content_parts_to_anthropic_blocks(content) if any(b.get("type") == "image" for b in converted): multimodal_blocks = converted - # Back-compat: some callers stash blocks under a private key. - if multimodal_blocks is None: + if multimodal_blocks is None: # back-compat: blocks stashed under a private key stashed = m.get("_anthropic_content_blocks") if isinstance(stashed, list) and stashed: text_content = content if isinstance(content, str) and content.strip() else None - multimodal_blocks = ( - [{"type": "text", "text": text_content}] + stashed - if text_content else list(stashed) - ) + multimodal_blocks = [_text_block(text_content)] + stashed if text_content else list(stashed) if multimodal_blocks: - result_content: Any = multimodal_blocks - elif isinstance(content, str): - result_content = content - else: - result_content = json.dumps(content) if content else "(no output)" - if not result_content: - result_content = "(no output)" + return multimodal_blocks + if isinstance(content, str): + return content or "(no output)" + return json.dumps(content) if content else "(no output)" + + +def _convert_tool_message_to_result(result: List[Dict[str, Any]], m: Dict[str, Any]) -> None: + """Append a tool_result to ``result``, merging into a trailing tool_result user + message when there is one. Mutates ``result`` in place.""" tool_result = { "type": "tool_result", "tool_use_id": _sanitize_tool_id(m.get("tool_call_id", "")), - "content": result_content, + "content": _tool_result_content(m), } - if isinstance(m.get("cache_control"), dict): - tool_result["cache_control"] = dict(m["cache_control"]) - # Merge consecutive tool results into one user message - if ( - result - and result[-1]["role"] == "user" - and isinstance(result[-1]["content"], list) - and result[-1]["content"] - and result[-1]["content"][0].get("type") == "tool_result" - ): - result[-1]["content"].append(tool_result) + cache_control = _cache_control_of(m) + if cache_control is not None: + tool_result["cache_control"] = dict(cache_control) + last = result[-1] if result else {} + if last.get("role") == "user" and isinstance(last.get("content"), list) and last["content"] \ + and last["content"][0].get("type") == "tool_result": + last["content"].append(tool_result) else: result.append({"role": "user", "content": [tool_result]}) def _convert_user_message(content: Any) -> Dict[str, Any]: - """Validate and convert a user message to anthropic format.""" + """Validate and convert a user message to Anthropic format.""" if isinstance(content, list): - converted_blocks = _convert_content_to_anthropic(content) kept_blocks = _fix_blank_text_blocks_in_list( - converted_blocks, - placeholder_text="(empty message)", - msg_index=-1, - role="user", - location="_convert_user_message", + _convert_content_to_anthropic(content), placeholder_text="(empty message)", + msg_index=-1, role="user", location="_convert_user_message", ) return {"role": "user", "content": kept_blocks} - else: - if not content or (isinstance(content, str) and not content.strip()): - content = "(empty message)" - return {"role": "user", "content": content} + if not content or (isinstance(content, str) and not content.strip()): + content = "(empty message)" + return {"role": "user", "content": content} + + +# --------------------------------------------------------------------------- +# Whole-list passes +# --------------------------------------------------------------------------- def _strip_orphaned_tool_blocks(result: List[Dict[str, Any]]) -> None: """Strip tool_use blocks with no matching tool_result, and vice versa. - Context compression or session truncation can remove either side of a - tool-call pair, or insert messages between a tool_use and its result. - Anthropic requires each tool_use to have a matching tool_result in the - IMMEDIATELY FOLLOWING user message — a global ID match is not enough. - Mutates ``result`` in place. + Compression/truncation can remove either side of a pair or insert messages + between them. Anthropic requires the tool_result in the IMMEDIATELY FOLLOWING + user message — a global id match is not enough. Mutates ``result`` in place. """ - # Pass 1: For each assistant message with tool_use blocks, check that - # EACH tool_use ID has a matching tool_result in the immediately following - # user message. Strip tool_use blocks that lack an adjacent result — - # Anthropic rejects non-adjacent pairs with HTTP 400 even when the IDs - # match somewhere later in the conversation. + # Pass 1: tool_use without an adjacent result. for i, m in enumerate(result): if m.get("role") != "assistant" or not isinstance(m.get("content"), list): continue - tool_use_ids_in_turn = { - b.get("id") - for b in m["content"] - if isinstance(b, dict) and b.get("type") == "tool_use" - } + tool_use_ids_in_turn = {b.get("id") for b in m["content"] if _block_type(b) == "tool_use"} if not tool_use_ids_in_turn: continue - - # Collect result IDs from the immediately following user message only. adjacent_result_ids: set = set() if i + 1 < len(result): nxt = result[i + 1] if nxt.get("role") == "user" and isinstance(nxt.get("content"), list): - for block in nxt["content"]: - if isinstance(block, dict) and block.get("type") == "tool_result": - adjacent_result_ids.add(block.get("tool_use_id")) - + adjacent_result_ids = {b.get("tool_use_id") for b in nxt["content"] if _block_type(b) == "tool_result"} orphaned = tool_use_ids_in_turn - adjacent_result_ids if not orphaned: continue - - kept = [ - b - for b in m["content"] - if not (isinstance(b, dict) and b.get("type") == "tool_use" and b.get("id") in orphaned) - ] - # If stripping an orphaned tool_use mutated a turn that also carries a - # signed thinking block, that block's Anthropic signature was computed - # against the ORIGINAL (un-stripped) turn content and is now invalid. - # Anthropic rejects the replayed turn with HTTP 400 "thinking blocks in - # the latest assistant message cannot be modified". Flag the turn so - # _manage_thinking_signatures can demote the dead signature instead of - # replaying it verbatim. See hermes-agent: extended-thinking + parallel - # tool batch interrupted mid-flight → non-retryable 400 crash-loop. - if len(kept) != len(m["content"]) and any( - isinstance(b, dict) and b.get("type") in {"thinking", "redacted_thinking"} - for b in m["content"] - ): + kept = [b for b in m["content"] if not (_block_type(b) == "tool_use" and b.get("id") in orphaned)] + # A signed thinking block on this turn was signed against the ORIGINAL + # content and is now dead (400 "thinking blocks in the latest assistant + # message cannot be modified"). Flag so _manage_thinking_signatures demotes it. + if len(kept) != len(m["content"]) and _has_block_type(m["content"], _THINKING_TYPES): m["_thinking_signature_invalidated"] = True - m["content"] = kept if kept else [{"type": "text", "text": "(tool call removed)"}] - - # Pass 2: Rebuild the set of tool_use IDs that survived pass 1, then - # strip tool_result blocks that no longer have any matching tool_use - # anywhere in the conversation. - surviving_tool_use_ids: set = set() - for m in result: - if m.get("role") == "assistant" and isinstance(m.get("content"), list): - for block in m["content"]: - if isinstance(block, dict) and block.get("type") == "tool_use": - surviving_tool_use_ids.add(block.get("id")) + m["content"] = kept if kept else [_text_block("(tool call removed)")] + # Pass 2: tool_result whose tool_use no longer exists anywhere. + surviving_tool_use_ids = { + b.get("id") + for m in result + if m.get("role") == "assistant" and isinstance(m.get("content"), list) + for b in m["content"] + if _block_type(b) == "tool_use" + } for m in result: if m.get("role") != "user" or not isinstance(m.get("content"), list): continue new_content = [ - b - for b in m["content"] - if not (isinstance(b, dict) and b.get("type") == "tool_result") - or b.get("tool_use_id") in surviving_tool_use_ids + b for b in m["content"] + if _block_type(b) != "tool_result" or b.get("tool_use_id") in surviving_tool_use_ids ] if len(new_content) != len(m["content"]): - m["content"] = new_content if new_content else [{"type": "text", "text": "(tool result removed)"}] + m["content"] = new_content if new_content else [_text_block("(tool result removed)")] + + +def _concat_content(prev: Any, curr: Any) -> Any: + """Merge two message contents: str+str joined by newline, list+list concatenated, + mixed shapes promoted to block lists.""" + if isinstance(prev, str) and isinstance(curr, str): + return prev + "\n" + curr + if isinstance(prev, str): + prev = [_text_block(prev)] + if isinstance(curr, str): + curr = [_text_block(curr)] + return prev + curr def _merge_consecutive_roles(result: List[Dict[str, Any]]) -> List[Dict[str, Any]]: - """Merge consecutive same-role messages to enforce Anthropic alternation. - - Returns a new list (caller must rebind ``result``). - """ + """Merge consecutive same-role messages to enforce alternation. Returns a new list.""" fixed = [] for m in result: - if fixed and fixed[-1]["role"] == m["role"]: - if m["role"] == "user": - prev_content = fixed[-1]["content"] - curr_content = m["content"] - if isinstance(prev_content, str) and isinstance(curr_content, str): - fixed[-1]["content"] = prev_content + "\n" + curr_content - elif isinstance(prev_content, list) and isinstance(curr_content, list): - fixed[-1]["content"] = prev_content + curr_content - else: - if isinstance(prev_content, str): - prev_content = [{"type": "text", "text": prev_content}] - if isinstance(curr_content, str): - curr_content = [{"type": "text", "text": curr_content}] - fixed[-1]["content"] = prev_content + curr_content - else: - # Consecutive assistant messages — merge text content. - # Propagate the orphan-strip signature-invalidation flag onto the - # surviving (prev) dict so _manage_thinking_signatures still sees it. - if m.get("_thinking_signature_invalidated"): - fixed[-1]["_thinking_signature_invalidated"] = True - # Drop thinking blocks from the *second* message: their - # signature was computed against a different turn boundary - # and becomes invalid once merged. - if isinstance(m["content"], list): - m["content"] = [ - b for b in m["content"] - if not (isinstance(b, dict) and b.get("type") in {"thinking", "redacted_thinking"}) - ] - prev_blocks = fixed[-1]["content"] - curr_blocks = m["content"] - if isinstance(prev_blocks, list) and isinstance(curr_blocks, list): - fixed[-1]["content"] = prev_blocks + curr_blocks - elif isinstance(prev_blocks, str) and isinstance(curr_blocks, str): - fixed[-1]["content"] = prev_blocks + "\n" + curr_blocks - else: - if isinstance(prev_blocks, str): - prev_blocks = [{"type": "text", "text": prev_blocks}] - if isinstance(curr_blocks, str): - curr_blocks = [{"type": "text", "text": curr_blocks}] - fixed[-1]["content"] = prev_blocks + curr_blocks - else: + if not (fixed and fixed[-1]["role"] == m["role"]): fixed.append(m) + continue + if m["role"] != "user": + # Keep the orphan-strip flag visible to _manage_thinking_signatures. + if m.get("_thinking_signature_invalidated"): + fixed[-1]["_thinking_signature_invalidated"] = True + # The second message's thinking blocks were signed against a + # different turn boundary and become invalid once merged. + if isinstance(m["content"], list): + m["content"] = [b for b in m["content"] if _block_type(b) not in _THINKING_TYPES] + fixed[-1]["content"] = _concat_content(fixed[-1]["content"], m["content"]) return fixed -def _manage_thinking_signatures( - result: List[Dict[str, Any]], base_url: str | None, model: str | None -) -> None: - """Strip or preserve thinking blocks based on endpoint type. +def _manage_thinking_signatures(result: List[Dict[str, Any]], base_url: str | None, model: str | None) -> None: + """Strip or preserve thinking blocks per endpoint. Mutates ``result`` in place. - Anthropic signs thinking blocks against the full turn content. - Any upstream mutation (context compression, session truncation, orphan - stripping, message merging) invalidates the signature, causing HTTP 400 - "Invalid signature in thinking block". - - Signatures are Anthropic-proprietary. Third-party endpoints (MiniMax, - Azure AI Foundry, AWS Bedrock, self-hosted proxies) cannot validate them - and will reject them outright. Kimi's /coding and DeepSeek's /anthropic - endpoints speak the Anthropic protocol upstream but require unsigned - thinking blocks (synthesised from ``reasoning_content``) to round-trip on - replayed assistant tool-call messages. See hermes-agent#13848 (Kimi) and - hermes-agent#16748 (DeepSeek). - - Nous Portal's ``/v1/messages`` route is the exception among third-party - hosts: it proxies Claude to Anthropic/Vertex/Bedrock and validates the - same signed thinking blocks. Sticky ``session_id`` keeps a conversation - on one upstream instance so those signatures stay warm — stripping them - here would 400 the first tool-loop turn ("thinking must be passed back"). - Portal therefore takes the native Anthropic replay path below. - - Mutates ``result`` in place. + Anthropic signs thinking blocks against the full turn; any upstream mutation + invalidates them (400 "Invalid signature in thinking block"), so on direct + Anthropic only the LATEST assistant turn keeps signed blocks. Signatures are + proprietary: third-party endpoints strip all thinking. Kimi replays as-is; + DeepSeek needs unsigned blocks round-tripped but rejects signed ones. Nous + Portal proxies Claude with sticky sessions and validates the same signatures, + so it takes the native path despite not being anthropic.com. """ - _THINKING_TYPES = frozenset(("thinking", "redacted_thinking")) - # Portal speaks Anthropic's thinking contract end-to-end; do not treat it - # as a signature-blind proxy even though the host is not anthropic.com. - _is_third_party = ( - _is_third_party_anthropic_endpoint(base_url) - and not _is_nous_portal_endpoint(base_url) - ) - - last_assistant_idx = None - for i in range(len(result) - 1, -1, -1): - if result[i].get("role") == "assistant": - last_assistant_idx = i - break + is_third_party = _is_third_party_anthropic_endpoint(base_url) and not _is_nous_portal_endpoint(base_url) + is_kimi = _is_kimi_family_endpoint(base_url, model) + is_deepseek = _is_deepseek_anthropic_endpoint(base_url) + last_assistant_idx = next((i for i in range(len(result) - 1, -1, -1) if result[i].get("role") == "assistant"), None) for idx, m in enumerate(result): if m.get("role") != "assistant" or not isinstance(m.get("content"), list): continue - - if _is_kimi_family_endpoint(base_url, model): - # Kimi does not enforce thinking signatures — replay as-is - # (shared cleanup below still strips cache markers + the internal flag). - pass - elif _is_deepseek_anthropic_endpoint(base_url): - # DeepSeek: strip signed, preserve unsigned. - new_content = [] - for b in m["content"]: - if not isinstance(b, dict) or b.get("type") not in _THINKING_TYPES: - new_content.append(b) - continue - if b.get("signature") or b.get("data"): - # Signed (or redacted-with-data) — upstream can't validate, strip. - continue - new_content.append(b) - m["content"] = new_content or [{"type": "text", "text": "(empty)"}] - elif _is_third_party or idx != last_assistant_idx: - # Third-party: strip ALL thinking blocks (signatures are proprietary). - # Direct Anthropic: strip from non-latest assistant messages only. - stripped = [ + if is_kimi: + pass # shared cleanup below still strips cache markers + the flag + elif is_deepseek: + # Strip signed (or redacted-with-data), keep unsigned. + new_content = [ b for b in m["content"] - if not (isinstance(b, dict) and b.get("type") in _THINKING_TYPES) + if _block_type(b) not in _THINKING_TYPES or not (b.get("signature") or b.get("data")) ] - m["content"] = stripped or [{"type": "text", "text": "(thinking elided)"}] + m["content"] = new_content or [_text_block("(empty)")] + elif is_third_party or idx != last_assistant_idx: + stripped = [b for b in m["content"] if _block_type(b) not in _THINKING_TYPES] + m["content"] = stripped or [_text_block("(thinking elided)")] else: - # Latest assistant on direct Anthropic: keep signed, downgrade unsigned - # to text so the reasoning isn't lost. - # - # Exception: if orphan-stripping (or another structural mutation) removed - # a tool_use block from THIS turn, every thinking signature on it was - # computed against the original turn content and is now dead. Anthropic - # rejects the turn either way — replaying the signed block 400s with - # "thinking blocks in the latest assistant message cannot be modified", - # and a bare signed block with no following tool_use is also invalid. - # Demote ALL thinking blocks on this turn to text so the turn replays - # cleanly and the model can re-plan from the surviving tool results. + # Latest assistant on direct Anthropic: keep signed, demote unsigned to + # text so the reasoning isn't lost. If orphan-stripping mutated THIS + # turn every signature is dead (and a bare signed block with no + # tool_use is also invalid), so demote ALL of them. signature_dead = bool(m.get("_thinking_signature_invalidated")) new_content = [] for b in m["content"]: - if not isinstance(b, dict) or b.get("type") not in _THINKING_TYPES: + if _block_type(b) not in _THINKING_TYPES: new_content.append(b) continue - if signature_dead: - thinking_text = b.get("thinking", "") - if thinking_text: - new_content.append({"type": "text", "text": thinking_text}) - continue - if b.get("type") == "redacted_thinking": - # Redacted blocks use 'data' for the signature payload — - # drop the block when 'data' is missing (can't be validated). - if b.get("data"): - new_content.append(b) - elif b.get("signature"): + is_redacted = b.get("type") == "redacted_thinking" + signed = b.get("data") if is_redacted else b.get("signature") # redacted 'data' IS the signature + if signed and not signature_dead: new_content.append(b) - else: - thinking_text = b.get("thinking", "") - if thinking_text: - new_content.append({"type": "text", "text": thinking_text}) - m["content"] = new_content or [{"type": "text", "text": "(empty)"}] + elif (signature_dead or not is_redacted) and b.get("thinking"): + new_content.append(_text_block(b["thinking"])) # demote to plain text + # else: redacted_thinking without data — unverifiable, dropped + m["content"] = new_content or [_text_block("(empty)")] - # Strip cache_control from any remaining thinking/redacted_thinking - # blocks — cache markers interfere with signature validation. + # cache_control on thinking blocks interferes with signature validation. for b in m["content"]: - if isinstance(b, dict) and b.get("type") in _THINKING_TYPES: + if _block_type(b) in _THINKING_TYPES: b.pop("cache_control", None) - - # Drop the internal bookkeeping flag — it must never reach the API payload. - m.pop("_thinking_signature_invalidated", None) + m.pop("_thinking_signature_invalidated", None) # internal flag, never on the wire def _evict_old_screenshots(result: List[Dict[str, Any]]) -> None: - """Keep only the most recent ``_MAX_KEEP_IMAGES`` computer-use screenshots. - - Base64 images cost ~1,465 tokens each and accumulate across tool calls. - Walk backward, keep the most recent N, replace older ones with a placeholder. - - Mutates ``result`` in place. - """ - _MAX_KEEP_IMAGES = 3 - _image_count = 0 + """Keep only the 3 most recent computer-use screenshots (~1,465 tokens each); + older images become a placeholder text block. Mutates ``result`` in place.""" + image_count = 0 for msg in reversed(result): content = msg.get("content") if not isinstance(content, list): continue for block in content: - if not isinstance(block, dict) or block.get("type") != "tool_result": + if _block_type(block) != "tool_result": continue inner = block.get("content") - if not isinstance(inner, list): + if not isinstance(inner, list) or not _has_block_type(inner, {"image"}): continue - has_image = any( - isinstance(b, dict) and b.get("type") == "image" - for b in inner - ) - if not has_image: - continue - _image_count += 1 - if _image_count > _MAX_KEEP_IMAGES: + image_count += 1 + if image_count > 3: block["content"] = [ - b if b.get("type") != "image" - else {"type": "text", "text": "[screenshot removed to save context]"} + b if b.get("type") != "image" else _text_block("[screenshot removed to save context]") for b in inner ] def _ensure_leading_user_turn(result: List[Dict[str, Any]]) -> None: - """Anthropic requires messages[0] to have role=user. + """Anthropic requires messages[0].role == user; prepend a placeholder turn otherwise. - After a second context compaction on the auto path the summary can be - emitted as role=assistant with nothing in front of it (the system prompt - lives outside messages[] or is extracted into the separate ``system`` - param), so messages[0] ends up assistant and the Messages API rejects - the request with HTTP 400 — often masked by a misleading - "tool_use ids were found without tool_result blocks" error (#52160). - - Mirror the Bedrock Converse adapter, which unconditionally prepends a - minimal user turn when the first message is not user - (convert_messages_to_converse). - - The inserted text block must be non-whitespace: Anthropic separately - rejects any text content block whose text is empty or whitespace-only - ("text content blocks must contain non-whitespace text"), so a single - space here traded the "leading assistant turn" 400 for that one (#69512 - class). Uses the same placeholder as every other synthesized filler - block in this module for consistency. + A second auto-compaction can leave a role=assistant summary first, which the + API rejects (often masked as a misleading tool_use/tool_result 400). The filler + must be non-whitespace text or it trades that 400 for the blank-block one. """ if result and result[0].get("role") != "user": - result.insert( - 0, {"role": "user", "content": [{"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER}]} - ) + result.insert(0, {"role": "user", "content": [_text_block(_EMPTY_TEXT_PLACEHOLDER)]}) def _fix_blank_text_blocks_in_list( @@ -1051,63 +736,35 @@ def _fix_blank_text_blocks_in_list( role: Any, location: str, ) -> List[Any]: - """Drop blank/whitespace-only text blocks from ``blocks``, in place logic. - - Non-text blocks (tool_use, tool_result, image, document, thinking, …) - and the relative order of everything else are left untouched. A - cache_control marker riding on a dropped block is relocated onto the - last surviving text/tool_use block so a breakpoint is never silently - lost. If nothing survives, a single non-blank placeholder text block - takes the dropped blocks' place (carrying the relocated cache_control, - if any) so the message never has empty content. - - Returns a new list; does not mutate ``blocks``. - """ + """Drop blank text blocks; relocate any cache_control they carried onto the last + surviving cacheable block; if nothing survives, substitute one placeholder block + (carrying the relocated marker). Non-text blocks and order are untouched. + Returns a new list; logs structure only (never text).""" kept: List[Any] = [] relocated_cache_control = None for block_index, blk in enumerate(blocks): - if ( - isinstance(blk, dict) - and blk.get("type") == "text" - and not (isinstance(blk.get("text"), str) and blk["text"].strip()) - ): - if isinstance(blk.get("cache_control"), dict): + if _is_blank_text_block(blk): + if _cache_control_of(blk) is not None: relocated_cache_control = blk["cache_control"] logger.warning( "Pre-call sanitizer: dropped blank text content block " - "(message_index=%d role=%s location=%s block_index=%d " - "block_type=text)", - msg_index, - role, - location, - block_index, + "(message_index=%d role=%s location=%s block_index=%d block_type=text)", + msg_index, role, location, block_index, ) continue kept.append(blk) if not kept: - placeholder: Dict[str, Any] = {"type": "text", "text": placeholder_text} - if relocated_cache_control is not None: - placeholder["cache_control"] = relocated_cache_control - kept.append(placeholder) - elif relocated_cache_control is not None: - _apply_assistant_cache_control_to_last_cacheable_block(kept, relocated_cache_control) + kept.append(_text_block(placeholder_text)) + _apply_assistant_cache_control_to_last_cacheable_block(kept, relocated_cache_control) return kept def _scrub_blank_text_blocks(result: List[Dict[str, Any]]) -> None: - """Final provider-boundary guard against blank Anthropic text blocks. + """Final boundary guard against blank text blocks (HTTP 400 "text content blocks + must contain non-whitespace text"), including inside tool_result content. - Anthropic rejects any text content block whose ``text`` is empty or - whitespace-only with HTTP 400 ("text content blocks must contain - non-whitespace text"). ``_convert_assistant_message``, - ``_convert_user_message`` and ``_ensure_leading_user_turn`` already - avoid emitting these for the paths that build them, but this pass runs - last — after every other transform in ``convert_messages_to_anthropic`` - — so a blank block from any current or future producer (including one - nested inside a ``tool_result``'s own content list) never reaches the - wire. Diagnostics are structural only: message index, role, content - location, block index/type. Never logs message text, tool arguments, - tokens, or credentials. Mutates ``result`` in place. + Runs LAST so a blank block from any current or future producer never reaches + the wire. Diagnostics are structural only. Mutates ``result`` in place. """ for msg_index, msg in enumerate(result): if not isinstance(msg, dict): @@ -1116,29 +773,44 @@ def _scrub_blank_text_blocks(result: List[Dict[str, Any]]) -> None: content = msg.get("content") if not isinstance(content, list) or not content: continue - placeholder_text = _EMPTY_TEXT_PLACEHOLDER if role == "assistant" else "(empty message)" new_content = _fix_blank_text_blocks_in_list( content, - placeholder_text=placeholder_text, - msg_index=msg_index, - role=role, - location="content", + placeholder_text=_EMPTY_TEXT_PLACEHOLDER if role == "assistant" else "(empty message)", + msg_index=msg_index, role=role, location="content", ) for blk in new_content: - if not isinstance(blk, dict) or blk.get("type") != "tool_result": + if _block_type(blk) != "tool_result": continue inner = blk.get("content") if isinstance(inner, list) and inner: blk["content"] = _fix_blank_text_blocks_in_list( - inner, - placeholder_text="(no output)", - msg_index=msg_index, - role=role, - location="tool_result", + inner, placeholder_text="(no output)", msg_index=msg_index, role=role, location="tool_result", ) msg["content"] = new_content +def _convert_system_content(content: Any) -> Any: + """System message content -> Anthropic ``system`` param (str, or block list when + cache_control is present). + + With cache markers the blocks are copied (never mutating the caller's dicts) + and blank text is replaced by the placeholder: Anthropic rejects blank system + blocks too, and a blank block carrying a breakpoint can't simply be dropped. + """ + if not isinstance(content, list): + return content + if not any(p.get("cache_control") for p in content if isinstance(p, dict)): + return "\n".join(p["text"] for p in content if p.get("type") == "text") + system = [] + for p in content: + if not isinstance(p, dict): + continue + if p.get("type") == "text" and isinstance(p.get("text"), str) and not p["text"].strip(): + p = {**p, "text": _EMPTY_TEXT_PLACEHOLDER} + system.append(p) + return system + + def convert_messages_to_anthropic( messages: List[Dict], base_url: str | None = None, @@ -1146,20 +818,11 @@ def convert_messages_to_anthropic( ) -> Tuple[Optional[Any], List[Dict]]: """Convert OpenAI-format messages to Anthropic format. - Returns (system_prompt, anthropic_messages). - System messages are extracted since Anthropic takes them as a separate param. - system_prompt is a string or list of content blocks (when cache_control present). - - When *base_url* is provided and points to a third-party Anthropic-compatible - endpoint, all thinking block signatures are stripped. Signatures are - Anthropic-proprietary — third-party endpoints cannot validate them and will - reject them with HTTP 400 "Invalid signature in thinking block". - - When *model* is provided and matches the Kimi / Moonshot family (or - *base_url* is a Kimi / Moonshot host), unsigned thinking blocks - synthesised from ``reasoning_content`` are preserved on replayed - assistant tool-call messages — Kimi requires the field to exist, even - if empty. + Returns ``(system, messages)``: system is extracted into its own param (a + string, or a block list when cache_control is present). ``base_url``/``model`` + drive thinking-signature policy — third-party endpoints strip signatures + (proprietary, they 400 on them); Kimi-family endpoints/models keep unsigned + reasoning_content-derived blocks, which Kimi requires even when empty. """ system = None result: List[Dict[str, Any]] = [] @@ -1167,52 +830,14 @@ def convert_messages_to_anthropic( for m in messages: role = m.get("role", "user") content = m.get("content", "") - if role == "system": - if isinstance(content, list): - # Preserve cache_control markers on content blocks - has_cache = any( - p.get("cache_control") for p in content if isinstance(p, dict) - ) - if has_cache: - # Copy blocks before coercing so the caller's message - # dicts are never mutated, then replace blank/whitespace - # text with the shared non-whitespace placeholder — - # Anthropic rejects a blank system text block with the - # same HTTP 400 as message blocks ("text content blocks - # must contain non-whitespace text"), and a blank block - # carrying a cache_control breakpoint cannot simply be - # dropped (#70909). - system = [] - for p in content: - if not isinstance(p, dict): - continue - if ( - p.get("type") == "text" - and isinstance(p.get("text"), str) - and not p["text"].strip() - ): - p = dict(p) - p["text"] = _EMPTY_TEXT_PLACEHOLDER - system.append(p) - else: - system = "\n".join( - p["text"] for p in content if p.get("type") == "text" - ) - else: - system = content - continue - - if role == "assistant": + system = _convert_system_content(content) + elif role == "assistant": result.append(_convert_assistant_message(m)) - continue - - if role == "tool": + elif role == "tool": _convert_tool_message_to_result(result, m) - continue - - # Regular user message - result.append(_convert_user_message(content)) + else: + result.append(_convert_user_message(content)) _strip_orphaned_tool_blocks(result) result = _merge_consecutive_roles(result) @@ -1222,4 +847,3 @@ def convert_messages_to_anthropic( _scrub_blank_text_blocks(result) return system, result -