Files
hermes-agent/agent/anthropic_adapter.py
Teknium 267cdbecf3 refactor(agent/adapters): simplify anthropic adapter, message convert, endpoints (-857 LOC)
Table-driven model capability checks and beta-header assembly, shared
_cache_control_of/_block_type/_image_block_from_data_url helpers in the message
converter, extracted _apply_claude_code_identity/_base_client_kwargs.
convert_messages_to_anthropic / convert_tools_to_anthropic and request kwargs
verified byte-identical against merge-base.
2026-09-02 13:29:47 -07:00

929 lines
41 KiB
Python

"""Anthropic Messages API adapter for Hermes Agent.
Translates between Hermes's internal OpenAI-style message format and
Anthropic's Messages API; all provider-specific logic is isolated here.
Auth supports:
- Regular API keys (sk-ant-api*) -> x-api-key header
- OAuth setup-tokens (sk-ant-oat*) -> Bearer auth + beta header
- Claude Code credentials (~/.claude.json or ~/.claude/.credentials.json) -> Bearer auth
"""
import logging
import math
import re
import subprocess
from pathlib import Path # noqa: F401 (tests patch ``anthropic_adapter.Path.home``)
from typing import Any, Dict, List, Optional
from utils import normalize_proxy_env_vars
# This module keeps client construction and the Messages API call itself; the
# three surfaces it used to inline live next to it and are re-exported below so
# long-standing ``from agent.anthropic_adapter import ...`` imports keep resolving:
# agent/anthropic_endpoints.py base-URL/endpoint-family predicates
# agent/anthropic_message_convert.py OpenAI -> Anthropic payload conversion
# agent/anthropic_credentials.py credential sources, OAuth, refresh commit
from agent.anthropic_endpoints import ( # noqa: F401
_KIMI_FAMILY_EXACT_SLUGS, _KIMI_FAMILY_MODEL_PREFIXES, _base_url_needs_context_1m_beta,
_is_azure_anthropic_endpoint, _is_deepseek_anthropic_endpoint, _is_kimi_coding_endpoint,
_is_kimi_family_endpoint, _is_minimax_anthropic_endpoint, _is_nous_portal_endpoint,
_is_opencode_endpoint, _is_third_party_anthropic_endpoint, _model_name_is_kimi_family,
_normalize_base_url_text, _requires_bearer_auth
)
from agent.anthropic_message_convert import ( # noqa: F401
_EMPTY_TEXT_PLACEHOLDER, _apply_assistant_cache_control_to_last_cacheable_block,
_content_parts_to_anthropic_blocks, _convert_assistant_message, _convert_content_part_to_anthropic,
_convert_content_to_anthropic, _convert_tool_message_to_result, _convert_user_message,
_ensure_leading_user_turn, _evict_old_screenshots, _extract_preserved_thinking_blocks,
_fix_blank_text_blocks_in_list, _image_source_from_openai_url, _is_bedrock_model_id,
_manage_thinking_signatures, _merge_consecutive_roles, _normalize_tool_input_schema, _safe_text,
_sanitize_replay_block, _sanitize_tool_id, _scrub_blank_text_blocks, _strip_orphaned_tool_blocks,
_to_plain_data, convert_messages_to_anthropic, convert_tools_to_anthropic, normalize_model_name
)
from agent.anthropic_credentials import ( # noqa: F401
_OAUTH_CLIENT_ID, _OAUTH_REDIRECT_URI, _OAUTH_SCOPES, _OAUTH_TOKEN_URL, _OAUTH_TOKEN_URLS,
_OAUTH_TOKEN_USER_AGENT, CredentialPersistError, _generate_pkce, _get_hermes_oauth_file, _getenv,
_is_oauth_token, _prefer_refreshable_claude_code_token, _read_claude_code_credentials_from_file,
_read_claude_code_credentials_from_keychain, _refresh_oauth_token, _resolve_anthropic_pool_token,
_resolve_claude_code_token_from_credentials, _write_claude_code_credentials,
_write_hermes_oauth_credentials, claude_code_credentials_path, is_claude_code_token_valid,
is_rotation_consumed_uncommitted, mark_rotation_consumed_uncommitted, read_claude_code_credentials,
read_hermes_oauth_credentials, refresh_anthropic_oauth_pure, resolve_anthropic_token,
run_hermes_oauth_login_pure, run_oauth_setup_token
)
try:
import hermes_cli as _hermes_cli
_HERMES_VERSION = str(_hermes_cli.__version__)
except Exception:
_HERMES_VERSION = "0.0.0"
# ``import anthropic`` is deliberately NOT at module top: the SDK costs ~220 ms
# of imports and every usage site is a cold user-triggered path. ``...`` is the
# "not yet tried" sentinel; None means tried and missing.
_anthropic_sdk: Any = ...
def _get_anthropic_sdk():
"""Return the ``anthropic`` SDK module, importing lazily. None if not installed."""
global _anthropic_sdk
if _anthropic_sdk is ...:
try:
from tools.lazy_deps import ensure as _lazy_ensure
_lazy_ensure("provider.anthropic", prompt=False)
except Exception: # ImportError or FeatureUnavailable — fall through to the import below
pass
try:
import anthropic as _sdk
_anthropic_sdk = _sdk
except ImportError:
_anthropic_sdk = None
return _anthropic_sdk
def _require_sdk(purpose: str, verb: str = "Install it with"):
"""``_get_anthropic_sdk()`` or ImportError naming the feature that needs it."""
sdk = _get_anthropic_sdk()
if sdk is None:
raise ImportError(
f"The 'anthropic' package is required for {purpose}. "
f"{verb}: pip install 'anthropic>=0.39.0'"
)
return sdk
logger = logging.getLogger(__name__)
THINKING_BUDGET = {"xhigh": 32000, "high": 16000, "medium": 8000, "low": 4000}
# Hermes effort -> Anthropic adaptive-thinking effort (output_config.effort).
# 4.7+ exposes low/medium/high/xhigh/max; Opus/Sonnet 4.6 have no xhigh, so
# callers downgrade xhigh->max there (see _supports_xhigh_effort). "minimal" is
# a legacy alias for low on every model.
ADAPTIVE_EFFORT_MAP = {
"ultra": "max",
"max": "max",
"xhigh": "xhigh",
"high": "high",
"medium": "medium",
"low": "low",
"minimal": "low",
}
# ── Anthropic thinking-mode classification ────────────────────────────
# Claude 4.6 replaced budget-based extended thinking with *adaptive* thinking;
# 4.7 additionally forbids the manual ``thinking`` block and drops
# temperature/top_p/top_k. Newer releases (4.8, named models like claude-fable-5)
# share no common version substring, so an allowlist of "modern" versions would
# go stale and silently route a new model down the legacy path. We therefore
# DEFAULT unknown Claude models to the modern contract and keep explicit
# *legacy* lists (mirroring _get_anthropic_max_output's default-to-newest).
# Non-Claude Anthropic-Messages models (minimax, qwen3, GLM, ...) are not Claude
# and fall through to the legacy manual-thinking path, which is what they need.
# Older Claude families that need manual thinking (budget_tokens only).
_LEGACY_MANUAL_THINKING_CLAUDE_SUBSTRINGS = (
"claude-3", # 3, 3.5, 3.7
"claude-opus-4-0", "claude-opus-4.0", "claude-opus-4-1", "claude-opus-4.1",
"claude-sonnet-4-0", "claude-sonnet-4.0",
"claude-opus-4-2025", "claude-sonnet-4-2025", # date-stamped 4.0 IDs
"claude-opus-4-5", "claude-opus-4.5",
"claude-sonnet-4-5", "claude-sonnet-4.5",
"claude-haiku-4-5", "claude-haiku-4.5",
)
# Adaptive families that reject the "xhigh" effort (arrived with Opus 4.7) and
# still accept sampling params.
_NO_XHIGH_CLAUDE_SUBSTRINGS = (
"claude-opus-4-6", "claude-opus-4.6",
"claude-sonnet-4-6", "claude-sonnet-4.6",
)
# Adaptive families where thinking is mandatory: ``thinking: {"type":
# "disabled"}`` answers HTTP 400 (Portal flags them ``reasoning.mandatory``).
# The failure is asymmetric — a missing entry 400s the turn, a spurious one
# only leaves thinking on — so when in doubt, add the family.
_MANDATORY_THINKING_CLAUDE_SUBSTRINGS = (
"claude-fable",
)
_FAST_MODE_SUPPORTED_SUBSTRINGS = ("opus-4-8", "opus-4.8", "opus-5")
def _is_claude_model(model: str | None) -> bool:
return "claude" in (model or "").lower()
def _model_matches(model: str, substrings) -> bool:
"""Case-insensitive substring match of ``model`` against a family list."""
m = model.lower()
return any(v in m for v in substrings)
# ── Max output token limits per Anthropic model ───────────────────────
# Anthropic requires max_tokens; a fixed 16384 starved thinking-enabled models
# (thinking tokens count toward the limit). Source: Anthropic docs + Cline catalog.
_ANTHROPIC_OUTPUT_LIMITS = {
"claude-fable": 128_000, # Mythos-class named models — 1M context, reasoning
"claude-sonnet-5": 128_000,
"claude-opus-4-8": 128_000,
"claude-opus-4-7": 128_000,
"claude-opus-4-6": 128_000,
"claude-sonnet-4-6": 64_000,
"claude-opus-4-5": 64_000,
"claude-sonnet-4-5": 64_000,
"claude-haiku-4-5": 64_000,
"claude-opus-4": 32_000,
"claude-sonnet-4": 64_000,
"claude-3-7-sonnet": 128_000,
"claude-3-5-sonnet": 8_192,
"claude-3-5-haiku": 8_192,
"claude-3-opus": 4_096,
"claude-3-sonnet": 4_096,
"claude-3-haiku": 4_096,
"minimax": 131_072, # third-party Anthropic-compatible
"qwen3": 65_536, # DashScope enforces max_tokens in [1, 65536]
}
# Unknown models get the highest current limit: future models are unlikely to
# have *less* output capacity.
_ANTHROPIC_DEFAULT_OUTPUT_LIMIT = 128_000
def _get_anthropic_max_output(model: str) -> int:
"""Max output tokens for ``model`` via longest substring match against
``_ANTHROPIC_OUTPUT_LIMITS`` (so date-stamped ids and ``:1m``/``:fast``
suffixes resolve, and ``claude-3-5-sonnet`` beats ``claude-3-5``). Dots are
normalized to hyphens so ``claude-opus-4.6`` matches ``claude-opus-4-6``.
"""
m = model.lower().replace(".", "-")
best_key = max((key for key in _ANTHROPIC_OUTPUT_LIMITS if key in m), key=len, default=None)
return _ANTHROPIC_OUTPUT_LIMITS[best_key] if best_key else _ANTHROPIC_DEFAULT_OUTPUT_LIMIT
def _resolve_positive_anthropic_max_tokens(value) -> Optional[int]:
"""``value`` floored to a positive int, or None when it is not a finite
positive number.
Anthropic 400s on max_tokens that are 0, negative, fractional or non-finite;
the ``max_tokens or fallback`` idiom catches 0 but lets ``-1``/``0.5``
through. Booleans are excluded explicitly (they subclass int).
"""
if isinstance(value, bool) or not isinstance(value, (int, float)):
return None
try:
if not math.isfinite(value):
return None
except Exception: # e.g. OverflowError for ints too large for float
return None
floored = int(value) # truncates toward zero for floats
return floored if floored > 0 else None
def _resolve_anthropic_messages_max_tokens(
requested,
model: str,
context_length: Optional[int] = None,
) -> int:
"""``requested`` when it is a positive finite number, else the model's output
ceiling. Raises ValueError if neither is positive. The context-window clamp
is the caller's job so the positive-value contract stays endpoint-agnostic.
"""
resolved = _resolve_positive_anthropic_max_tokens(requested)
if resolved is not None:
return resolved
fallback = _get_anthropic_max_output(model)
if fallback > 0:
return fallback
raise ValueError(
f"Anthropic Messages adapter requires a positive max_tokens value for "
f"model {model!r}; got {requested!r} and no model default resolved."
)
def _supports_adaptive_thinking(model: str) -> bool:
"""True for Claude models using adaptive thinking (4.6+): unknown Claude
models default to adaptive, the explicit legacy list stays manual, and
non-Claude models return False — except Kimi/Moonshot, whose Anthropic-
compatible endpoints implement the adaptive contract (incl. xhigh/display).
"""
if _model_name_is_kimi_family(model):
return True
if not _is_claude_model(model):
return False
return not _model_matches(model, _LEGACY_MANUAL_THINKING_CLAUDE_SUBSTRINGS)
def _supports_xhigh_effort(model: str) -> bool:
"""True for models accepting the 'xhigh' effort (Opus 4.7+). Opus/Sonnet 4.6
400 on it — callers downgrade xhigh->max when this returns False."""
return _supports_adaptive_thinking(model) and not _model_matches(model, _NO_XHIGH_CLAUDE_SUBSTRINGS)
def _accepts_thinking_disable(model: str) -> bool:
"""True when ``model`` accepts an explicit ``thinking: {"type": "disabled"}``.
Adaptive Claude thinks by default, so "off" only works if the disable is
sent; mandatory-thinking families 400 on it and keep the omit behavior.
Legacy manual-thinking models are opt-in via budget_tokens, so omission is
already off. Scoped to Claude: Kimi's documented disable is omission, and
sending it a new parameter on the strength of Claude's contract is a guess.
"""
return (
_is_claude_model(model)
and _supports_adaptive_thinking(model)
and not _model_matches(model, _MANDATORY_THINKING_CLAUDE_SUBSTRINGS)
)
def _forbids_sampling_params(model: str) -> bool:
"""True for models that 400 on any non-default temperature/top_p/top_k
(Opus 4.7 and later; unknown Claude defaults to forbidding). The 4.6 family
and the legacy manual-thinking families still accept them. Callers omit the
fields entirely — the API rejects anything non-null, even defaults."""
return _is_claude_model(model) and not _model_matches(
model, _NO_XHIGH_CLAUDE_SUBSTRINGS + _LEGACY_MANUAL_THINKING_CLAUDE_SUBSTRINGS
)
def _supports_fast_mode(model: str) -> bool:
"""True for models accepting ``speed: "fast"`` (Opus 4.8 / Opus 5, Claude API only).
Explicit allowlist, not a version floor: the matrix has flipped both ways.
Opus 4.6 had fast mode and lost it (requests silently run and bill at
standard speed, so listing it would show a toggle that does nothing); Opus
4.7 hard-400s on the param. Dedicated ``...-fast`` ids select fast inference
via the model field and must NOT also receive the speed parameter.
"""
return "-fast" not in model and any(v in model for v in _FAST_MODE_SUPPORTED_SUBSTRINGS)
# Beta headers safe on ordinary/native Anthropic requests. GA on Claude 4.6+
# (harmless no-op there) but older Claude and compatible endpoints still gate
# on them. Do NOT add ``context-1m-2025-08-07``: accounts without the
# long-context beta get HTTP 400 ("long context beta is not yet available for
# this subscription"), breaking short auxiliary calls. Bedrock/Azure still need
# it for 1M context and opt in on their own paths.
_COMMON_BETAS = [
"interleaved-thinking-2025-05-14",
"fine-grained-tool-streaming-2025-05-14",
]
# MiniMax's Anthropic-compatible endpoints fail tool-use requests when this beta
# is present.
_TOOL_STREAMING_BETA = "fine-grained-tool-streaming-2025-05-14"
_CONTEXT_1M_BETA = "context-1m-2025-08-07"
# Enables the ``speed: "fast"`` request parameter.
_FAST_MODE_BETA = "fast-mode-2026-02-01"
# Required for OAuth/subscription auth; matches Claude Code / pi-ai / OpenCode.
_OAUTH_ONLY_BETAS = [
"claude-code-20250219",
"oauth-2025-04-20",
]
# Claude Code identity — OAuth requests without it intermittently 500. Anthropic
# rejects OAuth requests whose user-agent version is too far behind the actual
# release, so the installed version is detected and this fallback kept current.
_CLAUDE_CODE_VERSION_FALLBACK = "2.1.74"
_claude_code_version_cache: Optional[str] = None
def _detect_claude_code_version() -> str:
"""Installed Claude Code version (``claude --version``), else the static fallback."""
for cmd in ("claude", "claude-code"):
try:
result = subprocess.run(
[cmd, "--version"],
capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5,
)
if result.returncode == 0 and result.stdout.strip():
version = result.stdout.strip().split()[0] # "2.1.74 (Claude Code)" or "2.1.74"
if version and version[0].isdigit():
return version
except Exception:
pass
return _CLAUDE_CODE_VERSION_FALLBACK
def _get_claude_code_version() -> str:
"""Lazily detect the installed Claude Code version when OAuth headers need it."""
global _claude_code_version_cache
if _claude_code_version_cache is None:
_claude_code_version_cache = _detect_claude_code_version()
return _claude_code_version_cache
_CLAUDE_CODE_SYSTEM_PREFIX = "You are Claude Code, Anthropic's official CLI for Claude."
_MCP_TOOL_PREFIX = "mcp__"
# Anthropic's OAuth billing classifier fingerprints certain Hermes tool
# schemas/prose as a third-party app and reroutes to the metered extra-usage
# lane (HTTP 400 "You're out of extra usage" on a valid subscription). Live A/B
# repros isolated two independent triggers — the ``session_search`` tool
# (schema/name/prose) and the ``memory`` tool (schema/name) — so both are
# aliased on the OAuth wire only; normalize_response reverses the mapping.
_OAUTH_TOOL_NAME_ALIASES = {
"session_search": "chat_history_lookup",
"memory": "context_notes",
}
_OAUTH_TOOL_NAME_REVERSE_ALIASES = {
wire_name: name for name, wire_name in _OAUTH_TOOL_NAME_ALIASES.items()
}
# Aliases ALSO safe to substitute in free-form prose (system prompt, tool
# descriptions). "memory" is ordinary English throughout the prompt and inside
# the memory tool's own parameter docs (an enum the model must emit verbatim),
# so rewriting it would corrupt guidance; a model that calls bare ``memory``
# still dispatches, since normalize_response resolves it through the registry.
_OAUTH_PROSE_ALIAS_NAMES = frozenset({"session_search"})
# Word-boundary matchers so a longer identifier containing the token (e.g.
# ``tools/session_search_tool.py`` in AGENTS.md) is left alone; ``\b`` treats
# ``_`` as a word char.
_OAUTH_PROSE_ALIAS_PATTERNS = tuple(
(re.compile(rf"\b{re.escape(name)}\b"), _OAUTH_TOOL_NAME_ALIASES[name])
for name in sorted(_OAUTH_PROSE_ALIAS_NAMES)
)
def _apply_oauth_prose_aliases(text: str) -> str:
"""Rewrite prose-safe tool-name tokens to their OAuth wire aliases."""
for pattern, wire_name in _OAUTH_PROSE_ALIAS_PATTERNS:
text = pattern.sub(wire_name, text)
return text
def _common_betas_for_base_url(
base_url: str | None,
*,
drop_context_1m_beta: bool = False,
) -> list[str]:
"""Beta headers safe for the configured endpoint.
MiniMax (Bearer-auth) rejects both the fine-grained-tool-streaming beta
(every tool-use message errors) and the 1M-context beta. Azure AI Foundry
also uses Bearer auth but keeps both — it needs the 1M beta for 1M context,
which native Anthropic does not get by default (some subscriptions reject
it; Bedrock opts in via its own client helper). ``drop_context_1m_beta``
strips the 1M beta after a subscription/endpoint rejected it.
"""
betas = list(_COMMON_BETAS)
if _base_url_needs_context_1m_beta(base_url) and not drop_context_1m_beta:
betas.append(_CONTEXT_1M_BETA)
if _is_minimax_anthropic_endpoint(base_url):
return [b for b in betas if b not in (_TOOL_STREAMING_BETA, _CONTEXT_1M_BETA)]
return betas
def _beta_header(betas: list) -> Dict[str, str]:
"""``{"anthropic-beta": ...}`` when there are betas, else ``{}``."""
return {"anthropic-beta": ",".join(betas)} if betas else {}
_ATTRIBUTION_HEADERS = {
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
"X-Title": "Hermes Agent",
}
def _attribution_headers() -> Dict[str, str]:
"""Same client-attribution set sent to OpenRouter / Vercel AI Gateway / Fireworks."""
return {**_ATTRIBUTION_HEADERS, "User-Agent": f"HermesAgent/{_HERMES_VERSION}"}
def _client_timeout(timeout):
"""httpx.Timeout with the caller's read timeout (default 900s) and a 10s connect."""
from httpx import Timeout
read = timeout if (isinstance(timeout, (int, float)) and timeout > 0) else 900.0
return Timeout(timeout=float(read), connect=10.0)
def _base_client_kwargs(base_url, timeout) -> tuple[str, Dict[str, Any]]:
"""Shared SDK constructor kwargs; returns ``(normalized_base_url, kwargs)``.
Retry is delegated to hermes's outer loop (``max_retries=0``): the SDK
default of 2 uses its own backoff that ignores Retry-After and double-
retries inside our loop, burning request slots against a bucket that won't
refill for minutes. Any trailing ``/v1`` is stripped because the SDK appends
``/v1/messages``. Azure endpoints need an ``api-version`` query param; it
goes through ``default_query`` so the base_url is not corrupted into
``/anthropic?api-version=.../v1/messages``.
"""
kwargs: Dict[str, Any] = {"timeout": _client_timeout(timeout), "max_retries": 0}
normalized = _normalize_base_url_text(base_url)
if normalized:
normalized = re.sub(r"/v1/?$", "", normalized.rstrip("/"))
kwargs["base_url"] = normalized
if _is_azure_anthropic_endpoint(normalized) and "api-version" not in normalized:
kwargs["default_query"] = {"api-version": "2025-04-15"}
return normalized, kwargs
def _build_anthropic_client_with_bearer_hook(
token_provider,
base_url: str = None,
timeout: float = None,
*,
drop_context_1m_beta: bool = False,
):
"""Anthropic-on-Foundry Entra ID variant of :func:`build_anthropic_client`.
The SDK stores ``api_key``/``auth_token`` as static strings, so per-request
bearer refresh (Microsoft's documented Foundry pattern) is done with a custom
``httpx.Client`` whose request hook mints a fresh JWT and rewrites
``Authorization`` on every request; the SDK skips its own auth when
``http_client`` is given. A placeholder ``auth_token`` is still required at
construction — the hook overrides it, and the sentinel makes any accidental
leak diagnosable in logs.
"""
sdk = _require_sdk("Azure Foundry Anthropic-style endpoints with Entra ID auth", verb="Install with")
normalize_proxy_env_vars()
from agent.azure_identity_adapter import build_bearer_http_client
normalized_base_url, kwargs = _base_client_kwargs(base_url, timeout)
kwargs["http_client"] = build_bearer_http_client(token_provider, timeout=kwargs["timeout"])
kwargs["auth_token"] = "entra-id-bearer-via-http-hook"
headers = _beta_header(_common_betas_for_base_url(normalized_base_url, drop_context_1m_beta=drop_context_1m_beta))
if headers:
kwargs["default_headers"] = headers
client = sdk.Anthropic(**kwargs)
# Same env-inference trap as build_anthropic_client: auth_token-only
# construction would otherwise also send ANTHROPIC_API_KEY as X-Api-Key.
client.api_key = None
return client
def build_anthropic_client(
api_key,
base_url: str = None,
timeout: float = None,
*,
drop_context_1m_beta: bool = False,
):
"""Create an Anthropic client, auto-detecting setup-tokens vs API keys.
``api_key`` is a static ``str`` (all key-based and OAuth flows) or a
``Callable[[], str]`` Entra ID bearer provider, which is routed through
:func:`_build_anthropic_client_with_bearer_hook`. ``timeout`` overrides the
900s read timeout (connect stays 10s) from the per-provider/per-model
``request_timeout_seconds`` config. ``drop_context_1m_beta`` strips
``context-1m-2025-08-07`` from the client-level beta header — used by the
reactive OAuth retry in run_agent when a subscription rejects it; fresh
clients keep the default so 1M-capable subscriptions keep the capability.
"""
sdk = _require_sdk("the Anthropic provider")
if callable(api_key) and not isinstance(api_key, str):
return _build_anthropic_client_with_bearer_hook(
api_key, base_url, timeout,
drop_context_1m_beta=drop_context_1m_beta,
)
normalize_proxy_env_vars()
normalized_base_url, kwargs = _base_client_kwargs(base_url, timeout)
if "default_query" in kwargs: # historical: this path also strips a stray trailing slash on Azure
kwargs["base_url"] = normalized_base_url.rstrip("/")
common_betas = _common_betas_for_base_url(normalized_base_url, drop_context_1m_beta=drop_context_1m_beta)
if _is_kimi_coding_endpoint(base_url):
# Kimi's /coding endpoint 403s without a User-Agent; the Kimi team asked
# for proper attribution instead of the ``claude-code/0.1.0`` minimum.
kwargs["api_key"] = api_key
headers = {**_attribution_headers(), **_beta_header(common_betas)}
elif _requires_bearer_auth(normalized_base_url):
# MiniMax & co. want the key in Authorization: Bearer. Checked before the
# OAuth shape test: their secrets lack the sk-ant-api prefix and would
# otherwise be misread as Anthropic OAuth/setup tokens.
kwargs["auth_token"] = api_key
headers = _beta_header(common_betas)
elif _is_third_party_anthropic_endpoint(base_url):
# Third-party proxies use their own x-api-key keys; skip OAuth detection
# (their keys don't follow the sk-ant-* convention).
kwargs["api_key"] = api_key
headers = _beta_header(common_betas)
elif _is_oauth_token(api_key):
# OAuth/setup-token -> Bearer auth + Claude Code identity. Anthropic
# routes OAuth by user-agent/headers; without the fingerprint, 500s.
kwargs["auth_token"] = api_key
headers = {
**_beta_header(common_betas + _OAUTH_ONLY_BETAS),
"user-agent": f"claude-code/{_get_claude_code_version()} (external, cli)",
"x-app": "cli",
}
else:
kwargs["api_key"] = api_key
headers = _beta_header(common_betas)
if _is_opencode_endpoint(base_url):
# OpenCode identifies clients by request headers (like OpenRouter). The
# OpenAI-wire paths get these from profile.default_headers, but this
# route builds its client here and never sees the profile.
for k, v in _attribution_headers().items():
headers.setdefault(k, v)
if headers:
kwargs["default_headers"] = headers
client = sdk.Anthropic(**kwargs)
# Bearer-only construction leaves ``api_key`` unset, so the SDK fills it from
# ANTHROPIC_API_KEY (loaded from ~/.hermes/.env) and sends dual auth —
# X-Api-Key *and* Authorization: Bearer — on every Portal/MiniMax/OAuth
# request. Clear it whenever we intentionally authenticated via auth_token.
if "auth_token" in kwargs and "api_key" not in kwargs:
client.api_key = None
return client
def build_anthropic_bedrock_client(region: str):
"""AnthropicBedrock client for Bedrock Claude models (boto3 default credential chain).
The SDK's native Bedrock adapter gives full Claude feature parity (prompt
caching, thinking budgets, adaptive thinking, fast mode) that Converse
lacks. The common betas plus ``context-1m-2025-08-07`` are attached: without
the latter Bedrock caps Opus 4.6/4.7 at 200K instead of 1M.
"""
sdk = _require_sdk("the Bedrock provider")
if not hasattr(sdk, "AnthropicBedrock"):
raise ImportError(
"anthropic.AnthropicBedrock not available. "
"Upgrade with: pip install 'anthropic>=0.39.0'"
)
return sdk.AnthropicBedrock(
aws_region=region,
timeout=_client_timeout(None),
max_retries=0, # retry belongs to hermes's outer loop (honors Retry-After)
default_headers=_beta_header([*_COMMON_BETAS, _CONTEXT_1M_BETA]),
)
def _normalize_to_mcp_wire(name: str) -> str:
"""OAuth wire form of a tool name (no aliasing): ``mcp__<...>``.
Anthropic's OAuth billing classifier treats a single-underscore ``mcp_``
tool name as a third-party-app fingerprint (HTTP 400 "Third-party apps now
draw from extra usage"); ``mcp__foo`` is accepted. Both bare Hermes tools
(``read_file``) and native MCP tools registered as ``mcp_<server>_<tool>``
must land on the double-underscore form — the latter was the gap a bare
prefix swap left open. normalize_response reverses both via registry lookup.
"""
if name.startswith("mcp__"):
return name # already correct, don't double-prefix
if name.startswith("mcp_"):
return "mcp__" + name[len("mcp_"):]
return _MCP_TOOL_PREFIX + name
def _oauth_wire_namer(anthropic_tools: List[Dict[str, Any]]):
"""Return ``name -> OAuth wire name`` for this request's tool set.
An alias must never collide with a wire name owned by a non-alias tool: two
identical tool names in one request is a hard 400, strictly worse than the
bug being fixed. Mirrors normalize_response's "registered tool wins" so
outbound and inbound agree on who owns a contested name.
"""
claimed = {
_normalize_to_mcp_wire(tool["name"])
for tool in (anthropic_tools or [])
if isinstance(tool.get("name"), str) and tool["name"] not in _OAUTH_TOOL_NAME_ALIASES
}
def to_wire(name: str) -> str:
if name in _OAUTH_TOOL_NAME_ALIASES:
aliased = _OAUTH_TOOL_NAME_ALIASES[name]
if _MCP_TOOL_PREFIX + aliased not in claimed:
name = aliased
return _normalize_to_mcp_wire(name)
return to_wire
_OAUTH_SYSTEM_REPLACEMENTS = (
("Hermes Agent", "Claude Code"),
("Hermes agent", "Claude Code"),
("hermes-agent", "claude-code"),
("Nous Research", "Anthropic"),
)
def _apply_claude_code_identity(system, anthropic_tools, anthropic_messages, to_wire):
"""OAuth transforms: Claude Code system prefix, product-name sanitizing (avoids
server-side content filters), tool/description aliasing, and the same tool
renames on replayed tool_use blocks so history matches ``tools[]``. Returns
the new ``system``; tools and messages are mutated in place.
"""
cc_block = {"type": "text", "text": _CLAUDE_CODE_SYSTEM_PREFIX}
if isinstance(system, list):
system = [cc_block] + system
elif isinstance(system, str) and system:
system = [cc_block, {"type": "text", "text": system}]
else:
system = [cc_block]
for block in system:
if isinstance(block, dict) and block.get("type") == "text":
text = block.get("text", "")
for old, new in _OAUTH_SYSTEM_REPLACEMENTS:
text = text.replace(old, new)
block["text"] = _apply_oauth_prose_aliases(text)
for tool in anthropic_tools or []:
if "name" in tool:
tool["name"] = to_wire(tool["name"])
description = tool.get("description")
if isinstance(description, str):
tool["description"] = _apply_oauth_prose_aliases(description) # prose-safe aliases only
for msg in anthropic_messages:
content = msg.get("content")
if isinstance(content, list):
for block in content:
if isinstance(block, dict) and block.get("type") == "tool_use" and "name" in block:
block["name"] = to_wire(block["name"]) # tool_result pairs by id, not name
return system
def _thinking_kwargs(reasoning_config: Dict[str, Any], model: str, effective_max_tokens: int) -> Dict[str, Any]:
"""Map ``reasoning_config`` to Anthropic thinking kwargs.
Adaptive models (Claude 4.6+, Kimi/Moonshot — the replay-validation 400s
that once motivated dropping the param for Kimi no longer occur) get
``thinking.type=adaptive`` + ``output_config.effort``; older models and
manual-only compat endpoints (MiniMax) get budget_tokens. Haiku has no
extended thinking. On 4.7+ ``thinking.display`` defaults to "omitted",
hiding the reasoning Hermes shows in its CLI, so "summarized" is requested
to keep the activity feed populated.
"""
if reasoning_config.get("enabled") is False:
# Adaptive models think by DEFAULT, so omitting the parameter is not a
# disable — the user silently keeps paying. Mandatory-thinking models
# 400 on the disable, so they keep the omission: a silently-ignored
# disable beats a dead turn.
return {"thinking": {"type": "disabled"}} if _accepts_thinking_disable(model) else {}
if "haiku" in model.lower():
return {}
effort = str(reasoning_config.get("effort", "medium")).lower()
budget = THINKING_BUDGET.get(effort, 8000)
if _supports_adaptive_thinking(model):
adaptive_effort = ADAPTIVE_EFFORT_MAP.get(effort, "medium")
if adaptive_effort == "xhigh" and not _supports_xhigh_effort(model):
adaptive_effort = "max"
return {
"thinking": {"type": "adaptive", "display": "summarized"},
"output_config": {"effort": adaptive_effort},
}
return {
"thinking": {"type": "enabled", "budget_tokens": budget},
"temperature": 1, # required when thinking is enabled on older models
"max_tokens": max(effective_max_tokens, budget + 4096),
}
def build_anthropic_kwargs(
model: str,
messages: List[Dict],
tools: Optional[List[Dict]],
max_tokens: Optional[int],
reasoning_config: Optional[Dict[str, Any]],
tool_choice: Optional[str] = None,
is_oauth: bool = False,
preserve_dots: bool = False,
context_length: Optional[int] = None,
base_url: str | None = None,
fast_mode: bool = False,
drop_context_1m_beta: bool = False,
) -> Dict[str, Any]:
"""Build kwargs for anthropic.messages.create().
Two easily confused concepts: ``max_tokens`` is the OUTPUT cap for one
response (Anthropic's name for it; their native SDK says max_output_tokens);
``context_length`` is the TOTAL window (input + output), enforced as
``input_tokens + max_tokens <= context_length``. ``max_tokens=None`` uses the
model's native output ceiling; if that exceeds ``context_length`` (small
local endpoints) it is clamped to ``context_length - 1``. The clamp ignores
prompt size — callers must catch "max_tokens too large given prompt" and
retry smaller (parse_available_output_tokens_from_error).
``is_oauth`` applies Claude Code compatibility transforms; ``preserve_dots``
keeps model-name dots (DashScope: qwen3.5-plus); a third-party ``base_url``
strips thinking signatures; ``fast_mode`` adds ``extra_body.speed="fast"``
plus the fast-mode beta on native Anthropic only.
"""
system, anthropic_messages = convert_messages_to_anthropic(
messages, base_url=base_url, model=model
)
anthropic_tools = convert_tools_to_anthropic(tools) if tools else []
# Nous Portal routes on its own catalog ids (``anthropic/claude-opus-4.8``);
# normalizing would make the model unresolvable there (prefix AND dots kept).
if not _is_nous_portal_endpoint(base_url):
model = normalize_model_name(model, preserve_dots=preserve_dots)
# Non-positive/non-finite values fail locally instead of 400-ing upstream.
effective_max_tokens = _resolve_anthropic_messages_max_tokens(
max_tokens, model, context_length=context_length
)
if context_length and effective_max_tokens > context_length:
effective_max_tokens = max(context_length - 1, 1)
to_wire = _oauth_wire_namer(anthropic_tools) if is_oauth else None
if to_wire:
system = _apply_claude_code_identity(system, anthropic_tools, anthropic_messages, to_wire)
kwargs: Dict[str, Any] = {
"model": model,
"messages": anthropic_messages,
"max_tokens": effective_max_tokens,
}
if system:
kwargs["system"] = system
if anthropic_tools:
kwargs["tools"] = anthropic_tools
if tool_choice == "auto" or tool_choice is None:
kwargs["tool_choice"] = {"type": "auto"}
elif tool_choice == "required":
kwargs["tool_choice"] = {"type": "any"}
elif tool_choice == "none":
kwargs.pop("tools", None) # no Anthropic "none" — omit tools to prevent use
elif isinstance(tool_choice, str):
# Under OAuth every tools[] entry is mcp__-prefixed/aliased; the forced
# name must go through the same normalizer or it (a) leaks the literal
# trigger string and (b) names a tool that no longer exists -> 400.
kwargs["tool_choice"] = {"type": "tool", "name": to_wire(tool_choice) if to_wire else tool_choice}
if reasoning_config and isinstance(reasoning_config, dict):
kwargs.update(_thinking_kwargs(reasoning_config, model, effective_max_tokens))
# Safety net so upstream 4.6 -> 4.7 migrations don't need coordinated edits
# everywhere callers (auxiliary_client, ...) set sampling params.
if _forbids_sampling_params(model):
for _sampling_key in ("temperature", "top_p", "top_k"):
kwargs.pop(_sampling_key, None)
# Fast mode: native Anthropic only — third-party providers reject the
# unknown beta/param and Anthropic scopes it to the Claude API (not
# Bedrock/Vertex/Foundry). Per-request extra_headers OVERRIDE the
# client-level anthropic-beta header, so rebuild the full beta list.
if fast_mode and not _is_third_party_anthropic_endpoint(base_url) and _supports_fast_mode(model):
kwargs.setdefault("extra_body", {})["speed"] = "fast"
betas = list(_common_betas_for_base_url(base_url, drop_context_1m_beta=drop_context_1m_beta))
if is_oauth:
betas.extend(_OAUTH_ONLY_BETAS)
betas.append(_FAST_MODE_BETA)
kwargs["extra_headers"] = _beta_header(betas)
return kwargs
# Keys exclusive to the OpenAI Responses / Codex shape; the Messages SDK raises
# ``TypeError: ... unexpected keyword argument`` on any of them.
_RESPONSES_ONLY_KWARGS = frozenset(
{"instructions", "input", "store", "parallel_tool_calls"}
)
def sanitize_anthropic_kwargs(api_kwargs: Any, *, log_prefix: str = "") -> Any:
"""Drop Responses-API-only keys before an Anthropic Messages SDK call.
Boundary guard for api_mode-flip races (a concurrent auxiliary call mutating
a shared agent between kwargs build and dispatch): a Responses-shaped payload
reaching ``messages.stream()`` dies with a non-retryable TypeError that takes
the whole turn and fallback chain with it. Mutates and returns
``api_kwargs``; logs a WARNING so the race stays visible.
"""
if not isinstance(api_kwargs, dict):
return api_kwargs
leaked = _RESPONSES_ONLY_KWARGS.intersection(api_kwargs)
if leaked:
for _key in leaked:
del api_kwargs[_key]
logger.warning(
"%sStripped Responses-only kwarg(s) %s from an Anthropic Messages "
"call (api_mode flip race — see #31673). The call will proceed; "
"this breadcrumb means a kwargs build ran under a Responses "
"api_mode while dispatch ran under anthropic_messages.",
log_prefix,
sorted(leaked),
)
return api_kwargs
def _is_stream_unavailable_error(exc: Exception) -> bool:
"""True when an Anthropic stream call should fall back to create()."""
err_lower = str(exc).lower()
if "stream" in err_lower and "not supported" in err_lower:
return True
if "invokemodelwithresponsestream" in err_lower:
from agent.bedrock_adapter import is_streaming_access_denied_error
return is_streaming_access_denied_error(exc)
return False
def create_anthropic_message(
client: Any,
api_kwargs: dict,
*,
log_prefix: str = "",
prefer_stream: bool = True,
on_stream_event=None,
on_response=None,
) -> Any:
"""Create an Anthropic message, aggregating via stream when available.
Some Anthropic-compatible gateways are SSE-only and answer ``create()`` with
``text/event-stream``, which the SDK surfaces as raw text (callers then
crash on ``.content``). So prefer ``messages.stream().get_final_message()``
like the main turn path, falling back to ``create()`` only for providers
that explicitly don't support streaming (restricted Bedrock roles).
Both callbacks are best-effort (exceptions swallowed) and fire only on the
streaming path. ``on_stream_event(event)`` lets liveness watchdogs see
forward progress (e.g. a slow compression summary is not "hung");
``on_response(httpx_response)`` exposes headers the parsed Message drops
(Nous Portal's ``x-nous-credits-*`` balance family).
"""
sanitize_anthropic_kwargs(api_kwargs, log_prefix=log_prefix)
messages_api = getattr(client, "messages", None)
stream_fn = getattr(messages_api, "stream", None)
if prefer_stream and callable(stream_fn):
stream_kwargs = dict(api_kwargs)
stream_kwargs.pop("stream", None)
try:
with stream_fn(**stream_kwargs) as stream:
if callable(on_response):
try:
on_response(getattr(stream, "response", None))
except Exception:
logger.debug("%son_response callback failed", log_prefix, exc_info=True)
if callable(on_stream_event):
# Consume manually so each event ticks the progress callback;
# get_final_message then returns the accumulated snapshot.
for _event in stream:
try:
on_stream_event(_event)
except TimeoutError:
# The callback is the caller's deadline seam: the host
# has given up, so abandon the stream (``with`` closes
# it) instead of streaming an answer nobody will read.
raise
except Exception:
logger.debug("%son_stream_event callback failed", log_prefix, exc_info=True)
return stream.get_final_message()
except TimeoutError:
raise
except Exception as exc:
if not _is_stream_unavailable_error(exc):
raise
logger.debug(
"%sAnthropic Messages stream unavailable; falling back to "
"messages.create(): %s",
log_prefix,
exc,
)
create_kwargs = dict(api_kwargs)
create_kwargs.pop("stream", None)
return messages_api.create(**create_kwargs)