diff --git a/agent/codex_responses_adapter.py b/agent/codex_responses_adapter.py index 0dadad59b9..1d114af397 100644 --- a/agent/codex_responses_adapter.py +++ b/agent/codex_responses_adapter.py @@ -770,7 +770,7 @@ _PREFLIGHT_OPTIONAL_FIELDS: tuple[tuple[str, Callable[[Any], bool], Optional[Cal # Cache routing/retention and tool-dispatch hints pass through as-is. *( (key, lambda v: v is not None, None) - for key in ("tool_choice", "parallel_tool_calls", "prompt_cache_key", "prompt_cache_retention") + for key in ("tool_choice", "parallel_tool_calls", "prompt_cache_key", "prompt_cache_retention", "prompt_cache_options") ), # Native compaction directive; eligibility is resolved in agent/native_compaction.py. ("context_management", lambda v: isinstance(v, list) and bool(v), None), diff --git a/agent/models_dev.py b/agent/models_dev.py index 5dd3e9707c..fddd02f2ca 100644 --- a/agent/models_dev.py +++ b/agent/models_dev.py @@ -522,6 +522,19 @@ _OVERRIDE_WARNED_KEYS: set = set() # shared by get_model_capabilities and get_model_info so the two unknown-model paths agree. _UNKNOWN_MODEL_BASE: Dict[str, Any] = {"limit": {"context": 200000, "output": 8192}, "tool_call": True} +# Account-gated models may be usable before models.dev has indexed them. Keep +# their capabilities available for an explicitly selected/discovered model +# without adding them to any picker catalog. +_BUILTIN_MODEL_METADATA: Dict[Tuple[str, str], Dict[str, Any]] = { + ("openai", "gpt-6-astra"): { + "limit": {"context": 1_050_000, "output": 128_000}, + "modalities": {"input": ["text", "image"], "output": ["text"]}, + "tool_call": True, + "reasoning": True, + "family": "gpt-6", + }, +} + def _load_model_overrides() -> Dict[str, Any]: """The ``model_overrides`` config section ({} on any failure). Deliberately not memoized: @@ -646,8 +659,11 @@ def _merge_catalog_entry_with_override(raw: Dict[str, Any], override: Dict[str, def _apply_overrides(provider: str, model: str, entry: Optional[Dict[str, Any]]) -> Optional[Dict[str, Any]]: """*entry* patched by its override; ``_UNKNOWN_MODEL_BASE`` patched by a fill-gap override on a catalog miss (selected AFTER lookup: _default only fills misses); None when neither exists.""" - override = _override_for(provider, model, catalog_hit=entry is not None) - return entry if override is None else _merge_catalog_entry_with_override(entry if entry is not None else _UNKNOWN_MODEL_BASE, override) + provider_key = PROVIDER_TO_MODELS_DEV.get((provider or "").strip(), (provider or "").strip()) + builtin = _BUILTIN_MODEL_METADATA.get((provider_key, (model or "").strip().lower())) + base = entry if entry is not None else builtin + override = _override_for(provider, model, catalog_hit=base is not None) + return base if override is None else _merge_catalog_entry_with_override(base if base is not None else _UNKNOWN_MODEL_BASE, override) def _entry_supports_vision(entry: Dict[str, Any]) -> bool: diff --git a/agent/reasoning_effort.py b/agent/reasoning_effort.py index 711517f8fe..e84c59b1ba 100644 --- a/agent/reasoning_effort.py +++ b/agent/reasoning_effort.py @@ -31,6 +31,9 @@ OPENAI_COMPAT_WIRE_EFFORTS: tuple[str, ...] = ("none", "minimal", "low", "medium #: both (clamps to low); ``max`` is gpt-5.6-only. CODEX_GPT56_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh", "max") CODEX_LEGACY_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh") +# GPT-6 Astra is account-gated and its Responses API accepts no disable/minimal +# wire level; callers normalize those requests to ``low`` at the transport boundary. +CODEX_ASTRA_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "xhigh", "max") #: xAI Responses — Grok 4.6+ accepts xhigh; older Grok tops out at high. XAI_GROK46_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "xhigh") @@ -80,6 +83,8 @@ META_AI_EFFORTS: tuple[str, ...] = ("minimal", "low", "medium", "high", "xhigh") def codex_supported_efforts(model: Optional[str]) -> tuple[str, ...]: """Supported effort set for an OpenAI/Codex Responses model.""" + if (model or "").strip().lower().rsplit("/", 1)[-1] == "gpt-6-astra": + return CODEX_ASTRA_EFFORTS return CODEX_GPT56_EFFORTS if "gpt-5.6" in (model or "").lower() else CODEX_LEGACY_EFFORTS diff --git a/agent/transports/codex.py b/agent/transports/codex.py index 183c7f3a30..a83b99f73c 100644 --- a/agent/transports/codex.py +++ b/agent/transports/codex.py @@ -11,7 +11,7 @@ import re from typing import Any, Callable, Optional from agent.reasoning_effort import ( - ACTUAL_RELAY_EFFORTS, XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort, + ACTUAL_RELAY_EFFORTS, CODEX_ASTRA_EFFORTS, XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort, # Same declared vocabulary + shared clamp as the main Codex transport (agent.reasoning_effort): # per-model — "max" is gpt-5.6-only, "minimal"/"ultra" always rejected (live-verified, #68365). codex_supported_efforts, @@ -211,6 +211,15 @@ def _resolve_reasoning(model: str, params: dict[str, Any]) -> tuple[Any, bool]: elif reasoning_config.get("effort"): reasoning_effort = reasoning_config["effort"] + # Astra has no wire-level disable/minimal setting. Preserve Hermes' user-facing + # controls by sending the documented lowest enabled level instead; this also keeps + # an explicit ``enabled: false`` request valid on the Responses API. + if model.strip().lower().rsplit("/", 1)[-1] == "gpt-6-astra": + requested = str(reasoning_effort or "").strip().lower() + if not reasoning_enabled or requested in {"", "none", "minimal", "disabled", "off"}: + return "low", True + return clamp_effort(reasoning_effort, CODEX_ASTRA_EFFORTS), True + # Wire vocabularies are declared in agent.reasoning_effort; the shared clamp policy (nearest weaker # supported level, never escalate, never invert the ladder) replaces the per-backend hand maps that # repeatedly leaked internal levels like "ultra" to the wire (#89503 class) or clamped one rung below a @@ -262,6 +271,37 @@ def _default_prompt_cache_retention_for_request(model: str, base_url: Any) -> Op return "24h" if _EXTENDED_PROMPT_CACHE_MODEL_RE.search(normalized) else None +def _is_astra_model(model: Any) -> bool: + return str(model or "").strip().lower().rsplit("/", 1)[-1] == "gpt-6-astra" + + +def _is_official_openai_responses_route(model: Any, base_url: Any) -> bool: + """Astra cache semantics apply only to the official OpenAI API host.""" + if not _is_astra_model(model): + return False + from utils import base_url_host_matches + + return base_url_host_matches(str(base_url or ""), "api.openai.com") + + +def _sanitize_astra_request_kwargs(kwargs: dict[str, Any], model: Any, base_url: Any) -> None: + """Apply Astra's model-specific restrictions after all request overrides are merged.""" + if not _is_astra_model(model): + return + for key in ("temperature", "top_p", "top_logprobs", "logprobs"): + kwargs.pop(key, None) + include = kwargs.get("include") + if isinstance(include, list): + kwargs["include"] = [item for item in include if "logprob" not in str(item).lower()] + if _is_official_openai_responses_route(model, base_url): + kwargs["prompt_cache_options"] = {"ttl": "30m"} + kwargs.pop("prompt_cache_retention", None) + else: + # A caller override must not accidentally send official-only cache syntax to a + # proxy or another provider that happens to accept the Astra model name. + kwargs.pop("prompt_cache_options", None) + + def _content_cache_key(instructions: str, tools: Optional[list[dict[str, Any]]], scope_id: str = "") -> Optional[str]: """``pck_`` of (scope_id, instructions, name-sorted tools), or None if nothing static. @@ -544,6 +584,8 @@ class ResponsesApiTransport(ProviderTransport): if params.get("request_overrides"): kwargs.update(params["request_overrides"]) + _sanitize_astra_request_kwargs(kwargs, model, params.get("base_url")) + _bound_prompt_cache_key_field(kwargs) # Older xAI models reject ``service_tier`` (HTTP 400); only Grok 4.6 accepts Priority Processing. diff --git a/agent/usage_pricing.py b/agent/usage_pricing.py index 9830e2e28b..214c63223a 100644 --- a/agent/usage_pricing.py +++ b/agent/usage_pricing.py @@ -109,6 +109,7 @@ class PricingEntry: input_cost_per_million_above: Optional[Decimal] = None output_cost_per_million_above: Optional[Decimal] = None cache_read_cost_per_million_above: Optional[Decimal] = None + cache_write_cost_per_million_above: Optional[Decimal] = None @dataclass(frozen=True) @@ -241,6 +242,20 @@ for _provider, _url, _version, _rows in _SNAPSHOTS: _OFFICIAL_DOCS_PRICING[(_provider, _model)] = _entry del _SNAPSHOTS, _provider, _url, _version, _rows, _models, _rates, _entry, _model +# GPT-6 Astra uses whole-request pricing above the 272K prompt tier. Keep this +# account-gated model out of generic static catalogs, but retain published billing +# metadata for an explicitly selected route. +_OFFICIAL_DOCS_PRICING[("openai", "gpt-6-astra")] = _snap( + "10.00", "50.00", "1.00", "12.50", + url="https://developers.openai.com/api/docs/models/gpt-6-astra", + version="openai-gpt-6-astra-2026-09", + tier_threshold_tokens=272_000, + input_cost_per_million_above=Decimal("20.00"), + output_cost_per_million_above=Decimal("75.00"), + cache_read_cost_per_million_above=Decimal("2.00"), + cache_write_cost_per_million_above=Decimal("25.00"), +) + # Context-tiered Gemini Pro: above 200k prompt tokens the *_above rates apply to # the whole request (see PricingEntry). _OFFICIAL_DOCS_PRICING[("google", "gemini-3.1-pro")] = _snap( @@ -558,7 +573,7 @@ def estimate_usage_cost( (usage.output_tokens, entry.output_cost_per_million, entry.output_cost_per_million_above, ()), (usage.cache_read_tokens, entry.cache_read_cost_per_million, entry.cache_read_cost_per_million_above, ("cache-read pricing unavailable for route",)), - (usage.cache_write_tokens, entry.cache_write_cost_per_million, None, + (usage.cache_write_tokens, entry.cache_write_cost_per_million, entry.cache_write_cost_per_million_above, ("cache-write pricing unavailable for route",)), ): if above and rate_above is not None: diff --git a/hermes_cli/codex_models.py b/hermes_cli/codex_models.py index 178800b2e7..ed570f1416 100644 --- a/hermes_cli/codex_models.py +++ b/hermes_cli/codex_models.py @@ -88,9 +88,16 @@ def _add_context_variants(model_ids: List[str]) -> List[str]: return out -def _finalize_codex_models(model_ids: List[str]) -> List[str]: - """Forward-compat synthesis + large-context variant synthesis.""" - return _add_context_variants(_add_forward_compat_models(model_ids)) +def _finalize_codex_models(model_ids: List[str], *, allow_astra: bool = False) -> List[str]: + """Forward-compat/context synthesis with an entitlement gate for Astra. + + Cached/configured model names are useful compatibility hints, but only the + account-scoped Codex endpoint is authoritative for current Astra access. + """ + finalized = _add_forward_compat_models(model_ids) + if not allow_astra: + finalized = [model for model in finalized if model.lower() != "gpt-6-astra"] + return _add_context_variants(finalized) def _extract_chatgpt_account_id(access_token: str) -> Optional[str]: @@ -159,7 +166,7 @@ def _fetch_models_from_api(access_token: str) -> List[str]: logger.debug("Failed to fetch Codex models from API: %s", exc) return [] - return _finalize_codex_models(_ranked_slugs(entries)) + return _finalize_codex_models(_ranked_slugs(entries), allow_astra=True) def _read_default_model(codex_home: Path) -> Optional[str]: @@ -194,7 +201,7 @@ def get_codex_model_ids(access_token: Optional[str] = None) -> List[str]: if access_token: api_models = _fetch_models_from_api(access_token) if api_models: - return _finalize_codex_models(api_models) + return _finalize_codex_models(api_models, allow_astra=True) default_model = _read_default_model(codex_home) return _finalize_codex_models(_dedupe([ *([default_model] if default_model else []), *_read_cache_models(codex_home), diff --git a/hermes_cli/models.py b/hermes_cli/models.py index 50e513b65a..4a4ff86d87 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -1203,7 +1203,12 @@ def _openai_catalog(normalized: str, force_refresh: bool) -> Optional[list[str]] curated = list(_PROVIDER_MODELS.get(normalized, [])) # Curated order, only models the account has access to; an account serving none of them (rare) # falls back to curated so the picker still offers sane defaults. - return [m for m in curated if m.lower() in live_lower] or curated or live + discovered = [m for m in curated if m.lower() in live_lower] + # Astra is intentionally absent from offline/static catalogs: the official API's + # account-scoped /models response is the only source that may advertise it. + if "gpt-6-astra" in live_lower: + discovered.append(next((m for m in live if m.lower() == "gpt-6-astra"), "gpt-6-astra")) + return discovered or curated or live def _custom_catalog(normalized: str, force_refresh: bool) -> Optional[list[str]]: diff --git a/tests/agent/test_models_dev.py b/tests/agent/test_models_dev.py index 94c3c62818..410452dbd6 100644 --- a/tests/agent/test_models_dev.py +++ b/tests/agent/test_models_dev.py @@ -825,6 +825,18 @@ class TestGetModelCapabilities: assert caps is not None assert caps.supports_vision is False + def test_astra_builtin_metadata_fills_catalog_lag_without_listing_it(self): + """An explicitly discovered Astra remains fully described before models.dev catches up.""" + with patch("agent.models_dev.fetch_models_dev", return_value={}): + caps = get_model_capabilities("openai", "gpt-6-astra") + + assert caps is not None + assert caps.context_window == 1_050_000 + assert caps.max_output_tokens == 128_000 + assert caps.supports_tools is True + assert caps.supports_vision is True + assert caps.supports_reasoning is True + # --------------------------------------------------------------------------- # Per-model metadata overrides (model_overrides config) diff --git a/tests/agent/test_usage_pricing.py b/tests/agent/test_usage_pricing.py index 5895985e83..346731df1e 100644 --- a/tests/agent/test_usage_pricing.py +++ b/tests/agent/test_usage_pricing.py @@ -11,6 +11,23 @@ from agent.usage_pricing import ( from decimal import Decimal +def test_astra_whole_request_price_tier_includes_cache_writes(): + below = estimate_usage_cost( + "gpt-6-astra", + CanonicalUsage(input_tokens=100_000, output_tokens=10_000, cache_read_tokens=10_000, cache_write_tokens=10_000), + provider="openai", + ) + above = estimate_usage_cost( + "gpt-6-astra", + CanonicalUsage(input_tokens=100_000, output_tokens=10_000, cache_read_tokens=100_000, cache_write_tokens=100_001), + provider="openai", + ) + + assert below.amount_usd == Decimal("1.635") + assert above.amount_usd == Decimal("5.450025") + assert above.pricing_version == "openai-gpt-6-astra-2026-09" + + diff --git a/tests/agent/transports/test_codex_transport.py b/tests/agent/transports/test_codex_transport.py index 463fea8bb4..d4946158bd 100644 --- a/tests/agent/transports/test_codex_transport.py +++ b/tests/agent/transports/test_codex_transport.py @@ -39,6 +39,42 @@ class TestCodexTransportBasic: class TestCodexBuildKwargs: + def test_astra_direct_request_applies_model_contract_after_overrides(self, transport): + kw = transport.build_kwargs( + model="gpt-6-astra", + messages=[{"role": "user", "content": "Hi"}], + tools=[], + base_url="https://api.openai.com/v1", + reasoning_config={"enabled": False, "effort": "none"}, + request_overrides={ + "temperature": 0.4, + "top_p": 0.9, + "top_logprobs": 5, + "logprobs": True, + "include": ["reasoning.encrypted_content", "message.output_text.logprobs"], + "prompt_cache_retention": "24h", + "prompt_cache_options": {"ttl": "1h"}, + }, + ) + + assert kw["reasoning"]["effort"] == "low" + assert kw["prompt_cache_options"] == {"ttl": "30m"} + assert "prompt_cache_retention" not in kw + assert kw["include"] == ["reasoning.encrypted_content"] + for unsupported in ("temperature", "top_p", "top_logprobs", "logprobs"): + assert unsupported not in kw + + def test_astra_proxy_does_not_receive_official_cache_options(self, transport): + kw = transport.build_kwargs( + model="gpt-6-astra", + messages=[{"role": "user", "content": "Hi"}], + tools=[], + base_url="https://responses.example.com/v1", + request_overrides={"prompt_cache_options": {"ttl": "30m"}}, + ) + + assert "prompt_cache_options" not in kw + def test_900k_context_variant_suffix_stripped_on_wire(self, transport): """``-900k`` large-context picker variants are Hermes-side aliases — the Codex backend only knows the base slug, so build_kwargs must diff --git a/tests/hermes_cli/test_codex_models.py b/tests/hermes_cli/test_codex_models.py index 7885293357..6408e8f6d6 100644 --- a/tests/hermes_cli/test_codex_models.py +++ b/tests/hermes_cli/test_codex_models.py @@ -112,6 +112,28 @@ def test_fetch_from_api_keeps_supported_in_api_false_models(monkeypatch): assert "gpt-5-internal" not in models +def test_astra_requires_live_codex_account_discovery(monkeypatch, tmp_path): + """Cached/configured Astra names must not manufacture current OAuth entitlement.""" + from hermes_cli import codex_models + + (tmp_path / "config.toml").write_text('model = "gpt-6-astra"\n', encoding="utf-8") + (tmp_path / "models_cache.json").write_text( + json.dumps({"models": [{"slug": "gpt-6-astra", "priority": 0}]}), + encoding="utf-8", + ) + monkeypatch.setenv("CODEX_HOME", str(tmp_path)) + monkeypatch.setattr(codex_models, "_fetch_models_from_api", lambda _token: []) + + assert "gpt-6-astra" not in get_codex_model_ids(access_token="stale-token") + + monkeypatch.setattr( + codex_models, + "_fetch_models_from_api", + lambda _token: codex_models._finalize_codex_models(["gpt-6-astra"], allow_astra=True), + ) + assert "gpt-6-astra" in get_codex_model_ids(access_token="entitled-token") + + @@ -238,4 +260,3 @@ class TestNormalizeModelForProvider: assert changed is True # Uses first from available list assert cli.model == "gpt-5.3-codex" - diff --git a/tests/hermes_cli/test_openai_picker_curated.py b/tests/hermes_cli/test_openai_picker_curated.py index 3273917d3e..9d95043fbf 100644 --- a/tests/hermes_cli/test_openai_picker_curated.py +++ b/tests/hermes_cli/test_openai_picker_curated.py @@ -70,3 +70,16 @@ def test_default_openai_endpoint_intersects_account_access(monkeypatch): assert result == list(curated[:2]) +def test_astra_is_offered_only_by_successful_account_discovery(monkeypatch): + """A gated preview may enrich the picker only when this API key lists it.""" + monkeypatch.setenv("OPENAI_API_KEY", "sk-fake") + monkeypatch.delenv("OPENAI_BASE_URL", raising=False) + + with patch.object(M, "fetch_api_models", return_value=["gpt-5.6-sol", "gpt-6-astra"]): + discovered = M.provider_model_ids("openai-api", force_refresh=True) + with patch.object(M, "fetch_api_models", return_value=["gpt-5.6-sol"]): + not_entitled = M.provider_model_ids("openai-api", force_refresh=True) + + assert "gpt-6-astra" in discovered + assert "gpt-6-astra" not in not_entitled +