feat(openai): add GPT-6 Astra baseline support
(cherry picked from commit a8c53d20c6b16cc35745e364e16bb7259166a1d3)
This commit is contained in:
@@ -770,7 +770,7 @@ _PREFLIGHT_OPTIONAL_FIELDS: tuple[tuple[str, Callable[[Any], bool], Optional[Cal
|
||||
# Cache routing/retention and tool-dispatch hints pass through as-is.
|
||||
*(
|
||||
(key, lambda v: v is not None, None)
|
||||
for key in ("tool_choice", "parallel_tool_calls", "prompt_cache_key", "prompt_cache_retention")
|
||||
for key in ("tool_choice", "parallel_tool_calls", "prompt_cache_key", "prompt_cache_retention", "prompt_cache_options")
|
||||
),
|
||||
# Native compaction directive; eligibility is resolved in agent/native_compaction.py.
|
||||
("context_management", lambda v: isinstance(v, list) and bool(v), None),
|
||||
|
||||
@@ -522,6 +522,19 @@ _OVERRIDE_WARNED_KEYS: set = set()
|
||||
# shared by get_model_capabilities and get_model_info so the two unknown-model paths agree.
|
||||
_UNKNOWN_MODEL_BASE: Dict[str, Any] = {"limit": {"context": 200000, "output": 8192}, "tool_call": True}
|
||||
|
||||
# Account-gated models may be usable before models.dev has indexed them. Keep
|
||||
# their capabilities available for an explicitly selected/discovered model
|
||||
# without adding them to any picker catalog.
|
||||
_BUILTIN_MODEL_METADATA: Dict[Tuple[str, str], Dict[str, Any]] = {
|
||||
("openai", "gpt-6-astra"): {
|
||||
"limit": {"context": 1_050_000, "output": 128_000},
|
||||
"modalities": {"input": ["text", "image"], "output": ["text"]},
|
||||
"tool_call": True,
|
||||
"reasoning": True,
|
||||
"family": "gpt-6",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _load_model_overrides() -> Dict[str, Any]:
|
||||
"""The ``model_overrides`` config section ({} on any failure). Deliberately not memoized:
|
||||
@@ -646,8 +659,11 @@ def _merge_catalog_entry_with_override(raw: Dict[str, Any], override: Dict[str,
|
||||
def _apply_overrides(provider: str, model: str, entry: Optional[Dict[str, Any]]) -> Optional[Dict[str, Any]]:
|
||||
"""*entry* patched by its override; ``_UNKNOWN_MODEL_BASE`` patched by a fill-gap override on a
|
||||
catalog miss (selected AFTER lookup: _default only fills misses); None when neither exists."""
|
||||
override = _override_for(provider, model, catalog_hit=entry is not None)
|
||||
return entry if override is None else _merge_catalog_entry_with_override(entry if entry is not None else _UNKNOWN_MODEL_BASE, override)
|
||||
provider_key = PROVIDER_TO_MODELS_DEV.get((provider or "").strip(), (provider or "").strip())
|
||||
builtin = _BUILTIN_MODEL_METADATA.get((provider_key, (model or "").strip().lower()))
|
||||
base = entry if entry is not None else builtin
|
||||
override = _override_for(provider, model, catalog_hit=base is not None)
|
||||
return base if override is None else _merge_catalog_entry_with_override(base if base is not None else _UNKNOWN_MODEL_BASE, override)
|
||||
|
||||
|
||||
def _entry_supports_vision(entry: Dict[str, Any]) -> bool:
|
||||
|
||||
@@ -31,6 +31,9 @@ OPENAI_COMPAT_WIRE_EFFORTS: tuple[str, ...] = ("none", "minimal", "low", "medium
|
||||
#: both (clamps to low); ``max`` is gpt-5.6-only.
|
||||
CODEX_GPT56_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh", "max")
|
||||
CODEX_LEGACY_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh")
|
||||
# GPT-6 Astra is account-gated and its Responses API accepts no disable/minimal
|
||||
# wire level; callers normalize those requests to ``low`` at the transport boundary.
|
||||
CODEX_ASTRA_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "xhigh", "max")
|
||||
|
||||
#: xAI Responses — Grok 4.6+ accepts xhigh; older Grok tops out at high.
|
||||
XAI_GROK46_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "xhigh")
|
||||
@@ -80,6 +83,8 @@ META_AI_EFFORTS: tuple[str, ...] = ("minimal", "low", "medium", "high", "xhigh")
|
||||
|
||||
def codex_supported_efforts(model: Optional[str]) -> tuple[str, ...]:
|
||||
"""Supported effort set for an OpenAI/Codex Responses model."""
|
||||
if (model or "").strip().lower().rsplit("/", 1)[-1] == "gpt-6-astra":
|
||||
return CODEX_ASTRA_EFFORTS
|
||||
return CODEX_GPT56_EFFORTS if "gpt-5.6" in (model or "").lower() else CODEX_LEGACY_EFFORTS
|
||||
|
||||
|
||||
|
||||
@@ -11,7 +11,7 @@ import re
|
||||
from typing import Any, Callable, Optional
|
||||
|
||||
from agent.reasoning_effort import (
|
||||
ACTUAL_RELAY_EFFORTS, XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort,
|
||||
ACTUAL_RELAY_EFFORTS, CODEX_ASTRA_EFFORTS, XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort,
|
||||
# Same declared vocabulary + shared clamp as the main Codex transport (agent.reasoning_effort):
|
||||
# per-model — "max" is gpt-5.6-only, "minimal"/"ultra" always rejected (live-verified, #68365).
|
||||
codex_supported_efforts,
|
||||
@@ -211,6 +211,15 @@ def _resolve_reasoning(model: str, params: dict[str, Any]) -> tuple[Any, bool]:
|
||||
elif reasoning_config.get("effort"):
|
||||
reasoning_effort = reasoning_config["effort"]
|
||||
|
||||
# Astra has no wire-level disable/minimal setting. Preserve Hermes' user-facing
|
||||
# controls by sending the documented lowest enabled level instead; this also keeps
|
||||
# an explicit ``enabled: false`` request valid on the Responses API.
|
||||
if model.strip().lower().rsplit("/", 1)[-1] == "gpt-6-astra":
|
||||
requested = str(reasoning_effort or "").strip().lower()
|
||||
if not reasoning_enabled or requested in {"", "none", "minimal", "disabled", "off"}:
|
||||
return "low", True
|
||||
return clamp_effort(reasoning_effort, CODEX_ASTRA_EFFORTS), True
|
||||
|
||||
# Wire vocabularies are declared in agent.reasoning_effort; the shared clamp policy (nearest weaker
|
||||
# supported level, never escalate, never invert the ladder) replaces the per-backend hand maps that
|
||||
# repeatedly leaked internal levels like "ultra" to the wire (#89503 class) or clamped one rung below a
|
||||
@@ -262,6 +271,37 @@ def _default_prompt_cache_retention_for_request(model: str, base_url: Any) -> Op
|
||||
return "24h" if _EXTENDED_PROMPT_CACHE_MODEL_RE.search(normalized) else None
|
||||
|
||||
|
||||
def _is_astra_model(model: Any) -> bool:
|
||||
return str(model or "").strip().lower().rsplit("/", 1)[-1] == "gpt-6-astra"
|
||||
|
||||
|
||||
def _is_official_openai_responses_route(model: Any, base_url: Any) -> bool:
|
||||
"""Astra cache semantics apply only to the official OpenAI API host."""
|
||||
if not _is_astra_model(model):
|
||||
return False
|
||||
from utils import base_url_host_matches
|
||||
|
||||
return base_url_host_matches(str(base_url or ""), "api.openai.com")
|
||||
|
||||
|
||||
def _sanitize_astra_request_kwargs(kwargs: dict[str, Any], model: Any, base_url: Any) -> None:
|
||||
"""Apply Astra's model-specific restrictions after all request overrides are merged."""
|
||||
if not _is_astra_model(model):
|
||||
return
|
||||
for key in ("temperature", "top_p", "top_logprobs", "logprobs"):
|
||||
kwargs.pop(key, None)
|
||||
include = kwargs.get("include")
|
||||
if isinstance(include, list):
|
||||
kwargs["include"] = [item for item in include if "logprob" not in str(item).lower()]
|
||||
if _is_official_openai_responses_route(model, base_url):
|
||||
kwargs["prompt_cache_options"] = {"ttl": "30m"}
|
||||
kwargs.pop("prompt_cache_retention", None)
|
||||
else:
|
||||
# A caller override must not accidentally send official-only cache syntax to a
|
||||
# proxy or another provider that happens to accept the Astra model name.
|
||||
kwargs.pop("prompt_cache_options", None)
|
||||
|
||||
|
||||
def _content_cache_key(instructions: str, tools: Optional[list[dict[str, Any]]], scope_id: str = "") -> Optional[str]:
|
||||
"""``pck_<sha256[:24]>`` of (scope_id, instructions, name-sorted tools), or None if nothing static.
|
||||
|
||||
@@ -544,6 +584,8 @@ class ResponsesApiTransport(ProviderTransport):
|
||||
if params.get("request_overrides"):
|
||||
kwargs.update(params["request_overrides"])
|
||||
|
||||
_sanitize_astra_request_kwargs(kwargs, model, params.get("base_url"))
|
||||
|
||||
_bound_prompt_cache_key_field(kwargs)
|
||||
|
||||
# Older xAI models reject ``service_tier`` (HTTP 400); only Grok 4.6 accepts Priority Processing.
|
||||
|
||||
@@ -109,6 +109,7 @@ class PricingEntry:
|
||||
input_cost_per_million_above: Optional[Decimal] = None
|
||||
output_cost_per_million_above: Optional[Decimal] = None
|
||||
cache_read_cost_per_million_above: Optional[Decimal] = None
|
||||
cache_write_cost_per_million_above: Optional[Decimal] = None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -241,6 +242,20 @@ for _provider, _url, _version, _rows in _SNAPSHOTS:
|
||||
_OFFICIAL_DOCS_PRICING[(_provider, _model)] = _entry
|
||||
del _SNAPSHOTS, _provider, _url, _version, _rows, _models, _rates, _entry, _model
|
||||
|
||||
# GPT-6 Astra uses whole-request pricing above the 272K prompt tier. Keep this
|
||||
# account-gated model out of generic static catalogs, but retain published billing
|
||||
# metadata for an explicitly selected route.
|
||||
_OFFICIAL_DOCS_PRICING[("openai", "gpt-6-astra")] = _snap(
|
||||
"10.00", "50.00", "1.00", "12.50",
|
||||
url="https://developers.openai.com/api/docs/models/gpt-6-astra",
|
||||
version="openai-gpt-6-astra-2026-09",
|
||||
tier_threshold_tokens=272_000,
|
||||
input_cost_per_million_above=Decimal("20.00"),
|
||||
output_cost_per_million_above=Decimal("75.00"),
|
||||
cache_read_cost_per_million_above=Decimal("2.00"),
|
||||
cache_write_cost_per_million_above=Decimal("25.00"),
|
||||
)
|
||||
|
||||
# Context-tiered Gemini Pro: above 200k prompt tokens the *_above rates apply to
|
||||
# the whole request (see PricingEntry).
|
||||
_OFFICIAL_DOCS_PRICING[("google", "gemini-3.1-pro")] = _snap(
|
||||
@@ -558,7 +573,7 @@ def estimate_usage_cost(
|
||||
(usage.output_tokens, entry.output_cost_per_million, entry.output_cost_per_million_above, ()),
|
||||
(usage.cache_read_tokens, entry.cache_read_cost_per_million, entry.cache_read_cost_per_million_above,
|
||||
("cache-read pricing unavailable for route",)),
|
||||
(usage.cache_write_tokens, entry.cache_write_cost_per_million, None,
|
||||
(usage.cache_write_tokens, entry.cache_write_cost_per_million, entry.cache_write_cost_per_million_above,
|
||||
("cache-write pricing unavailable for route",)),
|
||||
):
|
||||
if above and rate_above is not None:
|
||||
|
||||
@@ -88,9 +88,16 @@ def _add_context_variants(model_ids: List[str]) -> List[str]:
|
||||
return out
|
||||
|
||||
|
||||
def _finalize_codex_models(model_ids: List[str]) -> List[str]:
|
||||
"""Forward-compat synthesis + large-context variant synthesis."""
|
||||
return _add_context_variants(_add_forward_compat_models(model_ids))
|
||||
def _finalize_codex_models(model_ids: List[str], *, allow_astra: bool = False) -> List[str]:
|
||||
"""Forward-compat/context synthesis with an entitlement gate for Astra.
|
||||
|
||||
Cached/configured model names are useful compatibility hints, but only the
|
||||
account-scoped Codex endpoint is authoritative for current Astra access.
|
||||
"""
|
||||
finalized = _add_forward_compat_models(model_ids)
|
||||
if not allow_astra:
|
||||
finalized = [model for model in finalized if model.lower() != "gpt-6-astra"]
|
||||
return _add_context_variants(finalized)
|
||||
|
||||
|
||||
def _extract_chatgpt_account_id(access_token: str) -> Optional[str]:
|
||||
@@ -159,7 +166,7 @@ def _fetch_models_from_api(access_token: str) -> List[str]:
|
||||
logger.debug("Failed to fetch Codex models from API: %s", exc)
|
||||
return []
|
||||
|
||||
return _finalize_codex_models(_ranked_slugs(entries))
|
||||
return _finalize_codex_models(_ranked_slugs(entries), allow_astra=True)
|
||||
|
||||
|
||||
def _read_default_model(codex_home: Path) -> Optional[str]:
|
||||
@@ -194,7 +201,7 @@ def get_codex_model_ids(access_token: Optional[str] = None) -> List[str]:
|
||||
if access_token:
|
||||
api_models = _fetch_models_from_api(access_token)
|
||||
if api_models:
|
||||
return _finalize_codex_models(api_models)
|
||||
return _finalize_codex_models(api_models, allow_astra=True)
|
||||
default_model = _read_default_model(codex_home)
|
||||
return _finalize_codex_models(_dedupe([
|
||||
*([default_model] if default_model else []), *_read_cache_models(codex_home),
|
||||
|
||||
@@ -1203,7 +1203,12 @@ def _openai_catalog(normalized: str, force_refresh: bool) -> Optional[list[str]]
|
||||
curated = list(_PROVIDER_MODELS.get(normalized, []))
|
||||
# Curated order, only models the account has access to; an account serving none of them (rare)
|
||||
# falls back to curated so the picker still offers sane defaults.
|
||||
return [m for m in curated if m.lower() in live_lower] or curated or live
|
||||
discovered = [m for m in curated if m.lower() in live_lower]
|
||||
# Astra is intentionally absent from offline/static catalogs: the official API's
|
||||
# account-scoped /models response is the only source that may advertise it.
|
||||
if "gpt-6-astra" in live_lower:
|
||||
discovered.append(next((m for m in live if m.lower() == "gpt-6-astra"), "gpt-6-astra"))
|
||||
return discovered or curated or live
|
||||
|
||||
|
||||
def _custom_catalog(normalized: str, force_refresh: bool) -> Optional[list[str]]:
|
||||
|
||||
@@ -825,6 +825,18 @@ class TestGetModelCapabilities:
|
||||
assert caps is not None
|
||||
assert caps.supports_vision is False
|
||||
|
||||
def test_astra_builtin_metadata_fills_catalog_lag_without_listing_it(self):
|
||||
"""An explicitly discovered Astra remains fully described before models.dev catches up."""
|
||||
with patch("agent.models_dev.fetch_models_dev", return_value={}):
|
||||
caps = get_model_capabilities("openai", "gpt-6-astra")
|
||||
|
||||
assert caps is not None
|
||||
assert caps.context_window == 1_050_000
|
||||
assert caps.max_output_tokens == 128_000
|
||||
assert caps.supports_tools is True
|
||||
assert caps.supports_vision is True
|
||||
assert caps.supports_reasoning is True
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Per-model metadata overrides (model_overrides config)
|
||||
|
||||
@@ -11,6 +11,23 @@ from agent.usage_pricing import (
|
||||
from decimal import Decimal
|
||||
|
||||
|
||||
def test_astra_whole_request_price_tier_includes_cache_writes():
|
||||
below = estimate_usage_cost(
|
||||
"gpt-6-astra",
|
||||
CanonicalUsage(input_tokens=100_000, output_tokens=10_000, cache_read_tokens=10_000, cache_write_tokens=10_000),
|
||||
provider="openai",
|
||||
)
|
||||
above = estimate_usage_cost(
|
||||
"gpt-6-astra",
|
||||
CanonicalUsage(input_tokens=100_000, output_tokens=10_000, cache_read_tokens=100_000, cache_write_tokens=100_001),
|
||||
provider="openai",
|
||||
)
|
||||
|
||||
assert below.amount_usd == Decimal("1.635")
|
||||
assert above.amount_usd == Decimal("5.450025")
|
||||
assert above.pricing_version == "openai-gpt-6-astra-2026-09"
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -39,6 +39,42 @@ class TestCodexTransportBasic:
|
||||
|
||||
class TestCodexBuildKwargs:
|
||||
|
||||
def test_astra_direct_request_applies_model_contract_after_overrides(self, transport):
|
||||
kw = transport.build_kwargs(
|
||||
model="gpt-6-astra",
|
||||
messages=[{"role": "user", "content": "Hi"}],
|
||||
tools=[],
|
||||
base_url="https://api.openai.com/v1",
|
||||
reasoning_config={"enabled": False, "effort": "none"},
|
||||
request_overrides={
|
||||
"temperature": 0.4,
|
||||
"top_p": 0.9,
|
||||
"top_logprobs": 5,
|
||||
"logprobs": True,
|
||||
"include": ["reasoning.encrypted_content", "message.output_text.logprobs"],
|
||||
"prompt_cache_retention": "24h",
|
||||
"prompt_cache_options": {"ttl": "1h"},
|
||||
},
|
||||
)
|
||||
|
||||
assert kw["reasoning"]["effort"] == "low"
|
||||
assert kw["prompt_cache_options"] == {"ttl": "30m"}
|
||||
assert "prompt_cache_retention" not in kw
|
||||
assert kw["include"] == ["reasoning.encrypted_content"]
|
||||
for unsupported in ("temperature", "top_p", "top_logprobs", "logprobs"):
|
||||
assert unsupported not in kw
|
||||
|
||||
def test_astra_proxy_does_not_receive_official_cache_options(self, transport):
|
||||
kw = transport.build_kwargs(
|
||||
model="gpt-6-astra",
|
||||
messages=[{"role": "user", "content": "Hi"}],
|
||||
tools=[],
|
||||
base_url="https://responses.example.com/v1",
|
||||
request_overrides={"prompt_cache_options": {"ttl": "30m"}},
|
||||
)
|
||||
|
||||
assert "prompt_cache_options" not in kw
|
||||
|
||||
def test_900k_context_variant_suffix_stripped_on_wire(self, transport):
|
||||
"""``-900k`` large-context picker variants are Hermes-side aliases —
|
||||
the Codex backend only knows the base slug, so build_kwargs must
|
||||
|
||||
@@ -112,6 +112,28 @@ def test_fetch_from_api_keeps_supported_in_api_false_models(monkeypatch):
|
||||
assert "gpt-5-internal" not in models
|
||||
|
||||
|
||||
def test_astra_requires_live_codex_account_discovery(monkeypatch, tmp_path):
|
||||
"""Cached/configured Astra names must not manufacture current OAuth entitlement."""
|
||||
from hermes_cli import codex_models
|
||||
|
||||
(tmp_path / "config.toml").write_text('model = "gpt-6-astra"\n', encoding="utf-8")
|
||||
(tmp_path / "models_cache.json").write_text(
|
||||
json.dumps({"models": [{"slug": "gpt-6-astra", "priority": 0}]}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
monkeypatch.setenv("CODEX_HOME", str(tmp_path))
|
||||
monkeypatch.setattr(codex_models, "_fetch_models_from_api", lambda _token: [])
|
||||
|
||||
assert "gpt-6-astra" not in get_codex_model_ids(access_token="stale-token")
|
||||
|
||||
monkeypatch.setattr(
|
||||
codex_models,
|
||||
"_fetch_models_from_api",
|
||||
lambda _token: codex_models._finalize_codex_models(["gpt-6-astra"], allow_astra=True),
|
||||
)
|
||||
assert "gpt-6-astra" in get_codex_model_ids(access_token="entitled-token")
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -238,4 +260,3 @@ class TestNormalizeModelForProvider:
|
||||
assert changed is True
|
||||
# Uses first from available list
|
||||
assert cli.model == "gpt-5.3-codex"
|
||||
|
||||
|
||||
@@ -70,3 +70,16 @@ def test_default_openai_endpoint_intersects_account_access(monkeypatch):
|
||||
assert result == list(curated[:2])
|
||||
|
||||
|
||||
def test_astra_is_offered_only_by_successful_account_discovery(monkeypatch):
|
||||
"""A gated preview may enrich the picker only when this API key lists it."""
|
||||
monkeypatch.setenv("OPENAI_API_KEY", "sk-fake")
|
||||
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
||||
|
||||
with patch.object(M, "fetch_api_models", return_value=["gpt-5.6-sol", "gpt-6-astra"]):
|
||||
discovered = M.provider_model_ids("openai-api", force_refresh=True)
|
||||
with patch.object(M, "fetch_api_models", return_value=["gpt-5.6-sol"]):
|
||||
not_entitled = M.provider_model_ids("openai-api", force_refresh=True)
|
||||
|
||||
assert "gpt-6-astra" in discovered
|
||||
assert "gpt-6-astra" not in not_entitled
|
||||
|
||||
|
||||
Reference in New Issue
Block a user