feat(openai): add GPT-6 Astra baseline support

(cherry picked from commit a8c53d20c6b16cc35745e364e16bb7259166a1d3)
This commit is contained in:
Eva
2026-09-04 23:38:08 +07:00
committed by kshitij
parent a7198a8855
commit 2c315ff59b
12 changed files with 201 additions and 12 deletions

View File

@@ -770,7 +770,7 @@ _PREFLIGHT_OPTIONAL_FIELDS: tuple[tuple[str, Callable[[Any], bool], Optional[Cal
# Cache routing/retention and tool-dispatch hints pass through as-is.
*(
(key, lambda v: v is not None, None)
for key in ("tool_choice", "parallel_tool_calls", "prompt_cache_key", "prompt_cache_retention")
for key in ("tool_choice", "parallel_tool_calls", "prompt_cache_key", "prompt_cache_retention", "prompt_cache_options")
),
# Native compaction directive; eligibility is resolved in agent/native_compaction.py.
("context_management", lambda v: isinstance(v, list) and bool(v), None),

View File

@@ -522,6 +522,19 @@ _OVERRIDE_WARNED_KEYS: set = set()
# shared by get_model_capabilities and get_model_info so the two unknown-model paths agree.
_UNKNOWN_MODEL_BASE: Dict[str, Any] = {"limit": {"context": 200000, "output": 8192}, "tool_call": True}
# Account-gated models may be usable before models.dev has indexed them. Keep
# their capabilities available for an explicitly selected/discovered model
# without adding them to any picker catalog.
_BUILTIN_MODEL_METADATA: Dict[Tuple[str, str], Dict[str, Any]] = {
("openai", "gpt-6-astra"): {
"limit": {"context": 1_050_000, "output": 128_000},
"modalities": {"input": ["text", "image"], "output": ["text"]},
"tool_call": True,
"reasoning": True,
"family": "gpt-6",
},
}
def _load_model_overrides() -> Dict[str, Any]:
"""The ``model_overrides`` config section ({} on any failure). Deliberately not memoized:
@@ -646,8 +659,11 @@ def _merge_catalog_entry_with_override(raw: Dict[str, Any], override: Dict[str,
def _apply_overrides(provider: str, model: str, entry: Optional[Dict[str, Any]]) -> Optional[Dict[str, Any]]:
"""*entry* patched by its override; ``_UNKNOWN_MODEL_BASE`` patched by a fill-gap override on a
catalog miss (selected AFTER lookup: _default only fills misses); None when neither exists."""
override = _override_for(provider, model, catalog_hit=entry is not None)
return entry if override is None else _merge_catalog_entry_with_override(entry if entry is not None else _UNKNOWN_MODEL_BASE, override)
provider_key = PROVIDER_TO_MODELS_DEV.get((provider or "").strip(), (provider or "").strip())
builtin = _BUILTIN_MODEL_METADATA.get((provider_key, (model or "").strip().lower()))
base = entry if entry is not None else builtin
override = _override_for(provider, model, catalog_hit=base is not None)
return base if override is None else _merge_catalog_entry_with_override(base if base is not None else _UNKNOWN_MODEL_BASE, override)
def _entry_supports_vision(entry: Dict[str, Any]) -> bool:

View File

@@ -31,6 +31,9 @@ OPENAI_COMPAT_WIRE_EFFORTS: tuple[str, ...] = ("none", "minimal", "low", "medium
#: both (clamps to low); ``max`` is gpt-5.6-only.
CODEX_GPT56_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh", "max")
CODEX_LEGACY_EFFORTS: tuple[str, ...] = ("none", "low", "medium", "high", "xhigh")
# GPT-6 Astra is account-gated and its Responses API accepts no disable/minimal
# wire level; callers normalize those requests to ``low`` at the transport boundary.
CODEX_ASTRA_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "xhigh", "max")
#: xAI Responses — Grok 4.6+ accepts xhigh; older Grok tops out at high.
XAI_GROK46_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "xhigh")
@@ -80,6 +83,8 @@ META_AI_EFFORTS: tuple[str, ...] = ("minimal", "low", "medium", "high", "xhigh")
def codex_supported_efforts(model: Optional[str]) -> tuple[str, ...]:
"""Supported effort set for an OpenAI/Codex Responses model."""
if (model or "").strip().lower().rsplit("/", 1)[-1] == "gpt-6-astra":
return CODEX_ASTRA_EFFORTS
return CODEX_GPT56_EFFORTS if "gpt-5.6" in (model or "").lower() else CODEX_LEGACY_EFFORTS

View File

@@ -11,7 +11,7 @@ import re
from typing import Any, Callable, Optional
from agent.reasoning_effort import (
ACTUAL_RELAY_EFFORTS, XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort,
ACTUAL_RELAY_EFFORTS, CODEX_ASTRA_EFFORTS, XAI_GROK46_EFFORTS, XAI_LEGACY_EFFORTS, clamp_effort,
# Same declared vocabulary + shared clamp as the main Codex transport (agent.reasoning_effort):
# per-model — "max" is gpt-5.6-only, "minimal"/"ultra" always rejected (live-verified, #68365).
codex_supported_efforts,
@@ -211,6 +211,15 @@ def _resolve_reasoning(model: str, params: dict[str, Any]) -> tuple[Any, bool]:
elif reasoning_config.get("effort"):
reasoning_effort = reasoning_config["effort"]
# Astra has no wire-level disable/minimal setting. Preserve Hermes' user-facing
# controls by sending the documented lowest enabled level instead; this also keeps
# an explicit ``enabled: false`` request valid on the Responses API.
if model.strip().lower().rsplit("/", 1)[-1] == "gpt-6-astra":
requested = str(reasoning_effort or "").strip().lower()
if not reasoning_enabled or requested in {"", "none", "minimal", "disabled", "off"}:
return "low", True
return clamp_effort(reasoning_effort, CODEX_ASTRA_EFFORTS), True
# Wire vocabularies are declared in agent.reasoning_effort; the shared clamp policy (nearest weaker
# supported level, never escalate, never invert the ladder) replaces the per-backend hand maps that
# repeatedly leaked internal levels like "ultra" to the wire (#89503 class) or clamped one rung below a
@@ -262,6 +271,37 @@ def _default_prompt_cache_retention_for_request(model: str, base_url: Any) -> Op
return "24h" if _EXTENDED_PROMPT_CACHE_MODEL_RE.search(normalized) else None
def _is_astra_model(model: Any) -> bool:
return str(model or "").strip().lower().rsplit("/", 1)[-1] == "gpt-6-astra"
def _is_official_openai_responses_route(model: Any, base_url: Any) -> bool:
"""Astra cache semantics apply only to the official OpenAI API host."""
if not _is_astra_model(model):
return False
from utils import base_url_host_matches
return base_url_host_matches(str(base_url or ""), "api.openai.com")
def _sanitize_astra_request_kwargs(kwargs: dict[str, Any], model: Any, base_url: Any) -> None:
"""Apply Astra's model-specific restrictions after all request overrides are merged."""
if not _is_astra_model(model):
return
for key in ("temperature", "top_p", "top_logprobs", "logprobs"):
kwargs.pop(key, None)
include = kwargs.get("include")
if isinstance(include, list):
kwargs["include"] = [item for item in include if "logprob" not in str(item).lower()]
if _is_official_openai_responses_route(model, base_url):
kwargs["prompt_cache_options"] = {"ttl": "30m"}
kwargs.pop("prompt_cache_retention", None)
else:
# A caller override must not accidentally send official-only cache syntax to a
# proxy or another provider that happens to accept the Astra model name.
kwargs.pop("prompt_cache_options", None)
def _content_cache_key(instructions: str, tools: Optional[list[dict[str, Any]]], scope_id: str = "") -> Optional[str]:
"""``pck_<sha256[:24]>`` of (scope_id, instructions, name-sorted tools), or None if nothing static.
@@ -544,6 +584,8 @@ class ResponsesApiTransport(ProviderTransport):
if params.get("request_overrides"):
kwargs.update(params["request_overrides"])
_sanitize_astra_request_kwargs(kwargs, model, params.get("base_url"))
_bound_prompt_cache_key_field(kwargs)
# Older xAI models reject ``service_tier`` (HTTP 400); only Grok 4.6 accepts Priority Processing.

View File

@@ -109,6 +109,7 @@ class PricingEntry:
input_cost_per_million_above: Optional[Decimal] = None
output_cost_per_million_above: Optional[Decimal] = None
cache_read_cost_per_million_above: Optional[Decimal] = None
cache_write_cost_per_million_above: Optional[Decimal] = None
@dataclass(frozen=True)
@@ -241,6 +242,20 @@ for _provider, _url, _version, _rows in _SNAPSHOTS:
_OFFICIAL_DOCS_PRICING[(_provider, _model)] = _entry
del _SNAPSHOTS, _provider, _url, _version, _rows, _models, _rates, _entry, _model
# GPT-6 Astra uses whole-request pricing above the 272K prompt tier. Keep this
# account-gated model out of generic static catalogs, but retain published billing
# metadata for an explicitly selected route.
_OFFICIAL_DOCS_PRICING[("openai", "gpt-6-astra")] = _snap(
"10.00", "50.00", "1.00", "12.50",
url="https://developers.openai.com/api/docs/models/gpt-6-astra",
version="openai-gpt-6-astra-2026-09",
tier_threshold_tokens=272_000,
input_cost_per_million_above=Decimal("20.00"),
output_cost_per_million_above=Decimal("75.00"),
cache_read_cost_per_million_above=Decimal("2.00"),
cache_write_cost_per_million_above=Decimal("25.00"),
)
# Context-tiered Gemini Pro: above 200k prompt tokens the *_above rates apply to
# the whole request (see PricingEntry).
_OFFICIAL_DOCS_PRICING[("google", "gemini-3.1-pro")] = _snap(
@@ -558,7 +573,7 @@ def estimate_usage_cost(
(usage.output_tokens, entry.output_cost_per_million, entry.output_cost_per_million_above, ()),
(usage.cache_read_tokens, entry.cache_read_cost_per_million, entry.cache_read_cost_per_million_above,
("cache-read pricing unavailable for route",)),
(usage.cache_write_tokens, entry.cache_write_cost_per_million, None,
(usage.cache_write_tokens, entry.cache_write_cost_per_million, entry.cache_write_cost_per_million_above,
("cache-write pricing unavailable for route",)),
):
if above and rate_above is not None:

View File

@@ -88,9 +88,16 @@ def _add_context_variants(model_ids: List[str]) -> List[str]:
return out
def _finalize_codex_models(model_ids: List[str]) -> List[str]:
"""Forward-compat synthesis + large-context variant synthesis."""
return _add_context_variants(_add_forward_compat_models(model_ids))
def _finalize_codex_models(model_ids: List[str], *, allow_astra: bool = False) -> List[str]:
"""Forward-compat/context synthesis with an entitlement gate for Astra.
Cached/configured model names are useful compatibility hints, but only the
account-scoped Codex endpoint is authoritative for current Astra access.
"""
finalized = _add_forward_compat_models(model_ids)
if not allow_astra:
finalized = [model for model in finalized if model.lower() != "gpt-6-astra"]
return _add_context_variants(finalized)
def _extract_chatgpt_account_id(access_token: str) -> Optional[str]:
@@ -159,7 +166,7 @@ def _fetch_models_from_api(access_token: str) -> List[str]:
logger.debug("Failed to fetch Codex models from API: %s", exc)
return []
return _finalize_codex_models(_ranked_slugs(entries))
return _finalize_codex_models(_ranked_slugs(entries), allow_astra=True)
def _read_default_model(codex_home: Path) -> Optional[str]:
@@ -194,7 +201,7 @@ def get_codex_model_ids(access_token: Optional[str] = None) -> List[str]:
if access_token:
api_models = _fetch_models_from_api(access_token)
if api_models:
return _finalize_codex_models(api_models)
return _finalize_codex_models(api_models, allow_astra=True)
default_model = _read_default_model(codex_home)
return _finalize_codex_models(_dedupe([
*([default_model] if default_model else []), *_read_cache_models(codex_home),

View File

@@ -1203,7 +1203,12 @@ def _openai_catalog(normalized: str, force_refresh: bool) -> Optional[list[str]]
curated = list(_PROVIDER_MODELS.get(normalized, []))
# Curated order, only models the account has access to; an account serving none of them (rare)
# falls back to curated so the picker still offers sane defaults.
return [m for m in curated if m.lower() in live_lower] or curated or live
discovered = [m for m in curated if m.lower() in live_lower]
# Astra is intentionally absent from offline/static catalogs: the official API's
# account-scoped /models response is the only source that may advertise it.
if "gpt-6-astra" in live_lower:
discovered.append(next((m for m in live if m.lower() == "gpt-6-astra"), "gpt-6-astra"))
return discovered or curated or live
def _custom_catalog(normalized: str, force_refresh: bool) -> Optional[list[str]]:

View File

@@ -825,6 +825,18 @@ class TestGetModelCapabilities:
assert caps is not None
assert caps.supports_vision is False
def test_astra_builtin_metadata_fills_catalog_lag_without_listing_it(self):
"""An explicitly discovered Astra remains fully described before models.dev catches up."""
with patch("agent.models_dev.fetch_models_dev", return_value={}):
caps = get_model_capabilities("openai", "gpt-6-astra")
assert caps is not None
assert caps.context_window == 1_050_000
assert caps.max_output_tokens == 128_000
assert caps.supports_tools is True
assert caps.supports_vision is True
assert caps.supports_reasoning is True
# ---------------------------------------------------------------------------
# Per-model metadata overrides (model_overrides config)

View File

@@ -11,6 +11,23 @@ from agent.usage_pricing import (
from decimal import Decimal
def test_astra_whole_request_price_tier_includes_cache_writes():
below = estimate_usage_cost(
"gpt-6-astra",
CanonicalUsage(input_tokens=100_000, output_tokens=10_000, cache_read_tokens=10_000, cache_write_tokens=10_000),
provider="openai",
)
above = estimate_usage_cost(
"gpt-6-astra",
CanonicalUsage(input_tokens=100_000, output_tokens=10_000, cache_read_tokens=100_000, cache_write_tokens=100_001),
provider="openai",
)
assert below.amount_usd == Decimal("1.635")
assert above.amount_usd == Decimal("5.450025")
assert above.pricing_version == "openai-gpt-6-astra-2026-09"

View File

@@ -39,6 +39,42 @@ class TestCodexTransportBasic:
class TestCodexBuildKwargs:
def test_astra_direct_request_applies_model_contract_after_overrides(self, transport):
kw = transport.build_kwargs(
model="gpt-6-astra",
messages=[{"role": "user", "content": "Hi"}],
tools=[],
base_url="https://api.openai.com/v1",
reasoning_config={"enabled": False, "effort": "none"},
request_overrides={
"temperature": 0.4,
"top_p": 0.9,
"top_logprobs": 5,
"logprobs": True,
"include": ["reasoning.encrypted_content", "message.output_text.logprobs"],
"prompt_cache_retention": "24h",
"prompt_cache_options": {"ttl": "1h"},
},
)
assert kw["reasoning"]["effort"] == "low"
assert kw["prompt_cache_options"] == {"ttl": "30m"}
assert "prompt_cache_retention" not in kw
assert kw["include"] == ["reasoning.encrypted_content"]
for unsupported in ("temperature", "top_p", "top_logprobs", "logprobs"):
assert unsupported not in kw
def test_astra_proxy_does_not_receive_official_cache_options(self, transport):
kw = transport.build_kwargs(
model="gpt-6-astra",
messages=[{"role": "user", "content": "Hi"}],
tools=[],
base_url="https://responses.example.com/v1",
request_overrides={"prompt_cache_options": {"ttl": "30m"}},
)
assert "prompt_cache_options" not in kw
def test_900k_context_variant_suffix_stripped_on_wire(self, transport):
"""``-900k`` large-context picker variants are Hermes-side aliases —
the Codex backend only knows the base slug, so build_kwargs must

View File

@@ -112,6 +112,28 @@ def test_fetch_from_api_keeps_supported_in_api_false_models(monkeypatch):
assert "gpt-5-internal" not in models
def test_astra_requires_live_codex_account_discovery(monkeypatch, tmp_path):
"""Cached/configured Astra names must not manufacture current OAuth entitlement."""
from hermes_cli import codex_models
(tmp_path / "config.toml").write_text('model = "gpt-6-astra"\n', encoding="utf-8")
(tmp_path / "models_cache.json").write_text(
json.dumps({"models": [{"slug": "gpt-6-astra", "priority": 0}]}),
encoding="utf-8",
)
monkeypatch.setenv("CODEX_HOME", str(tmp_path))
monkeypatch.setattr(codex_models, "_fetch_models_from_api", lambda _token: [])
assert "gpt-6-astra" not in get_codex_model_ids(access_token="stale-token")
monkeypatch.setattr(
codex_models,
"_fetch_models_from_api",
lambda _token: codex_models._finalize_codex_models(["gpt-6-astra"], allow_astra=True),
)
assert "gpt-6-astra" in get_codex_model_ids(access_token="entitled-token")
@@ -238,4 +260,3 @@ class TestNormalizeModelForProvider:
assert changed is True
# Uses first from available list
assert cli.model == "gpt-5.3-codex"

View File

@@ -70,3 +70,16 @@ def test_default_openai_endpoint_intersects_account_access(monkeypatch):
assert result == list(curated[:2])
def test_astra_is_offered_only_by_successful_account_discovery(monkeypatch):
"""A gated preview may enrich the picker only when this API key lists it."""
monkeypatch.setenv("OPENAI_API_KEY", "sk-fake")
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
with patch.object(M, "fetch_api_models", return_value=["gpt-5.6-sol", "gpt-6-astra"]):
discovered = M.provider_model_ids("openai-api", force_refresh=True)
with patch.object(M, "fetch_api_models", return_value=["gpt-5.6-sol"]):
not_entitled = M.provider_model_ids("openai-api", force_refresh=True)
assert "gpt-6-astra" in discovered
assert "gpt-6-astra" not in not_entitled