Merge pull request #116344 from NousResearch/boa-w3-session-guard

session.create refuses a model its provider cannot serve instead of a dead first turn (#96817, salvage #96845)
This commit is contained in:
Teknium
2026-09-19 14:32:15 -07:00
committed by GitHub
5 changed files with 130 additions and 2 deletions

View File

@@ -283,6 +283,41 @@ _STATIC_FAMILY_PREFIXES = {
_STATIC_LABELS = {"openai-codex": "OpenAI Codex", "xai-oauth": "xAI Grok OAuth (SuperGrok / Premium+)"}
def _family_head(model_id: str) -> str:
"""Vendor family token of a model id: ``gpt-5.5`` → ``gpt``, ``claude-opus-5`` → ``claude``."""
return re.split(r"[-./:]", model_id.strip().lower(), maxsplit=1)[0]
def static_model_provider_conflict(model_name: str, provider: Optional[str], *, limit: int = 5) -> Optional[dict[str, Any]]:
"""Offline model×provider coherence from the curated catalogs only (no network: this runs on
``session.create``). ``None`` = coherent or undecidable — custom / aggregator / catalog-less
providers, names in the provider's own family (a newer ``gpt-*`` the curated list lacks) and
names no vendor lists (hidden or preview slugs) stay permissive. A conflict is a name outside
the provider's family that another native vendor's catalog lists — or any foreign-family name
on the OAuth catalogs with a strict family gate (``_STATIC_FAMILY_PREFIXES``) (#96817)."""
from hermes_cli import models as _m
requested = (model_name or "").strip()
normalized = _m.normalize_provider(provider)
catalog = list(_m._PROVIDER_MODELS.get(normalized, ()))
if not requested or not catalog or normalized == "moa" or normalized in _m._AGGREGATOR_PROVIDERS:
return None
if _m._model_in_provider_catalog(requested.lower(), _m._provider_keys(normalized)):
return None
if _family_head(requested) in {_family_head(m) for m in catalog}:
return None
strict = normalized in _STATIC_FAMILY_PREFIXES
if not strict and next(_m._static_catalog_matches(requested, normalized), None) is None:
return None
suggestions = get_close_matches(requested, catalog, n=limit, cutoff=0.4) or catalog[:limit]
label = _m._PROVIDER_LABELS.get(normalized, normalized)
return {
"model": requested, "provider": normalized, "suggestions": suggestions,
"message": (f"Model `{requested}` is not served by provider `{normalized}` ({label}). "
f"Closest {label} models: " + ", ".join(f"`{s}`" for s in suggestions) + "."),
}
def _validate_static_catalog(req: _Request) -> Optional[dict[str, Any]]:
"""openai-codex / xai-oauth: no /v1/models probing — validate against the curated catalog.
Returns None (fall through) when the catalog is empty."""

View File

@@ -0,0 +1,57 @@
"""``session.create`` rejects a model the selected provider cannot serve (#96817).
Before the gate the session was minted and the FIRST turn died with the provider's 404 — a dead
chat. Custom / unknown providers and same-family or unlisted names stay permissive.
"""
import pytest
@pytest.fixture
def _create(monkeypatch, tmp_path):
monkeypatch.setattr("hermes_cli.banner.prefetch_update_check", lambda: None)
from tui_gateway import server
(tmp_path / "config.yaml").write_text("model:\n default: claude-opus-5\n provider: anthropic\n", encoding="utf-8")
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
monkeypatch.setattr(server, "_sessions", {})
monkeypatch.setattr(server, "_load_cfg", lambda: {})
monkeypatch.setattr(server, "_profile_home", lambda *a: None)
monkeypatch.setattr(server, "_enable_gateway_prompts", lambda: None)
monkeypatch.setattr(server, "_schedule_agent_build", lambda *a: None)
monkeypatch.setattr(server, "_schedule_session_cap_enforcement", lambda: None)
monkeypatch.setattr(server, "_register_session_cwd", lambda *a: None)
monkeypatch.setattr(server, "_project_info_for_cwd", lambda *a: None)
return lambda params: (server._methods["session.create"]("r1", {"cols": 80, **params}), server._sessions)
@pytest.mark.parametrize("params", [
{"model": "gpt-5.5", "provider": "anthropic"},
{"model": "gpt-5.5"}, # provider implied by the profile config
{"model": "deepseek/deepseek-v4-flash-0731", "provider": "openai-codex"}, # the incident pair
])
def test_session_create_rejects_incoherent_model_provider_pair_before_any_state(_create, params):
response, sessions = _create(params)
error = response["error"]
assert error["code"] == -32602
assert params["model"] in error["message"]
assert error["data"]["provider"] in error["message"]
assert error["data"]["model"] == params["model"]
assert 1 <= len(error["data"]["suggestions"]) <= 5
assert all(s in error["message"] for s in error["data"]["suggestions"])
assert sessions == {}
@pytest.mark.parametrize("params", [
{"model": "claude-opus-5", "provider": "anthropic"}, # coherent
{"model": "claude-opus-5-20261001", "provider": "anthropic"}, # same family, not (yet) listed
{"model": "gpt-5.5", "provider": "custom:local"}, # custom endpoint: Hermes cannot know its models
{"model": "gpt-5.5", "provider": "openrouter"}, # aggregator
])
def test_session_create_keeps_coherent_unlisted_and_custom_pairs(_create, params):
response, sessions = _create(params)
assert "error" not in response, response
session = sessions[response["result"]["session_id"]]
assert session["model_override"] == {"model": params["model"], "provider": params["provider"]}

View File

@@ -325,6 +325,13 @@ def _create_overrides(params: dict) -> tuple:
@method("session.create")
def _(rid, params: dict) -> dict:
# ``profile`` (app-global remote mode): stored so the build and every turn re-bind HERMES_HOME.
profile_home = _profile_home(profile := (params.get("profile") or "").strip() or None)
# Reject an incoherent model×provider pair BEFORE any state exists: minting it only defers the
# failure to the first turn's provider 404 (#96817). Custom/unknown providers stay permissive.
from .methods_session_model_guard import model_override_conflict
if conflict := model_override_conflict(params, _profile_build_scope(profile_home)):
return _err(rid, -32602, conflict.pop("message"), conflict)
(sid, source), key = _new_runtime_ids(params), _new_session_key()
history = _coerce_seed_history(params.get("messages"))
# Branch: links back so list_sessions_rich keeps it visible and the sidebar nests it.
@@ -335,8 +342,6 @@ def _(rid, params: dict) -> dict:
with contextlib.suppress(Exception):
explicit_cwd = bool(raw_cwd) and os.path.isdir(os.path.abspath(os.path.expanduser(raw_cwd)))
_enable_gateway_prompts()
# ``profile`` (app-global remote mode): stored so the build and every turn re-bind HERMES_HOME.
profile_home = _profile_home(profile := (params.get("profile") or "").strip() or None)
session_model_override, create_reasoning_override, create_service_tier_override = _create_overrides(params)
now = time.time()
with _sessions_lock:

View File

@@ -0,0 +1,27 @@
"""``session.create`` model×provider coherence gate (#96817).
A composer, script or older client can pin a model the selected provider cannot serve
(``gpt-5.5`` on ``anthropic``); the session used to be minted fine and the FIRST turn died with
the provider's 404, leaving a dead chat. The gate is offline (curated catalogs only) and stays
permissive wherever Hermes cannot know better — see ``models_validate.static_model_provider_conflict``.
"""
from __future__ import annotations
def model_override_conflict(params: dict, build_scope) -> dict | None:
"""The conflict record for the create params' model override, or ``None`` when coherent /
undecidable. Without an explicit ``provider`` the pair is judged against the provider the
session would actually build with (profile config, then env) inside ``build_scope`` — the
handler's ``_profile_build_scope(profile_home)`` context manager."""
model = str(params.get("model") or "").strip()
if not model:
return None
from hermes_cli.models_validate import static_model_provider_conflict
from hermes_cli.runtime_provider import resolve_requested_provider
provider = str(params.get("provider") or "").strip()
if not provider:
with build_scope:
provider = resolve_requested_provider()
return static_model_provider_conflict(model, provider)

View File

@@ -59,6 +59,10 @@ terminal.resize clipboard.paste image.attach
Within one authenticated gateway, resuming or activating a live session attaches another event subscriber rather than replacing the previous connection. Streaming and terminal events go to all attached clients; disconnecting one client does not end a session another client is viewing. Existing submit exclusivity and configured busy-input policy remain in force. Attached clients can steer the session's subagents; browser-controller results still require the connection that registered that controller. This does not enable independent gateway processes to write the same session, nor does it imply durable prompt admission across an owner restart.
### Model overrides on `session.create`
`session.create` accepts per-session `model` / `provider` overrides. A pair the provider cannot serve (`model: gpt-5.5` with `provider: anthropic`, or with no `provider` when the profile's configured provider is Anthropic) is refused up front with JSON-RPC code `-32602` instead of minting a session whose first turn fails at the provider; `error.data` carries `model`, `provider` and up to five `suggestions` from that provider's catalog, and `error.message` repeats them. The check is offline and only refuses names Hermes knows belong elsewhere: custom endpoints (`custom`, `custom:<name>`), aggregators (OpenRouter, Nous, …), models in the provider's own family that the curated list has not caught up with, and names no catalog lists are all accepted as before.
### Rewinding history on `prompt.submit`
A rewind / edit / regenerate is a `prompt.submit` that drops part of the stored transcript before running the new turn. Because that write is a destructive rewrite of the session's durable rows, the gateway honors it only when the client states its intent: