diff --git a/hermes_cli/models_validate.py b/hermes_cli/models_validate.py index 8eccd07c06..7e98b402b5 100644 --- a/hermes_cli/models_validate.py +++ b/hermes_cli/models_validate.py @@ -283,6 +283,41 @@ _STATIC_FAMILY_PREFIXES = { _STATIC_LABELS = {"openai-codex": "OpenAI Codex", "xai-oauth": "xAI Grok OAuth (SuperGrok / Premium+)"} +def _family_head(model_id: str) -> str: + """Vendor family token of a model id: ``gpt-5.5`` → ``gpt``, ``claude-opus-5`` → ``claude``.""" + return re.split(r"[-./:]", model_id.strip().lower(), maxsplit=1)[0] + + +def static_model_provider_conflict(model_name: str, provider: Optional[str], *, limit: int = 5) -> Optional[dict[str, Any]]: + """Offline model×provider coherence from the curated catalogs only (no network: this runs on + ``session.create``). ``None`` = coherent or undecidable — custom / aggregator / catalog-less + providers, names in the provider's own family (a newer ``gpt-*`` the curated list lacks) and + names no vendor lists (hidden or preview slugs) stay permissive. A conflict is a name outside + the provider's family that another native vendor's catalog lists — or any foreign-family name + on the OAuth catalogs with a strict family gate (``_STATIC_FAMILY_PREFIXES``) (#96817).""" + from hermes_cli import models as _m + + requested = (model_name or "").strip() + normalized = _m.normalize_provider(provider) + catalog = list(_m._PROVIDER_MODELS.get(normalized, ())) + if not requested or not catalog or normalized == "moa" or normalized in _m._AGGREGATOR_PROVIDERS: + return None + if _m._model_in_provider_catalog(requested.lower(), _m._provider_keys(normalized)): + return None + if _family_head(requested) in {_family_head(m) for m in catalog}: + return None + strict = normalized in _STATIC_FAMILY_PREFIXES + if not strict and next(_m._static_catalog_matches(requested, normalized), None) is None: + return None + suggestions = get_close_matches(requested, catalog, n=limit, cutoff=0.4) or catalog[:limit] + label = _m._PROVIDER_LABELS.get(normalized, normalized) + return { + "model": requested, "provider": normalized, "suggestions": suggestions, + "message": (f"Model `{requested}` is not served by provider `{normalized}` ({label}). " + f"Closest {label} models: " + ", ".join(f"`{s}`" for s in suggestions) + "."), + } + + def _validate_static_catalog(req: _Request) -> Optional[dict[str, Any]]: """openai-codex / xai-oauth: no /v1/models probing — validate against the curated catalog. Returns None (fall through) when the catalog is empty.""" diff --git a/tests/tui_gateway/test_session_create_model_provider_guard.py b/tests/tui_gateway/test_session_create_model_provider_guard.py new file mode 100644 index 0000000000..b108276a1d --- /dev/null +++ b/tests/tui_gateway/test_session_create_model_provider_guard.py @@ -0,0 +1,57 @@ +"""``session.create`` rejects a model the selected provider cannot serve (#96817). + +Before the gate the session was minted and the FIRST turn died with the provider's 404 — a dead +chat. Custom / unknown providers and same-family or unlisted names stay permissive. +""" + +import pytest + + +@pytest.fixture +def _create(monkeypatch, tmp_path): + monkeypatch.setattr("hermes_cli.banner.prefetch_update_check", lambda: None) + from tui_gateway import server + + (tmp_path / "config.yaml").write_text("model:\n default: claude-opus-5\n provider: anthropic\n", encoding="utf-8") + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + monkeypatch.setattr(server, "_sessions", {}) + monkeypatch.setattr(server, "_load_cfg", lambda: {}) + monkeypatch.setattr(server, "_profile_home", lambda *a: None) + monkeypatch.setattr(server, "_enable_gateway_prompts", lambda: None) + monkeypatch.setattr(server, "_schedule_agent_build", lambda *a: None) + monkeypatch.setattr(server, "_schedule_session_cap_enforcement", lambda: None) + monkeypatch.setattr(server, "_register_session_cwd", lambda *a: None) + monkeypatch.setattr(server, "_project_info_for_cwd", lambda *a: None) + return lambda params: (server._methods["session.create"]("r1", {"cols": 80, **params}), server._sessions) + + +@pytest.mark.parametrize("params", [ + {"model": "gpt-5.5", "provider": "anthropic"}, + {"model": "gpt-5.5"}, # provider implied by the profile config + {"model": "deepseek/deepseek-v4-flash-0731", "provider": "openai-codex"}, # the incident pair +]) +def test_session_create_rejects_incoherent_model_provider_pair_before_any_state(_create, params): + response, sessions = _create(params) + + error = response["error"] + assert error["code"] == -32602 + assert params["model"] in error["message"] + assert error["data"]["provider"] in error["message"] + assert error["data"]["model"] == params["model"] + assert 1 <= len(error["data"]["suggestions"]) <= 5 + assert all(s in error["message"] for s in error["data"]["suggestions"]) + assert sessions == {} + + +@pytest.mark.parametrize("params", [ + {"model": "claude-opus-5", "provider": "anthropic"}, # coherent + {"model": "claude-opus-5-20261001", "provider": "anthropic"}, # same family, not (yet) listed + {"model": "gpt-5.5", "provider": "custom:local"}, # custom endpoint: Hermes cannot know its models + {"model": "gpt-5.5", "provider": "openrouter"}, # aggregator +]) +def test_session_create_keeps_coherent_unlisted_and_custom_pairs(_create, params): + response, sessions = _create(params) + + assert "error" not in response, response + session = sessions[response["result"]["session_id"]] + assert session["model_override"] == {"model": params["model"], "provider": params["provider"]} diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index 5932ceb160..042bfa3ebd 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -325,6 +325,13 @@ def _create_overrides(params: dict) -> tuple: @method("session.create") def _(rid, params: dict) -> dict: + # ``profile`` (app-global remote mode): stored so the build and every turn re-bind HERMES_HOME. + profile_home = _profile_home(profile := (params.get("profile") or "").strip() or None) + # Reject an incoherent model×provider pair BEFORE any state exists: minting it only defers the + # failure to the first turn's provider 404 (#96817). Custom/unknown providers stay permissive. + from .methods_session_model_guard import model_override_conflict + if conflict := model_override_conflict(params, _profile_build_scope(profile_home)): + return _err(rid, -32602, conflict.pop("message"), conflict) (sid, source), key = _new_runtime_ids(params), _new_session_key() history = _coerce_seed_history(params.get("messages")) # Branch: links back so list_sessions_rich keeps it visible and the sidebar nests it. @@ -335,8 +342,6 @@ def _(rid, params: dict) -> dict: with contextlib.suppress(Exception): explicit_cwd = bool(raw_cwd) and os.path.isdir(os.path.abspath(os.path.expanduser(raw_cwd))) _enable_gateway_prompts() - # ``profile`` (app-global remote mode): stored so the build and every turn re-bind HERMES_HOME. - profile_home = _profile_home(profile := (params.get("profile") or "").strip() or None) session_model_override, create_reasoning_override, create_service_tier_override = _create_overrides(params) now = time.time() with _sessions_lock: diff --git a/tui_gateway/methods_session_model_guard.py b/tui_gateway/methods_session_model_guard.py new file mode 100644 index 0000000000..1a1d90dcf0 --- /dev/null +++ b/tui_gateway/methods_session_model_guard.py @@ -0,0 +1,27 @@ +"""``session.create`` model×provider coherence gate (#96817). + +A composer, script or older client can pin a model the selected provider cannot serve +(``gpt-5.5`` on ``anthropic``); the session used to be minted fine and the FIRST turn died with +the provider's 404, leaving a dead chat. The gate is offline (curated catalogs only) and stays +permissive wherever Hermes cannot know better — see ``models_validate.static_model_provider_conflict``. +""" + +from __future__ import annotations + + +def model_override_conflict(params: dict, build_scope) -> dict | None: + """The conflict record for the create params' model override, or ``None`` when coherent / + undecidable. Without an explicit ``provider`` the pair is judged against the provider the + session would actually build with (profile config, then env) inside ``build_scope`` — the + handler's ``_profile_build_scope(profile_home)`` context manager.""" + model = str(params.get("model") or "").strip() + if not model: + return None + from hermes_cli.models_validate import static_model_provider_conflict + from hermes_cli.runtime_provider import resolve_requested_provider + + provider = str(params.get("provider") or "").strip() + if not provider: + with build_scope: + provider = resolve_requested_provider() + return static_model_provider_conflict(model, provider) diff --git a/website/docs/developer-guide/programmatic-integration.md b/website/docs/developer-guide/programmatic-integration.md index 45adae4604..4b050ad822 100644 --- a/website/docs/developer-guide/programmatic-integration.md +++ b/website/docs/developer-guide/programmatic-integration.md @@ -59,6 +59,10 @@ terminal.resize clipboard.paste image.attach Within one authenticated gateway, resuming or activating a live session attaches another event subscriber rather than replacing the previous connection. Streaming and terminal events go to all attached clients; disconnecting one client does not end a session another client is viewing. Existing submit exclusivity and configured busy-input policy remain in force. Attached clients can steer the session's subagents; browser-controller results still require the connection that registered that controller. This does not enable independent gateway processes to write the same session, nor does it imply durable prompt admission across an owner restart. +### Model overrides on `session.create` + +`session.create` accepts per-session `model` / `provider` overrides. A pair the provider cannot serve (`model: gpt-5.5` with `provider: anthropic`, or with no `provider` when the profile's configured provider is Anthropic) is refused up front with JSON-RPC code `-32602` instead of minting a session whose first turn fails at the provider; `error.data` carries `model`, `provider` and up to five `suggestions` from that provider's catalog, and `error.message` repeats them. The check is offline and only refuses names Hermes knows belong elsewhere: custom endpoints (`custom`, `custom:`), aggregators (OpenRouter, Nous, …), models in the provider's own family that the curated list has not caught up with, and names no catalog lists are all accepted as before. + ### Rewinding history on `prompt.submit` A rewind / edit / regenerate is a `prompt.submit` that drops part of the stored transcript before running the new turn. Because that write is a destructive rewrite of the session's durable rows, the gateway honors it only when the client states its intent: