diff --git a/hermes_cli/cli_model_switch_mixin.py b/hermes_cli/cli_model_switch_mixin.py index 82062fc6bf..ccc91ac249 100644 --- a/hermes_cli/cli_model_switch_mixin.py +++ b/hermes_cli/cli_model_switch_mixin.py @@ -59,7 +59,18 @@ def stored_session_route(session_meta, *, current_model, current_provider): provider_changed = bool(provider) and provider != current_provider if stored_model == current_model and not provider_changed: return None - return stored_model, provider, base_url, (runtime.get("api_mode") or None), provider_changed + api_mode = runtime.get("api_mode") or None + # A row's api_mode/base_url were written for whichever model the session last ran. Providers that + # pick the wire per model (OpenCode Zen/Go, Copilot, Nous) re-derive both from the stored model, or a + # resumed opencode-go session keeps a MiniMax-era anthropic_messages route for a chat_completions + # model (#96066) — the CLI/oneshot twin of tui_gateway's _rederive_per_model_route. + from hermes_cli.model_switch import model_derived_api_mode + derived = model_derived_api_mode(provider or "", stored_model) + if derived is not None: + from hermes_cli.models import normalize_opencode_base_url + api_mode = derived + base_url = normalize_opencode_base_url(provider, api_mode, base_url) or None + return stored_model, provider, base_url, api_mode, provider_changed def _heal_bare_custom_provider(provider, *, base_url, model): diff --git a/tests/hermes_cli/test_resume_model_restore.py b/tests/hermes_cli/test_resume_model_restore.py index e9f14b4009..9d842fce64 100644 --- a/tests/hermes_cli/test_resume_model_restore.py +++ b/tests/hermes_cli/test_resume_model_restore.py @@ -307,6 +307,26 @@ def test_restore_session_model_heals_bare_custom_stored_rows(monkeypatch): assert stub.provider == "openrouter" +def test_restore_session_model_rederives_per_model_wire_for_opencode_rows(monkeypatch): + """A row persisted while an opencode-go session ran an anthropic_messages model (MiniMax) must not + pin that wire onto a chat_completions model on resume — api_mode and the relay URL follow the + stored model, and a fixed-wire provider's row is still honored verbatim (#96066).""" + import hermes_cli.runtime_provider as rp + monkeypatch.setattr(rp, "resolve_runtime_provider", lambda **kw: {"api_key": "go-key"}) + stub = _make_stub(provider="opencode-go", requested_provider="opencode-go", + base_url="https://opencode.ai/zen/go/v1", api_mode="chat_completions") + stub._restore_session_model(_row(model="deepseek-v4-flash-vision-exp", model_config={ + "gateway_runtime": {"provider": "opencode-go", "base_url": "https://opencode.ai/zen/go", + "api_mode": "anthropic_messages"}})) + assert (stub.api_mode, stub.base_url) == ("chat_completions", "https://opencode.ai/zen/go/v1") + + stub = _make_stub() + stub._restore_session_model(_row(model="MiniMax-M2.5", model_config={ + "gateway_runtime": {"provider": "minimax", "base_url": "https://api.minimax.io/anthropic", + "api_mode": "anthropic_messages"}})) + assert (stub.api_mode, stub.base_url) == ("anthropic_messages", "https://api.minimax.io/anthropic") + + # ── round trip: persist → get_session shape → restore ─────────────── diff --git a/website/docs/integrations/providers.md b/website/docs/integrations/providers.md index 785dbb8256..53c80a49bf 100644 --- a/website/docs/integrations/providers.md +++ b/website/docs/integrations/providers.md @@ -62,7 +62,7 @@ You need at least one way to connect to an LLM. Use `hermes model` to switch pro Both built-in OpenCode providers send an opaque, per-conversation `x-opencode-session` header on every request (main turns on every transport plus auxiliary calls such as compression, titles, approval checks, skills-hub lookups and `/btw` side questions — including the ones that run in the background after the turn has ended; headless Kanban `specify`/`decompose` and dashboard estimate calls use a per-task key). OpenCode uses it to pin a conversation to one backend so its prompt cache stays warm; the value is derived from the Hermes session id (or the Kanban task id) and carries no personal data. -The two built-in OpenCode providers each pin their own relay on `opencode.ai` (`opencode-zen` → `/zen/v1`, `opencode-go` → `/zen/go/v1`). A `model.base_url` left behind by the other relay is healed to the selected provider's relay, and the model you pick (`-m`, `/model`, a fallback entry or a channel override) decides which relay is used — so switching from a Zen model to a Go-only one never sends the request to Zen. A custom provider you define under `providers:` whose name extends a family slug (for example `opencode-go-bridge`) still gets the family's per-model API-mode routing and `/v1` handling, but its `base_url` is taken as declared: name it after the relay it actually points at. +The two built-in OpenCode providers each pin their own relay on `opencode.ai` (`opencode-zen` → `/zen/v1`, `opencode-go` → `/zen/go/v1`). A `model.base_url` left behind by the other relay is healed to the selected provider's relay, and the model you pick (`-m`, `/model`, a fallback entry or a channel override) decides which relay is used — so switching from a Zen model to a Go-only one never sends the request to Zen. The same per-model routing is re-applied when a session is resumed (`hermes --resume`, `/resume`, the TUI and desktop resume paths): a wire format or relay URL recorded while the session ran a different OpenCode model never carries over to the model the session is reopened on. OpenCode models whose id carries a `-vision` marker (for example `deepseek-v4-flash-vision-exp`) are treated as vision-capable even before models.dev lists them, so `agent.image_input_mode: auto` attaches images natively without a `supports_vision` override. A custom provider you define under `providers:` whose name extends a family slug (for example `opencode-go-bridge`) still gets the family's per-model API-mode routing and `/v1` handling, but its `base_url` is taken as declared: name it after the relay it actually points at. For the official API-key path, see the dedicated [Google Gemini guide](../guides/google-gemini.md).