From 4a15e049fd7a6766cf6c44e465faff73590297a0 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 03:43:19 -0700 Subject: [PATCH] fix(cli): re-resolve reasoning after every startup model move, not just the auth fallback _resolve_cli_reasoning's invariant is that every path moving self.model re-resolves reasoning_config before the agent build. The previous commit covered only the auth fallback swap; the custom-entry runtime.model swap, the provider-default fill and the provider normalization in _ensure_runtime_credentials, plus the first-run picker re-sync in _offer_first_run_setup, still left the launch model's effort in place. Hoist a single re-resolve to the end of _ensure_runtime_credentials (guarded on the model actually changing since entry and on no explicit --reasoning), which also covers the fallback path since that is its only caller; add the same guarded call after the first-run picker re-sync, which runs before _ensure_runtime_credentials sees the move. --- hermes_cli/cli_agent_setup_mixin.py | 25 +++++++++++++------ tests/hermes_cli/test_cli_first_run_setup.py | 20 +++++++++++++++ .../test_cli_provider_resolution.py | 20 +++++++++++++++ 3 files changed, 57 insertions(+), 8 deletions(-) diff --git a/hermes_cli/cli_agent_setup_mixin.py b/hermes_cli/cli_agent_setup_mixin.py index d26399b185..f3356b1a52 100644 --- a/hermes_cli/cli_agent_setup_mixin.py +++ b/hermes_cli/cli_agent_setup_mixin.py @@ -182,6 +182,7 @@ class CLIAgentSetupMixin: from hermes_cli.runtime_provider import resolve_runtime_provider, format_runtime_provider_error _primary_exc = None runtime = None + _model_at_entry = self.model try: # target_model: the ladder's model-keyed rungs (Zen/Go api_mode, Copilot/Nous # api_mode) must see the model this CLI will actually send, not config's `default`, @@ -271,6 +272,16 @@ class CLIAgentSetupMixin: # Fixes #651. model_changed = self._normalize_model_for_provider(resolved_provider) + # Startup resolved reasoning_config for the launch model; whichever path above moved + # self.model (auth fallback, custom-entry model, provider default, normalization) leaves a + # per-model contract the lazily built agent would otherwise miss (an always-thinking model + # 400s on the primary's effort). Same chokepoint as /model, /new and --resume; an explicit + # --reasoning is the user's intent for this run and outranks the new model's config. + if self.model != _model_at_entry and getattr(self, "_explicit_reasoning_config", None) is None: + from hermes_cli.cli_model_switch_mixin import _resolve_cli_reasoning + _resolve_cli_reasoning(self) + logger.info("Model moved to %s: reasoning_config resolved: %s", self.model, self.reasoning_config) + # AIAgent/OpenAI client holds auth at init, so rebuild on key/routing/model change. if (credentials_changed or routing_changed or model_changed) and self.agent is not None: self.agent = None @@ -326,14 +337,7 @@ class CLIAgentSetupMixin: platform="cli") self.requested_provider = _fb_provider self.model = _fb_model - # Startup resolved reasoning_config for the launch model; the fallback model has its - # own per-model contract (an always-thinking model 400s on the primary's effort). - # Same chokepoint as /model, /new and --resume; an explicit --reasoning is the - # user's intent for this run and outranks the fallback model's config. - if getattr(self, "_explicit_reasoning_config", None) is None: - from hermes_cli.cli_model_switch_mixin import _resolve_cli_reasoning - _resolve_cli_reasoning(self) - logger.info("Fallback %s: reasoning_config resolved: %s", self.model, self.reasoning_config) + # reasoning_config follows the swap in _ensure_runtime_credentials (the only caller). return runtime except Exception: continue @@ -400,6 +404,11 @@ class CLIAgentSetupMixin: self.requested_provider = (_model_cfg.get("provider") or "").strip() or self.requested_provider _new_model = (_model_cfg.get("default") or _model_cfg.get("model") or "").strip() self.model = _new_model or self.model + # The picker's model has its own per-model reasoning contract (see + # _resolve_cli_reasoning); an explicit --reasoning stays the user's intent. + if _new_model and getattr(self, "_explicit_reasoning_config", None) is None: + from hermes_cli.cli_model_switch_mixin import _resolve_cli_reasoning + _resolve_cli_reasoning(self) except Exception as exc: logger.debug("first-run config re-sync failed: %s", exc) # Force credential re-resolution + agent rebuild on next use. diff --git a/tests/hermes_cli/test_cli_first_run_setup.py b/tests/hermes_cli/test_cli_first_run_setup.py index 625e4382d4..5882a36388 100644 --- a/tests/hermes_cli/test_cli_first_run_setup.py +++ b/tests/hermes_cli/test_cli_first_run_setup.py @@ -190,6 +190,26 @@ def test_offer_first_run_setup_routes_into_shared_picker(monkeypatch): assert shell.agent is None +def test_offer_first_run_setup_re_resolves_reasoning_for_picked_model(monkeypatch): + """The picker moves self.model; the CLI-level reasoning_config must follow it before the + lazily built agent inherits the launch model's effort.""" + cli = _import_cli() + monkeypatch.setitem(cli.CLI_CONFIG, "agent", { + **cli.CLI_CONFIG.get("agent", {}), "reasoning_effort": "medium", + "reasoning_overrides": {"hermes-4-405b": "high"}}) + shell = _make_shell(cli, monkeypatch) + assert shell.reasoning_config["effort"] == "medium" + monkeypatch.setattr("hermes_cli.main.select_provider_and_model", lambda: None) + monkeypatch.setattr("builtins.input", lambda *a, **k: "y") + monkeypatch.setattr("hermes_cli.config.load_config", + lambda: {"model": {"provider": "nous", "default": "hermes-4-405b"}}) + monkeypatch.setattr(shell, "_runtime_credentials_ready", lambda: True) + + assert shell._offer_first_run_setup() is True + assert shell.model == "hermes-4-405b" + assert shell.reasoning_config["effort"] == "high" + + def test_offer_first_run_setup_declined(monkeypatch): cli = _import_cli() shell = _make_shell(cli, monkeypatch) diff --git a/tests/hermes_cli/test_cli_provider_resolution.py b/tests/hermes_cli/test_cli_provider_resolution.py index 3c6fd435a6..c0baed7bcf 100644 --- a/tests/hermes_cli/test_cli_provider_resolution.py +++ b/tests/hermes_cli/test_cli_provider_resolution.py @@ -564,6 +564,26 @@ def test_startup_fallback_re_resolves_reasoning_for_the_fallback_model(monkeypat assert shell.reasoning_config["effort"] == expected_effort +def test_custom_entry_model_swap_re_resolves_reasoning(monkeypatch): + """`hermes chat --model `: the runtime's explicit `model` replaces the + slug, so the CLI-level reasoning_config must follow to that model's per-model override.""" + cli = _import_cli() + monkeypatch.setattr(cli, "_cprint", lambda *a, **k: None) + monkeypatch.setitem(cli.CLI_CONFIG, "agent", { + **cli.CLI_CONFIG.get("agent", {}), "reasoning_effort": "medium", + "reasoning_overrides": {"real-model": "high"}}) + monkeypatch.setattr( + "hermes_cli.runtime_provider.resolve_runtime_provider", + lambda **kw: {"provider": "custom", "name": "my-lan", "model": "real-model", "api_mode": "chat_completions", + "base_url": "http://10.0.0.7:11434/v1", "api_key": "sk-lan", "source": "custom"}) + shell = cli.HermesCLI(model="my-lan", compact=True, max_turns=1) + assert shell.reasoning_config["effort"] == "medium" + + assert shell._ensure_runtime_credentials() is True + assert shell.model == "real-model" + assert shell.reasoning_config["effort"] == "high" + +