From adc3698bce0b96f26f52c213deb1ceb7044591f4 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Sat, 19 Sep 2026 00:08:03 -0700 Subject: [PATCH 1/2] fix(tui_gateway): adopt fallback_providers edits into open Desktop/TUI chats at turn start Desktop and TUI sessions keep one AIAgent across turns and _make_agent read the fallback chain exactly once. A chat opened before `hermes fallback add` therefore kept an empty _fallback_chain forever: on a Codex 429 usage_limit_reached the pool exhausted both credentials, the loop found "No fallback providers configured", and the turn ended in a provider error even though a healthy fallback was on disk. The messaging gateway already re-applies its cached agents' chain per message (GatewayRunner._apply_fallback_chain_to_agent, #60955). Run the same helper on the turn thread right beside the model/compression config syncs, fail-open, so a chain added or edited while a chat is open reaches its next turn without a new chat. Live probe (fake primary replaying the Codex 429 body, 2-entry pool, fake fallback answering 200, real AIAgent loop): before -> 5 requests to the primary, no switch, "rate-limited every one of 3 attempts"; after -> pool exhausts, fallback activates, completed=True "hello from B". Part of #95066 (atoms 1+2: chain refresh and fallback activation; the error-card "Switch provider" button remounting the live session is a separate Desktop UI change) Salvages #95139 (slim redo of the turn-start sync; the JWT account dedupe hunk was dropped: on main the pool rotation exhausts both entries and falls back in the same turn, so it is not needed to reach the fallback). Co-authored-by: BrunoBza <189763786+BrunoBza@users.noreply.github.com> --- .../test_fallback_chain_hot_reload.py | 43 +++++++++++++++++++ tui_gateway/agent_callbacks.py | 18 ++++++++ tui_gateway/prompt_turn.py | 1 + .../user-guide/features/fallback-providers.md | 1 + 4 files changed, 63 insertions(+) create mode 100644 tests/tui_gateway/test_fallback_chain_hot_reload.py diff --git a/tests/tui_gateway/test_fallback_chain_hot_reload.py b/tests/tui_gateway/test_fallback_chain_hot_reload.py new file mode 100644 index 0000000000..fd7d4e71cb --- /dev/null +++ b/tests/tui_gateway/test_fallback_chain_hot_reload.py @@ -0,0 +1,43 @@ +"""Desktop/TUI sessions must adopt a ``fallback_providers`` chain added after the chat was opened. + +Regression for #95066: ``_make_agent`` read the chain once, so a session born before ``hermes fallback +add`` kept an empty ``_fallback_chain`` forever and a Codex ``usage_limit_reached`` 429 ended in a +provider error instead of switching to the configured fallback. +""" + +from __future__ import annotations + +import inspect +from types import SimpleNamespace + +from tui_gateway import server + + +def _session(chain=None): + agent = SimpleNamespace( + _fallback_chain=list(chain or []), _fallback_model=None, _fallback_index=0, + _fallback_activated=False, _rate_limited_until=0, _unavailable_fallback_keys=set(), + ) + return {"agent": agent, "session_key": "session-95066"}, agent + + +def test_chain_added_after_open_reaches_the_live_agent(monkeypatch): + session, agent = _session() + fallback = [{"provider": "xai-oauth", "model": "grok-4.6"}] + monkeypatch.setattr(server, "_load_cfg", lambda: {"fallback_providers": fallback}) + + server._sync_agent_fallback_with_config("sid", session) + + assert agent._fallback_chain == fallback + assert agent._fallback_model == fallback[0] + assert agent._fallback_index == 0 + + +def test_turn_admission_syncs_fallback_chain_before_running(): + # Wiring guard: the sync must run on the turn thread right beside the model/compression syncs. + from tui_gateway import prompt_turn + + source = inspect.getsource(prompt_turn) + sync_idx = source.find("_sync_agent_fallback_with_config(sid, session)") + assert sync_idx > 0 + assert source.find("_sync_agent_compression_with_config(sid, session)") < sync_idx < source.find("st.agent = agent = session[\"agent\"]") diff --git a/tui_gateway/agent_callbacks.py b/tui_gateway/agent_callbacks.py index adc2085c2b..2141feb4e7 100644 --- a/tui_gateway/agent_callbacks.py +++ b/tui_gateway/agent_callbacks.py @@ -310,6 +310,24 @@ def _load_fallback_model(): return get_fallback_chain(_load_cfg()) +def _sync_agent_fallback_with_config(sid: str, session: dict) -> None: + """Adopt ``fallback_providers`` edits into the cached agent at turn start. + + Desktop/TUI chats keep one agent across turns, and ``_make_agent`` reads the chain once: a chat + opened before ``hermes fallback add`` kept an empty chain forever and a provider-quota 429 ended in + a provider error with a healthy fallback configured (#95066). Same per-turn contract the messaging + gateway applies to its cached agents; fail-open so a torn config read never blocks the turn. + """ + agent = session.get("agent") + if agent is None: + return + try: + from gateway.run import GatewayRunner + GatewayRunner._apply_fallback_chain_to_agent(agent, _load_fallback_model()) + except Exception as e: + logger.warning("fallback chain sync failed for %s: %s", sid, e) + + def _background_agent_kwargs(agent, task_id: str) -> dict: cfg = _load_cfg() diff --git a/tui_gateway/prompt_turn.py b/tui_gateway/prompt_turn.py index b06a99dade..748692f700 100644 --- a/tui_gateway/prompt_turn.py +++ b/tui_gateway/prompt_turn.py @@ -519,6 +519,7 @@ def _prepare_turn_input(sid: str, session: dict, st: _TurnRun, text: Any, images _apply_pending_model_switch(sid, session) _sync_agent_model_with_config(sid, session) _sync_agent_compression_with_config(sid, session) + _sync_agent_fallback_with_config(sid, session) # chain added after the chat opened reaches this turn _sync_bot_capabilities(sid, session) # Bot Chat: adopt Settings->Capabilities edits st.agent = agent = session["agent"] # Snapshot after the model sync: a deferred switch's history mutation belongs to this turn. diff --git a/website/docs/user-guide/features/fallback-providers.md b/website/docs/user-guide/features/fallback-providers.md index d8cf8f73af..054dbdae82 100644 --- a/website/docs/user-guide/features/fallback-providers.md +++ b/website/docs/user-guide/features/fallback-providers.md @@ -188,6 +188,7 @@ fallback_providers: |---------|-------------------| | CLI sessions | ✔ | | Messaging gateway (Telegram, Discord, etc.) | ✔ | +| Desktop app / TUI chats | ✔ (a chain added or edited while a chat is open applies from its next turn) | | Subagent delegation | ✔ (`delegation.fallback_providers` when set; otherwise only unpinned children inherit the parent chain; `[]` disables) | | Cron jobs | ✔ (cron agents inherit configured fallback providers) | | Auxiliary tasks on `provider: auto` | ✔ (try per-task fallback, then the main fallback chain before built-in aux discovery) | From 064c84751613a35781576c1b15d7d0bc82b261c1 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Sat, 19 Sep 2026 01:43:15 -0700 Subject: [PATCH 2/2] fix(tui_gateway,cli): per-turn fallback sync fails closed on a torn config; classic CLI syncs too (#95066) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review follow-up. `_sync_agent_fallback_with_config` read the chain through `_load_cfg()`, which fails open to `{}` on a torn/invalid config.yaml — `get_fallback_chain({}) == []` then WIPED the live agent's chain (the messaging gateway's `_refresh_fallback_model` keeps its last known-good chain). The sync now reads `load_user_config_effective(fail_closed=True)` and skips the apply when the read fails, so only a config that actually loaded can remove a chain. The wiring test drives `_prepare_turn_input` (turn admission) instead of string-matching the prompt_turn source, and covers removal and the active-fallback cooldown skip. Sibling surface: the classic interactive CLI holds one long-lived agent and read the chain once in `HermesCLI.__init__`; `HermesCLI.chat` now applies the same fail-closed per-turn sync (and refreshes `self._fallback_model` so a rebuilt agent starts from the fresh chain). Docs table caveat added. --- hermes_cli/cli_chat_turn_mixin.py | 16 +++++ .../test_cli_fallback_chain_hot_reload.py | 56 +++++++++++++++ .../test_fallback_chain_hot_reload.py | 70 +++++++++++++++---- tui_gateway/agent_callbacks.py | 12 +++- .../user-guide/features/fallback-providers.md | 2 +- 5 files changed, 137 insertions(+), 19 deletions(-) create mode 100644 tests/hermes_cli/test_cli_fallback_chain_hot_reload.py diff --git a/hermes_cli/cli_chat_turn_mixin.py b/hermes_cli/cli_chat_turn_mixin.py index 4aaa1a27dd..09574e84bb 100644 --- a/hermes_cli/cli_chat_turn_mixin.py +++ b/hermes_cli/cli_chat_turn_mixin.py @@ -27,6 +27,21 @@ class CLIChatTurnMixin: # process exit code (see cli._run_single_query_mode) read this instead. _last_turn_result = None + def _sync_fallback_chain_with_config(self, agent) -> None: + """Adopt ``fallback_providers`` edits made while this chat is open (#95066) — the same + per-turn, fail-closed contract as the Desktop/TUI and messaging gateways: a torn config.yaml + keeps the last known-good chain instead of reading as "chain removed".""" + from cli import logger + try: + from gateway.run import GatewayRunner + from hermes_cli.config_effective import load_user_config_effective + from hermes_cli.fallback_config import get_fallback_chain + self._fallback_model = get_fallback_chain(load_user_config_effective(fail_closed=True)) + except Exception as e: + logger.debug("fallback chain sync skipped (keeping current chain): %s", e) + return + GatewayRunner._apply_fallback_chain_to_agent(agent, self._fallback_model) + def chat(self, message, images: list = None, voice_input: bool = False) -> Optional[str]: """Run one user turn; returns the agent's response, or None on error. @@ -60,6 +75,7 @@ class CLIChatTurnMixin: agent = self.agent if agent is None: return None + self._sync_fallback_chain_with_config(agent) # chain added after this chat opened reaches this turn message = self._chat_route_images(message, images) if isinstance(message, str) and not isinstance(message, TimelineNotification): diff --git a/tests/hermes_cli/test_cli_fallback_chain_hot_reload.py b/tests/hermes_cli/test_cli_fallback_chain_hot_reload.py new file mode 100644 index 0000000000..a7e45b8759 --- /dev/null +++ b/tests/hermes_cli/test_cli_fallback_chain_hot_reload.py @@ -0,0 +1,56 @@ +"""An open classic-CLI chat adopts a ``fallback_providers`` chain added after it started (#95066). + +``HermesCLI`` holds one long-lived agent and read the chain once in ``__init__``; ``hermes fallback +add`` from another terminal never reached the open chat. The turn loop (``HermesCLI.chat``) now +re-reads the chain fail-closed, so a torn config.yaml keeps the last known-good chain. +""" + +from __future__ import annotations + +from types import SimpleNamespace + +import pytest + +import cli +from hermes_cli.config import get_config_path + +FALLBACK = [{"provider": "xai-oauth", "model": "grok-4.6"}] + + +class _StopAfterSync(Exception): + pass + + +def _chat_turn(monkeypatch, shell, config_text: str) -> None: + """Drive ``HermesCLI.chat`` past the fallback sync against ``config_text`` as the live config.yaml.""" + path = get_config_path() + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(config_text, encoding="utf-8") + monkeypatch.setattr(shell, "_ensure_runtime_credentials", lambda: True) + monkeypatch.setattr(shell, "_resolve_turn_agent_config", + lambda message: {"signature": shell._active_agent_route_signature, "model": None, "runtime": None}) + monkeypatch.setattr(shell, "_init_agent", lambda **kw: True) + + def stop(message, images): + raise _StopAfterSync() + + monkeypatch.setattr(shell, "_chat_route_images", stop) + with pytest.raises(_StopAfterSync): + shell.chat("hello") + + +def test_chat_turn_adopts_chain_added_after_the_cli_opened_and_keeps_it_on_torn_config(monkeypatch): + shell = cli.HermesCLI(compact=True, max_turns=1) + shell.agent = SimpleNamespace( + _fallback_chain=[], _fallback_model=None, _fallback_index=0, + _fallback_activated=False, _rate_limited_until=0, _unavailable_fallback_keys=set(), + ) + + _chat_turn(monkeypatch, shell, "fallback_providers:\n - provider: xai-oauth\n model: grok-4.6\n") + assert shell.agent._fallback_chain == FALLBACK + assert shell.agent._fallback_model == FALLBACK[0] + assert shell._fallback_model == FALLBACK # a rebuilt agent starts from the fresh chain too + + # Torn mid-edit write: keep the last known-good chain rather than wiping it. + _chat_turn(monkeypatch, shell, "fallback_providers: [\n - provider: {{{\n") + assert shell.agent._fallback_chain == FALLBACK diff --git a/tests/tui_gateway/test_fallback_chain_hot_reload.py b/tests/tui_gateway/test_fallback_chain_hot_reload.py index fd7d4e71cb..28a1245035 100644 --- a/tests/tui_gateway/test_fallback_chain_hot_reload.py +++ b/tests/tui_gateway/test_fallback_chain_hot_reload.py @@ -3,41 +3,81 @@ Regression for #95066: ``_make_agent`` read the chain once, so a session born before ``hermes fallback add`` kept an empty ``_fallback_chain`` forever and a Codex ``usage_limit_reached`` 429 ended in a provider error instead of switching to the configured fallback. + +Both tests drive ``_prepare_turn_input`` (turn admission, the production entry) up to the sync's +successor so the wiring — not just the helper — is pinned. """ from __future__ import annotations -import inspect +import time from types import SimpleNamespace +import pytest + from tui_gateway import server +FALLBACK = [{"provider": "xai-oauth", "model": "grok-4.6"}] + + +class _StopAfterSync(Exception): + pass + def _session(chain=None): agent = SimpleNamespace( - _fallback_chain=list(chain or []), _fallback_model=None, _fallback_index=0, + _fallback_chain=list(chain or []), _fallback_model=(chain or [None])[0], _fallback_index=0, _fallback_activated=False, _rate_limited_until=0, _unavailable_fallback_keys=set(), ) return {"agent": agent, "session_key": "session-95066"}, agent -def test_chain_added_after_open_reaches_the_live_agent(monkeypatch): +def _admit_turn(monkeypatch, tmp_path, session, config_text: str) -> None: + """Run turn admission against ``config_text`` as the live config.yaml; stop right after the sync.""" + cfg_path = tmp_path / "config.yaml" + cfg_path.write_text(config_text, encoding="utf-8") + monkeypatch.setattr(server, "_active_config_path", lambda: cfg_path) + monkeypatch.setattr(server, "_profile_runtime_scope_tokens", lambda profile_home: None) + monkeypatch.setattr(server, "_set_session_context", lambda *a, **k: []) + monkeypatch.setattr(server, "_wire_callbacks", lambda sid: None) + for name in ("_apply_pending_model_switch", "_sync_agent_model_with_config", "_sync_agent_compression_with_config"): + monkeypatch.setattr(server, name, lambda sid, session: None) + + def stop(sid, session): + raise _StopAfterSync() + + monkeypatch.setattr(server, "_sync_bot_capabilities", stop) + st = server._TurnRun(agent=None, one_turn_restore=None, terminal_callback=None, receipt_committed=False) + with pytest.raises(_StopAfterSync): + server._prepare_turn_input("sid", session, st, "hello", []) + + +def test_chain_added_after_open_reaches_the_live_agent_at_turn_admission(monkeypatch, tmp_path): session, agent = _session() - fallback = [{"provider": "xai-oauth", "model": "grok-4.6"}] - monkeypatch.setattr(server, "_load_cfg", lambda: {"fallback_providers": fallback}) - server._sync_agent_fallback_with_config("sid", session) + _admit_turn(monkeypatch, tmp_path, session, + "fallback_providers:\n - provider: xai-oauth\n model: grok-4.6\n") - assert agent._fallback_chain == fallback - assert agent._fallback_model == fallback[0] + assert agent._fallback_chain == FALLBACK + assert agent._fallback_model == FALLBACK[0] assert agent._fallback_index == 0 -def test_turn_admission_syncs_fallback_chain_before_running(): - # Wiring guard: the sync must run on the turn thread right beside the model/compression syncs. - from tui_gateway import prompt_turn +def test_torn_config_keeps_the_last_known_good_chain_but_removal_still_applies(monkeypatch, tmp_path): + session, agent = _session(FALLBACK) - source = inspect.getsource(prompt_turn) - sync_idx = source.find("_sync_agent_fallback_with_config(sid, session)") - assert sync_idx > 0 - assert source.find("_sync_agent_compression_with_config(sid, session)") < sync_idx < source.find("st.agent = agent = session[\"agent\"]") + # Torn mid-edit write: an unparsable config.yaml must NOT read as "chain removed". + _admit_turn(monkeypatch, tmp_path, session, "fallback_providers: [\n - provider: {{{\n") + assert agent._fallback_chain == FALLBACK + assert agent._fallback_model == FALLBACK[0] + + # Control: a valid config with the chain removed clears it on the next turn. + _admit_turn(monkeypatch, tmp_path, session, "model:\n provider: openai\n") + assert agent._fallback_chain == [] + assert agent._fallback_model is None + + # While a cooldown holds the agent on an activated fallback, the sync leaves the chain alone. + agent._fallback_chain, agent._fallback_activated = list(FALLBACK), True + agent._rate_limited_until = time.monotonic() + 600 + _admit_turn(monkeypatch, tmp_path, session, "model:\n provider: openai\n") + assert agent._fallback_chain == FALLBACK diff --git a/tui_gateway/agent_callbacks.py b/tui_gateway/agent_callbacks.py index 2141feb4e7..c4ae1a37a1 100644 --- a/tui_gateway/agent_callbacks.py +++ b/tui_gateway/agent_callbacks.py @@ -316,16 +316,22 @@ def _sync_agent_fallback_with_config(sid: str, session: dict) -> None: Desktop/TUI chats keep one agent across turns, and ``_make_agent`` reads the chain once: a chat opened before ``hermes fallback add`` kept an empty chain forever and a provider-quota 429 ended in a provider error with a healthy fallback configured (#95066). Same per-turn contract the messaging - gateway applies to its cached agents; fail-open so a torn config read never blocks the turn. + gateway applies to its cached agents (``GatewayRunner._refresh_fallback_model``): the config is + read fail-closed, so a torn/invalid config.yaml keeps the agent's last known-good chain instead of + ``_load_cfg()``'s fail-open ``{}`` reading as "chain removed" and wiping it. Never blocks the turn. """ agent = session.get("agent") if agent is None: return try: from gateway.run import GatewayRunner - GatewayRunner._apply_fallback_chain_to_agent(agent, _load_fallback_model()) + from hermes_cli.config_effective import load_user_config_effective + from hermes_cli.fallback_config import get_fallback_chain + chain = get_fallback_chain(load_user_config_effective(_active_config_path(), fail_closed=True)) except Exception as e: - logger.warning("fallback chain sync failed for %s: %s", sid, e) + logger.warning("fallback chain sync skipped for %s (keeping current chain): %s", sid, e) + return + GatewayRunner._apply_fallback_chain_to_agent(agent, chain) def _background_agent_kwargs(agent, task_id: str) -> dict: diff --git a/website/docs/user-guide/features/fallback-providers.md b/website/docs/user-guide/features/fallback-providers.md index 054dbdae82..f1741b77d7 100644 --- a/website/docs/user-guide/features/fallback-providers.md +++ b/website/docs/user-guide/features/fallback-providers.md @@ -186,7 +186,7 @@ fallback_providers: | Context | Fallback Supported | |---------|-------------------| -| CLI sessions | ✔ | +| CLI sessions | ✔ (a chain added or edited while a chat is open applies from its next turn) | | Messaging gateway (Telegram, Discord, etc.) | ✔ | | Desktop app / TUI chats | ✔ (a chain added or edited while a chat is open applies from its next turn) | | Subagent delegation | ✔ (`delegation.fallback_providers` when set; otherwise only unpinned children inherit the parent chain; `[]` disables) |