From ff6b9f722d440eaf2498d23c0a347fce09171f38 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 00:49:01 -0700 Subject: [PATCH 01/65] fix(gateway): a non-interactive Windows `gateway start` never installs login auto-start MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `start()` asked "Install it now so the gateway starts on login?" with default Yes, then called `install()` which asked two more default-Yes questions; on a redirected stdin (`< /dev/null`, scripts, agent tool calls) all three answered themselves and a bare `start` wrote a Startup-folder login item, spawned the gateway, and start() spawned it a second time (#113977). - The persistence question is answered only explicitly: `HERMES_GATEWAY_INSTALL_START_ON_LOGIN` (which start() never consulted before) or a prompt on a real TTY. Without a TTY, or under `HERMES_NONINTERACTIVE=1`, the answer is No: the gateway starts, nothing persistent is written, and the explicit `hermes gateway install` command is printed. - Declining on a TTY also starts the gateway — the command is `start`; only persistence was declined. - A Yes hands `start_now=True, start_on_login=True` to install(), which spawns and reports (including the UAC hand-off), so start() no longer double-spawns and no longer prints the false "Gateway install did not complete in this process" after an install that was skipped by intent. - `install --no-start-on-login` without `--start-now` now hints `hermes gateway run` instead of the `hermes gateway start` that used to offer the very auto-start just declined. Supersedes #113981 (@KoNit-K), which flipped the first prompt's non-TTY default but left the env override, the hint text and the false warning in place. --- hermes_cli/gateway_windows.py | 27 ++++++++----- tests/hermes_cli/test_gateway_windows.py | 47 +++++++++++++++++++++++ website/docs/user-guide/windows-native.md | 4 +- 3 files changed, 67 insertions(+), 11 deletions(-) diff --git a/hermes_cli/gateway_windows.py b/hermes_cli/gateway_windows.py index 56eae9a9fd..9bafd6f462 100644 --- a/hermes_cli/gateway_windows.py +++ b/hermes_cli/gateway_windows.py @@ -785,7 +785,7 @@ def install( _start_or_report_running() else: print("ℹ Gateway not started and no auto-start service installed.") - print(" Run later with: hermes gateway start") + print(" Run in the foreground later with: hermes gateway run") return task_name = get_task_name() @@ -1493,17 +1493,24 @@ def start() -> None: return if not is_task_registered() and not is_startup_entry_installed(): - from hermes_cli.setup import prompt_yes_no + # Login persistence is a lasting system change: a bare ``start`` installs it only on an explicit + # answer — the HERMES_GATEWAY_INSTALL_START_ON_LOGIN override or a real TTY prompt — never on a + # non-TTY default (#113977). Declining still starts the gateway; the command is ``start``. + start_on_login = _install_choice_from_env("HERMES_GATEWAY_INSTALL_START_ON_LOGIN") + if start_on_login is None: + from hermes_cli.setup import is_interactive_stdin, is_noninteractive, prompt_yes_no - print("✗ Gateway service is not installed") - if not prompt_yes_no(" Install it now so the gateway starts on login?", True): - print(" Run: hermes gateway install") - return - install(force=False) - if not is_task_registered() and not is_startup_entry_installed(): - print("⚠ Gateway install did not complete in this process.") - print(" If a UAC prompt opened, approve it, then run: hermes gateway start") + print("✗ Gateway service is not installed") + if is_noninteractive() or not is_interactive_stdin(): + start_on_login = False + else: + start_on_login = prompt_yes_no(" Install it now so the gateway starts on login?", True) + if start_on_login: + # install() starts the gateway itself (start_now) and reports the outcome — including a UAC + # hand-off to an elevated child — so there is nothing left to spawn or to warn about here. + install(force=False, start_now=True, start_on_login=True) return + print("ℹ Login auto-start not installed; add it later with: hermes gateway install") elif is_task_registered(): reconcile_scheduled_task(get_task_name()) # like systemd's regenerate-on-stale before a start diff --git a/tests/hermes_cli/test_gateway_windows.py b/tests/hermes_cli/test_gateway_windows.py index b0e1117742..a997492149 100644 --- a/tests/hermes_cli/test_gateway_windows.py +++ b/tests/hermes_cli/test_gateway_windows.py @@ -481,6 +481,53 @@ def test_reconcile_scheduled_task_reregisters_only_on_drift(monkeypatch, tmp_pat assert not any(c[0] in ("/Delete", "/Create") for c in calls) +def _arrange_uninstalled_start(monkeypatch): + """start() with no Scheduled Task / Startup entry; returns (install_calls, spawn_count).""" + installs, spawns = [], [] + monkeypatch.delenv("HERMES_GATEWAY_INSTALL_START_ON_LOGIN", raising=False) + monkeypatch.delenv("HERMES_NONINTERACTIVE", raising=False) + monkeypatch.setattr(gateway_windows, "_assert_windows", lambda: None) + monkeypatch.setattr(gateway_windows, "_print_start_attestation_warning", lambda: None) + monkeypatch.setattr(gateway_windows, "_gateway_pids", lambda: []) + monkeypatch.setattr(gateway_windows, "is_task_registered", lambda: False) + monkeypatch.setattr(gateway_windows, "is_startup_entry_installed", lambda: False) + monkeypatch.setattr(gateway_windows, "install", lambda **kwargs: installs.append(kwargs)) + monkeypatch.setattr(gateway_windows, "_spawn_detached", lambda: spawns.append(1) or 4242) + monkeypatch.setattr(gateway_windows, "_report_gateway_start", lambda via: None) + return installs, spawns + + +def test_start_without_tty_starts_the_gateway_but_never_installs_login_persistence(monkeypatch, capsys): + """`hermes gateway start < /dev/null` must not answer the persistence question with a default Yes + (#113977); it starts the gateway once and points at the explicit install command.""" + installs, spawns = _arrange_uninstalled_start(monkeypatch) + monkeypatch.setattr(setup, "is_interactive_stdin", lambda: False) + monkeypatch.setattr(setup, "prompt_yes_no", lambda *a, **k: pytest.fail("no prompt without a TTY")) + + gateway_windows.start() + + assert installs == [] and spawns == [1] + out = capsys.readouterr().out + assert "hermes gateway install" in out and "did not complete" not in out + + +def test_start_on_tty_hands_both_answers_to_install_and_honours_the_env_opt_out(monkeypatch): + """Yes → one install() carrying start_now+start_on_login (install spawns; start() must not spawn + again). HERMES_GATEWAY_INSTALL_START_ON_LOGIN=0 → no question, no install, a plain start.""" + installs, spawns = _arrange_uninstalled_start(monkeypatch) + monkeypatch.setattr(setup, "is_interactive_stdin", lambda: True) + monkeypatch.setattr(setup, "prompt_yes_no", lambda *a, **k: True) + + gateway_windows.start() + assert installs == [{"force": False, "start_now": True, "start_on_login": True}] and spawns == [] + + installs.clear() + monkeypatch.setenv("HERMES_GATEWAY_INSTALL_START_ON_LOGIN", "0") + monkeypatch.setattr(setup, "prompt_yes_no", lambda *a, **k: pytest.fail("env override must skip the prompt")) + gateway_windows.start() + assert installs == [] and spawns == [1] + + diff --git a/website/docs/user-guide/windows-native.md b/website/docs/user-guide/windows-native.md index 42d26afea1..e895e5822b 100644 --- a/website/docs/user-guide/windows-native.md +++ b/website/docs/user-guide/windows-native.md @@ -186,7 +186,7 @@ Flags used when spawning: `DETACHED_PROCESS | CREATE_NEW_PROCESS_GROUP | CREATE_ ```powershell hermes gateway status # Merged view: schtasks + Startup folder + running PID -hermes gateway start # Starts the scheduled task now +hermes gateway start # Starts the gateway in the background (asks about login auto-start only on a TTY when nothing is installed) hermes gateway stop # Writes the planned-stop marker, waits for the gateway to drain (≤ agent.restart_drain_timeout, capped at 30 s), then force-kills only if it is still alive hermes gateway restart # Same drain-first stop, then a fresh start hermes gateway uninstall # Removes schtasks entry, Startup shortcut, pid file @@ -194,6 +194,8 @@ hermes gateway uninstall # Removes schtasks entry, Startup shortcut, pid file `hermes gateway status` is idempotent — call it a thousand times in a row and it will never accidentally kill the gateway. (Pre-PR #21561 it silently did, via `os.kill(pid, 0)` colliding with `CTRL_C_EVENT` at the C level — see "process management internals" below if you care about the story.) +Login auto-start is only ever installed on an explicit answer: `hermes gateway install`, a `Y` on a real terminal, or `HERMES_GATEWAY_INSTALL_START_ON_LOGIN=1`. A scripted or piped `hermes gateway start` (no TTY, or `HERMES_NONINTERACTIVE=1`) starts the gateway without touching the Scheduled Task or the Startup folder; set `HERMES_GATEWAY_INSTALL_START_ON_LOGIN=0` to skip the question on a terminal too. + ### Why not a Windows Service? Services require admin rights to install and tie the gateway's lifecycle to machine boot, not user login. The typical Hermes user wants: log in → gateway available, log out → gateway gone. Scheduled Tasks do exactly that without elevation. If you genuinely want a service, use `nssm` or `sc create` manually — but you probably don't. From be0a21445236588baf4bec96758c1588a25fd192 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 07:39:08 -0700 Subject: [PATCH 02/65] fix(gateway): a NUL-stdin Windows `gateway start` is non-interactive too MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Windows CRT reports isatty()==True for every character device, the NUL device included, so `hermes gateway start < NUL` (stdin=DEVNULL) still took the TTY branch: prompt_yes_no() hit EOF, defaulted to Yes and registered a Scheduled Task — the silent install #113977 describes. Confirm isatty with GetConsoleMode(STD_INPUT_HANDLE); a handle no console accepts is treated like a pipe. Pipes and HERMES_NONINTERACTIVE keep their behaviour. --- hermes_cli/gateway_windows.py | 20 +++++++++++++++++++- tests/hermes_cli/test_gateway_windows.py | 23 +++++++++++++++++++++++ 2 files changed, 42 insertions(+), 1 deletion(-) diff --git a/hermes_cli/gateway_windows.py b/hermes_cli/gateway_windows.py index 9bafd6f462..7a34788147 100644 --- a/hermes_cli/gateway_windows.py +++ b/hermes_cli/gateway_windows.py @@ -683,6 +683,22 @@ def _spawn_detached(script_path: Path | None = None, home: Path | None = None) - return proc.pid +def _stdin_is_interactive(*, isatty: bool, console_mode_ok: bool | None) -> bool: + """A human can answer a prompt only on a real console. The Windows CRT reports isatty()==True for + every character device — the NUL device included (`hermes gateway start < NUL`, stdin=DEVNULL) — so + isatty must be confirmed by GetConsoleMode accepting the handle (#113977). ``console_mode_ok`` is + None where that fact does not exist (not Windows) and isatty alone decides.""" + return isatty and console_mode_ok is not False + + +def _stdin_console_mode_ok() -> bool | None: + if sys.platform != "win32": + return None + kernel32 = ctypes.windll.kernel32 + handle = kernel32.GetStdHandle(-10) # STD_INPUT_HANDLE + return bool(kernel32.GetConsoleMode(handle, ctypes.byref(ctypes.c_ulong()))) + + def _install_choice_from_env(name: str) -> bool | None: raw = os.environ.get(name) if raw is None: @@ -1501,7 +1517,9 @@ def start() -> None: from hermes_cli.setup import is_interactive_stdin, is_noninteractive, prompt_yes_no print("✗ Gateway service is not installed") - if is_noninteractive() or not is_interactive_stdin(): + if is_noninteractive() or not _stdin_is_interactive( + isatty=is_interactive_stdin(), console_mode_ok=_stdin_console_mode_ok() + ): start_on_login = False else: start_on_login = prompt_yes_no(" Install it now so the gateway starts on login?", True) diff --git a/tests/hermes_cli/test_gateway_windows.py b/tests/hermes_cli/test_gateway_windows.py index a997492149..d4a62e1931 100644 --- a/tests/hermes_cli/test_gateway_windows.py +++ b/tests/hermes_cli/test_gateway_windows.py @@ -494,9 +494,32 @@ def _arrange_uninstalled_start(monkeypatch): monkeypatch.setattr(gateway_windows, "install", lambda **kwargs: installs.append(kwargs)) monkeypatch.setattr(gateway_windows, "_spawn_detached", lambda: spawns.append(1) or 4242) monkeypatch.setattr(gateway_windows, "_report_gateway_start", lambda via: None) + monkeypatch.setattr(gateway_windows, "_stdin_console_mode_ok", lambda: True) return installs, spawns +def test_stdin_interactive_only_when_isatty_and_a_console_answers_get_console_mode(): + """Windows CRT isatty() is True for the NUL device (`hermes gateway start < NUL`, stdin=DEVNULL), so + isatty alone must not open the prompt; off Windows (no console-mode fact) isatty decides (#113977).""" + assert gateway_windows._stdin_is_interactive(isatty=True, console_mode_ok=False) is False # NUL + assert gateway_windows._stdin_is_interactive(isatty=True, console_mode_ok=True) is True # console + assert gateway_windows._stdin_is_interactive(isatty=False, console_mode_ok=True) is False # pipe + assert gateway_windows._stdin_is_interactive(isatty=True, console_mode_ok=None) is True # POSIX tty + + +def test_start_with_nul_stdin_starts_the_gateway_but_never_installs_login_persistence(monkeypatch, capsys): + """isatty says TTY, GetConsoleMode says no console: `< NUL` gets the same treatment as a pipe.""" + installs, spawns = _arrange_uninstalled_start(monkeypatch) + monkeypatch.setattr(setup, "is_interactive_stdin", lambda: True) + monkeypatch.setattr(gateway_windows, "_stdin_console_mode_ok", lambda: False) + monkeypatch.setattr(setup, "prompt_yes_no", lambda *a, **k: pytest.fail("no prompt on a NUL stdin")) + + gateway_windows.start() + + assert installs == [] and spawns == [1] + assert "hermes gateway install" in capsys.readouterr().out + + def test_start_without_tty_starts_the_gateway_but_never_installs_login_persistence(monkeypatch, capsys): """`hermes gateway start < /dev/null` must not answer the persistence question with a default Yes (#113977); it starts the gateway once and points at the explicit install command.""" From 402743e044f07ea434e9948f0525ff34e916c79c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Chlo=C3=A9?= Date: Thu, 17 Sep 2026 20:03:04 +0200 Subject: [PATCH 03/65] plugin-catalog: add pstack (community) --- plugin-catalog/pstack.yaml | 15 +++++++++++++++ 1 file changed, 15 insertions(+) create mode 100644 plugin-catalog/pstack.yaml diff --git a/plugin-catalog/pstack.yaml b/plugin-catalog/pstack.yaml new file mode 100644 index 0000000000..b8c5480f55 --- /dev/null +++ b/plugin-catalog/pstack.yaml @@ -0,0 +1,15 @@ +name: pstack +repo: https://github.com/Zoeille/pstack +sha: 932d8174eba6581d3a117c2d2b7796e7eb27953e +description: "Engineering rigor stack: poteto-mode orchestrator with 7 playbooks (new-feature, bugfix, refactor, testing, one-shot, migrate, explore), plus how, architect, interrogate, swarm, and unslop skills." +maintainer: Zoeille +tier: community +category: tools +version: "0.1.0" +docs_url: https://github.com/Zoeille/pstack +platforms: [] +capabilities: + provides_tools: [] + provides_hooks: [] + provides_middleware: [] + requires_env: [] From 60fd4e872316ff805838cd044810f88ad91b1331 Mon Sep 17 00:00:00 2001 From: teknium Date: Fri, 18 Sep 2026 14:26:43 -0700 Subject: [PATCH 04/65] chore: map contributor email for @Zoeille --- contributors/emails/Zoeille@users.noreply.github.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/Zoeille@users.noreply.github.com diff --git a/contributors/emails/Zoeille@users.noreply.github.com b/contributors/emails/Zoeille@users.noreply.github.com new file mode 100644 index 0000000000..ec7ac6dcaa --- /dev/null +++ b/contributors/emails/Zoeille@users.noreply.github.com @@ -0,0 +1 @@ +Zoeille From a9dbbe9fccf2d6c6edaff672dacf0b61c79c26b5 Mon Sep 17 00:00:00 2001 From: Adolanium <94890352+Adolanium@users.noreply.github.com> Date: Wed, 16 Sep 2026 08:12:03 +0300 Subject: [PATCH 05/65] feat(catalog): add OpenAlex research tools --- plugin-catalog/openalex.yaml | 32 +++++++++++++++----------------- 1 file changed, 15 insertions(+), 17 deletions(-) diff --git a/plugin-catalog/openalex.yaml b/plugin-catalog/openalex.yaml index abdcb45958..90df1f0802 100644 --- a/plugin-catalog/openalex.yaml +++ b/plugin-catalog/openalex.yaml @@ -1,28 +1,26 @@ name: openalex repo: https://github.com/Adolanium/hermes-plugin-openalex -sha: 04e673e34bfd09ea9cf20626373ce48df096f813 -description: OpenAlex for Hermes. Search 250M scholarly works, authors, journals and institutions, with - a USD budget guard, per-call cost prediction, and response shaping that keeps a 2.8 MB paper inside - the context window. +sha: 47aafae87256ef634f32c18a69bf35d39c2c5682 +description: Scholarly search, citation traversal, grouped counts, and fulltext access through OpenAlex with session spending limits. maintainer: Adolanium tier: community category: tools docs_url: https://github.com/Adolanium/hermes-plugin-openalex#readme capabilities: provides_tools: - - openalex_account - - openalex_classify - - openalex_count - - openalex_fields - - openalex_fulltext - - openalex_get - - openalex_harvest - - openalex_related - - openalex_resolve - - openalex_search + - openalex_resolve + - openalex_get + - openalex_count + - openalex_search + - openalex_related + - openalex_account + - openalex_fields + - openalex_fulltext + - openalex_classify + - openalex_harvest provides_hooks: - - on_session_reset - - on_session_start + - on_session_start + - on_session_reset provides_middleware: [] requires_env: - - OPENALEX_API_KEY + - OPENALEX_API_KEY From 0b78a26903fb9d939849234b056b0f47fafdaf05 Mon Sep 17 00:00:00 2001 From: Dafka <322898297+dafka007@users.noreply.github.com> Date: Wed, 16 Sep 2026 14:10:06 +0300 Subject: [PATCH 06/65] feat(plugin-catalog): add hermes-security-audit --- plugin-catalog/hermes-security-audit.yaml | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) create mode 100644 plugin-catalog/hermes-security-audit.yaml diff --git a/plugin-catalog/hermes-security-audit.yaml b/plugin-catalog/hermes-security-audit.yaml new file mode 100644 index 0000000000..561f6c033f --- /dev/null +++ b/plugin-catalog/hermes-security-audit.yaml @@ -0,0 +1,16 @@ +name: hermes-security-audit +repo: https://github.com/dafka007/hermes-security-audit +sha: 60a7661bbe2f255c507d620275c59782cabe0335 +description: Approval-gated local security audit for Hermes Agent using Gitleaks, OSV-Scanner, and Semgrep CE. +maintainer: dafka007 +tier: community +category: tools +docs_url: https://github.com/dafka007/hermes-security-audit#readme +platforms: [] +capabilities: + provides_tools: + - security_audit + provides_hooks: + - pre_tool_call + provides_middleware: [] + requires_env: [] From 7317669fde3a47732d46b269cfd138856386e752 Mon Sep 17 00:00:00 2001 From: drkpxl Date: Wed, 16 Sep 2026 13:27:34 -0600 Subject: [PATCH 07/65] Add drkpxl-skills to plugin catalog MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Community skills collection by the repo owner (drkpxl). Anchor skill: morning-briefing — a personal daily newspaper that gathers weather, calendar, AQI, curated 36-hour news (Reddit RSS + X + web search with per-story QR codes), and newsletter digests, renders a single 8.5x11 newsprint page, and prints it every morning. Also bundles: Tiny Air (US AQI via MCP), Bro, Copywriting, Delegate to pi, Prototype, Research, Requesting Code Review. SKILL.md follows the HARDLINE authoring standards (validated: 54-char description, modern section order, native tool references). SHA pinned at d81a9b5ee427a378839828d4a8eab3925468b7ce. --- plugin-catalog/drkpxl-skills.yaml | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) create mode 100644 plugin-catalog/drkpxl-skills.yaml diff --git a/plugin-catalog/drkpxl-skills.yaml b/plugin-catalog/drkpxl-skills.yaml new file mode 100644 index 0000000000..2a190f91bb --- /dev/null +++ b/plugin-catalog/drkpxl-skills.yaml @@ -0,0 +1,18 @@ +name: drkpxl-skills +repo: https://github.com/drkpxl/drkpxl-skills +sha: d81a9b5ee427a378839828d4a8eab3925468b7ce +description: 'Personal daily newspaper skill for Hermes: weather, calendar, air quality, + curated 36-hour news from Reddit RSS + X + web search with per-story QR codes, + and newsletter digests, rendered to a single 8.5x11 newsprint page and printed + every morning. Plus six more installable skills: Tiny Air (US air quality MCP), + Bro (jargon translator), Copywriting, Delegate to pi, Prototype, Research, + and Requesting Code Review.' +maintainer: drkpxl +tier: community +docs_url: https://github.com/drkpxl/drkpxl-skills#readme +platforms: [macos, linux] +capabilities: + provides_tools: [] + provides_hooks: [] + provides_middleware: [] + requires_env: [] \ No newline at end of file From dd254399e48579d752e37af4c81f0f9cfa23fe40 Mon Sep 17 00:00:00 2001 From: drkpxl Date: Wed, 16 Sep 2026 13:53:16 -0600 Subject: [PATCH 08/65] Scope catalog entry to the dedicated morning-briefing repo Split out from drkpxl/drkpxl-skills so the entry matches the product: one skill, one repo, name on the tin. The new repo contains only the morning-briefing skill. Entry name, repo URL, SHA pin, description, and category all updated. --- plugin-catalog/drkpxl-skills.yaml | 18 ------------------ plugin-catalog/morning-briefing.yaml | 17 +++++++++++++++++ 2 files changed, 17 insertions(+), 18 deletions(-) delete mode 100644 plugin-catalog/drkpxl-skills.yaml create mode 100644 plugin-catalog/morning-briefing.yaml diff --git a/plugin-catalog/drkpxl-skills.yaml b/plugin-catalog/drkpxl-skills.yaml deleted file mode 100644 index 2a190f91bb..0000000000 --- a/plugin-catalog/drkpxl-skills.yaml +++ /dev/null @@ -1,18 +0,0 @@ -name: drkpxl-skills -repo: https://github.com/drkpxl/drkpxl-skills -sha: d81a9b5ee427a378839828d4a8eab3925468b7ce -description: 'Personal daily newspaper skill for Hermes: weather, calendar, air quality, - curated 36-hour news from Reddit RSS + X + web search with per-story QR codes, - and newsletter digests, rendered to a single 8.5x11 newsprint page and printed - every morning. Plus six more installable skills: Tiny Air (US air quality MCP), - Bro (jargon translator), Copywriting, Delegate to pi, Prototype, Research, - and Requesting Code Review.' -maintainer: drkpxl -tier: community -docs_url: https://github.com/drkpxl/drkpxl-skills#readme -platforms: [macos, linux] -capabilities: - provides_tools: [] - provides_hooks: [] - provides_middleware: [] - requires_env: [] \ No newline at end of file diff --git a/plugin-catalog/morning-briefing.yaml b/plugin-catalog/morning-briefing.yaml new file mode 100644 index 0000000000..4ac4871887 --- /dev/null +++ b/plugin-catalog/morning-briefing.yaml @@ -0,0 +1,17 @@ +name: morning-briefing +repo: https://github.com/drkpxl/morning-briefing +sha: 8198c77f8adcc78307e7fdb852f20e91ec46c5a2 +description: 'Personal daily newspaper for Hermes: gathers weather, calendar, AQI, curated + 36-hour news (timestamped Reddit RSS + X + web search, per-story QR codes), and + newsletter digests, renders one 8.5x11 newsprint page, and prints it every morning + via a cron job.' +maintainer: drkpxl +tier: community +category: automation +docs_url: https://github.com/drkpxl/morning-briefing#readme +platforms: [macos, linux] +capabilities: + provides_tools: [] + provides_hooks: [] + provides_middleware: [] + requires_env: [] \ No newline at end of file From 71950e61bb7cd7458d412a5be869f92091127ac8 Mon Sep 17 00:00:00 2001 From: drkpxl Date: Fri, 18 Sep 2026 10:33:43 -0600 Subject: [PATCH 09/65] Bump SHA: add portable plugin.json manifest to morning-briefing repo The pinned tree now carries an Agent Plugins v1 plugin.json at the repo root, so 'hermes plugins validate' passes and the catalog installer finds its manifest. Security scan clean (no cautions). New pin: 347ce44a1e71173285b2943a7a97bbdece8327b2 --- plugin-catalog/morning-briefing.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/plugin-catalog/morning-briefing.yaml b/plugin-catalog/morning-briefing.yaml index 4ac4871887..a1998abb90 100644 --- a/plugin-catalog/morning-briefing.yaml +++ b/plugin-catalog/morning-briefing.yaml @@ -1,6 +1,6 @@ name: morning-briefing repo: https://github.com/drkpxl/morning-briefing -sha: 8198c77f8adcc78307e7fdb852f20e91ec46c5a2 +sha: 347ce44a1e71173285b2943a7a97bbdece8327b2 description: 'Personal daily newspaper for Hermes: gathers weather, calendar, AQI, curated 36-hour news (timestamped Reddit RSS + X + web search, per-story QR codes), and newsletter digests, renders one 8.5x11 newsprint page, and prints it every morning From aac41a3d3d6a91fc8d64f2bbdc616e66729c7e9b Mon Sep 17 00:00:00 2001 From: drkpxl Date: Fri, 18 Sep 2026 15:38:14 -0600 Subject: [PATCH 10/65] chore: map contributor email --- contributors/emails/drkpxl@users.noreply.github.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/drkpxl@users.noreply.github.com diff --git a/contributors/emails/drkpxl@users.noreply.github.com b/contributors/emails/drkpxl@users.noreply.github.com new file mode 100644 index 0000000000..59387bb2af --- /dev/null +++ b/contributors/emails/drkpxl@users.noreply.github.com @@ -0,0 +1 @@ +drkpxl From c24358032fbf6a54501286aea4a0986890632105 Mon Sep 17 00:00:00 2001 From: Egor Loktev Date: Wed, 16 Sep 2026 01:31:06 +0300 Subject: [PATCH 11/65] feat(catalog): add pinned artifact-relay plugin --- plugin-catalog/artifact-relay.yaml | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 plugin-catalog/artifact-relay.yaml diff --git a/plugin-catalog/artifact-relay.yaml b/plugin-catalog/artifact-relay.yaml new file mode 100644 index 0000000000..1063490bf1 --- /dev/null +++ b/plugin-catalog/artifact-relay.yaml @@ -0,0 +1,14 @@ +name: artifact-relay +repo: https://github.com/eloktev/hermes-artifact-relay +sha: c91fa9229121dd616a275a45447301eacca45e6f +description: Publish and read private Markdown or HTML reports through your Artifact Relay server. +maintainer: eloktev +tier: community +category: tools +docs_url: https://github.com/eloktev/hermes-artifact-relay#readme +platforms: [linux, macos, windows] +capabilities: + provides_tools: [artifact_read, artifact_publish] + provides_hooks: [] + provides_middleware: [] + requires_env: [ARTIFACT_RELAY_API_TOKEN] From b73a290de4cfda8f3e5de97d36b4daf80c11bd98 Mon Sep 17 00:00:00 2001 From: teknium Date: Fri, 18 Sep 2026 14:43:07 -0700 Subject: [PATCH 12/65] chore: map contributor email for @eloktev --- contributors/emails/eloktev@users.noreply.github.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/eloktev@users.noreply.github.com diff --git a/contributors/emails/eloktev@users.noreply.github.com b/contributors/emails/eloktev@users.noreply.github.com new file mode 100644 index 0000000000..8885873168 --- /dev/null +++ b/contributors/emails/eloktev@users.noreply.github.com @@ -0,0 +1 @@ +eloktev From 6c7a0e59c6c6022ece7e68dc3bac73b069ba6cb8 Mon Sep 17 00:00:00 2001 From: devin Date: Thu, 17 Sep 2026 11:58:09 +0800 Subject: [PATCH 13/65] feat(catalog): add Search1API plugin MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Registers superagents-lab/hermes-search1api (v0.1.0) — five tools backed by the Search1API API: search1api_search, search1api_news, search1api_crawl, search1api_sitemap, search1api_trending. Vendored official Python SDK; only third-party dependency is httpx (already a Hermes core dependency). hermes plugins validate (main @ 6005aa1): all checks pass, security scan safe; declared tools match registrations. Submitted by the plugin repo owner (superagents-lab). --- plugin-catalog/search1api.yaml | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) create mode 100644 plugin-catalog/search1api.yaml diff --git a/plugin-catalog/search1api.yaml b/plugin-catalog/search1api.yaml new file mode 100644 index 0000000000..db01202012 --- /dev/null +++ b/plugin-catalog/search1api.yaml @@ -0,0 +1,19 @@ +name: search1api +repo: https://github.com/superagents-lab/hermes-search1api +sha: 14c861746c50ef5ab32edec8b5c9ad0815f40cec +description: Live web search, news, page crawl, sitemap, and trending via Search1API +maintainer: superagents-lab +tier: community +category: web +docs_url: https://github.com/superagents-lab/hermes-search1api#readme +version: "0.1.0" +capabilities: + provides_tools: + - search1api_search + - search1api_news + - search1api_crawl + - search1api_sitemap + - search1api_trending + provides_hooks: [] + provides_middleware: [] + requires_env: [] From 2d0ea1bc2e006d4e210b50157e0f002219df4e91 Mon Sep 17 00:00:00 2001 From: Edward Irby Date: Wed, 16 Sep 2026 20:29:50 -0700 Subject: [PATCH 14/65] plugin-catalog: add 'you' (You.com skills + MCP, community) Agent Plugins v1 package (plugin.json + mcp.json + skills/) from youdotcom-oss/agent-skills, pinned at 08f3d80 (tag v2026.09.17-08f3d80). Owner-submitted. --- plugin-catalog/you.yaml | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) create mode 100644 plugin-catalog/you.yaml diff --git a/plugin-catalog/you.yaml b/plugin-catalog/you.yaml new file mode 100644 index 0000000000..e009e2968c --- /dev/null +++ b/plugin-catalog/you.yaml @@ -0,0 +1,20 @@ +name: you +repo: https://github.com/youdotcom-oss/agent-skills +sha: 08f3d80e7ca9518ccc64b38b365a78cc4a762691 +description: 'You.com skills for live web search, URL content extraction, cited research, finance + research, and integration discovery via the You.com MCP servers (api.you.com). Portable Agent + Plugins v1 package (mcp.json + skills/, five skills bundled); auth via YDC_API_KEY, OAuth, or + MPP/x402 — the free profile works keyless.' +maintainer: youdotcom-oss +tier: community +category: web +requires_hermes: ">=0.20" +docs_url: https://github.com/youdotcom-oss/agent-skills#readme +version: "0.6.0" +image: https://raw.githubusercontent.com/youdotcom-oss/agent-skills/08f3d80e7ca9518ccc64b38b365a78cc4a762691/assets/logo.png +platforms: [] +capabilities: + provides_tools: [] + provides_hooks: [] + provides_middleware: [] + requires_env: [] From fde394c426e7cce7e4637bd3a47bd45a30a650f6 Mon Sep 17 00:00:00 2001 From: rajitkhanna Date: Tue, 15 Sep 2026 14:13:36 -0700 Subject: [PATCH 15/65] feat(plugin-catalog): add Prism provider --- plugin-catalog/prism.yaml | 15 +++++++++++++++ 1 file changed, 15 insertions(+) create mode 100644 plugin-catalog/prism.yaml diff --git a/plugin-catalog/prism.yaml b/plugin-catalog/prism.yaml new file mode 100644 index 0000000000..c173a90895 --- /dev/null +++ b/plugin-catalog/prism.yaml @@ -0,0 +1,15 @@ +name: prism +repo: https://github.com/prismhq/hermes-prism-provider +sha: 1606d8a2fb6d049d61852962ddbc186562f6830d +description: Prism OpenAI-compatible inference with long-context DeepSeek models. +maintainer: Prism +tier: community +category: models +requires_hermes: ">=0.21.3" +docs_url: https://github.com/prismhq/hermes-prism-provider#readme +capabilities: + provides_tools: [] + provides_hooks: [] + provides_middleware: [] + requires_env: + - PRISM_API_KEY From 5a3f3e7ce547551d9410c427b16625a11dc95694 Mon Sep 17 00:00:00 2001 From: rajitkhanna Date: Tue, 15 Sep 2026 17:12:13 -0700 Subject: [PATCH 16/65] fix(plugin-catalog): pin clean Prism model IDs --- plugin-catalog/prism.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/plugin-catalog/prism.yaml b/plugin-catalog/prism.yaml index c173a90895..4ae5da66c7 100644 --- a/plugin-catalog/prism.yaml +++ b/plugin-catalog/prism.yaml @@ -1,6 +1,6 @@ name: prism repo: https://github.com/prismhq/hermes-prism-provider -sha: 1606d8a2fb6d049d61852962ddbc186562f6830d +sha: 5354c0f0382996b3f56c33dc7c2de6a628d2715e description: Prism OpenAI-compatible inference with long-context DeepSeek models. maintainer: Prism tier: community From 249526bc2d0ab7da7ce74dafe04bdf56b0431387 Mon Sep 17 00:00:00 2001 From: rajitkhanna Date: Tue, 15 Sep 2026 17:56:55 -0700 Subject: [PATCH 17/65] chore(plugin-catalog): update Prism pin --- plugin-catalog/prism.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/plugin-catalog/prism.yaml b/plugin-catalog/prism.yaml index 4ae5da66c7..fa8303810b 100644 --- a/plugin-catalog/prism.yaml +++ b/plugin-catalog/prism.yaml @@ -1,6 +1,6 @@ name: prism repo: https://github.com/prismhq/hermes-prism-provider -sha: 5354c0f0382996b3f56c33dc7c2de6a628d2715e +sha: f68a9bff76a1dba30389f56a671292e48b22887a description: Prism OpenAI-compatible inference with long-context DeepSeek models. maintainer: Prism tier: community From a11bac476bb2a2129fdffca9511d41260fa5f51f Mon Sep 17 00:00:00 2001 From: chenxue <17203886+0genlab@users.noreply.github.com> Date: Thu, 17 Sep 2026 14:34:10 +0800 Subject: [PATCH 18/65] plugin-catalog: add aihubmix (community/models provider) --- plugin-catalog/aihubmix.yaml | 15 +++++++++++++++ 1 file changed, 15 insertions(+) create mode 100644 plugin-catalog/aihubmix.yaml diff --git a/plugin-catalog/aihubmix.yaml b/plugin-catalog/aihubmix.yaml new file mode 100644 index 0000000000..9d6aaff0d0 --- /dev/null +++ b/plugin-catalog/aihubmix.yaml @@ -0,0 +1,15 @@ +name: aihubmix +repo: https://github.com/AIhubmix/hermes-provider-aihubmix +sha: f0b6b37858970ced32b73447faaaaad6db899779 +subdir: aihubmix +description: AIHubMix — unified OpenAI-compatible access to 400+ models; /model picker filtered to tool-capable routes. +maintainer: AIHubMix +tier: community +category: models +docs_url: https://github.com/AIhubmix/hermes-provider-aihubmix#readme +capabilities: + provides_tools: [] + provides_hooks: [] + provides_middleware: [] + requires_env: + - AIHUBMIX_API_KEY From 4ab7dec7562dc05d3f2d46b03be98877610b22df Mon Sep 17 00:00:00 2001 From: openclaw Date: Fri, 18 Sep 2026 00:09:16 +0100 Subject: [PATCH 19/65] fix(compaction): name the emitted task heading in the update instruction The iterative-update instruction told the summarizer to update "## Active Task", but the template it is given emits HISTORICAL_TASK_HEADING ("## Historical Task Snapshot"). Leftover from #44454, which renamed the heading in the template and SUMMARY_PREFIX but not in this literal. A summarizer that follows the literal emits a section _HISTORICAL_TASK_SECTION_RE cannot match, so _ground_historical_task_snapshot prepends instead of replacing and the summary carries two task sections. Only the grounded one is disclaimed by SUMMARY_PREFIX, leaving the stale one readable as live work - the hijack class #44454 closed. Co-Authored-By: Claude Opus 5 --- agent/context_compressor.py | 2 +- .../test_context_compressor_task_heading.py | 40 +++++++++++++++++++ 2 files changed, 41 insertions(+), 1 deletion(-) create mode 100644 tests/agent/test_context_compressor_task_heading.py diff --git a/agent/context_compressor.py b/agent/context_compressor.py index d2cf2a9d80..7504cc67c8 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -3531,7 +3531,7 @@ PREVIOUS SUMMARY: NEW TURNS TO INCORPORATE: {content_to_summarize}{_memory_section} -Update the summary using this exact structure. PRESERVE all existing information that is still relevant. ADD new completed actions to the numbered list (continue numbering). Move items from "In Progress" to "Completed Actions" when done. Move answered questions to "Resolved Questions". Update "Active State" to reflect current state. Remove information only if it is clearly obsolete. CRITICAL: Update "## Active Task" to reflect the user's most recent unfulfilled input — this includes any question, decision request, or discussion turn that the assistant has not yet answered. Only write "None" if the last exchange was fully resolved. +Update the summary using this exact structure. PRESERVE all existing information that is still relevant. ADD new completed actions to the numbered list (continue numbering). Move items from "In Progress" to "Completed Actions" when done. Move answered questions to "Resolved Questions". Update "Active State" to reflect current state. Remove information only if it is clearly obsolete. CRITICAL: Update "{HISTORICAL_TASK_HEADING}" to reflect the user's most recent unfulfilled input — this includes any question, decision request, or discussion turn that the assistant has not yet answered. Only write "None" if the last exchange was fully resolved. {_template_sections}""" else: diff --git a/tests/agent/test_context_compressor_task_heading.py b/tests/agent/test_context_compressor_task_heading.py new file mode 100644 index 0000000000..247d247716 --- /dev/null +++ b/tests/agent/test_context_compressor_task_heading.py @@ -0,0 +1,40 @@ +"""The iterative-update instruction must name the heading the template emits.""" + +from types import SimpleNamespace + +from agent.context_compressor import ( + ContextCompressor, + HISTORICAL_TASK_HEADING, + _HISTORICAL_TASK_SECTION_RE, +) + + +def _iterative_prompt() -> str: + """The prompt built when a previous summary exists (the update path).""" + stub = SimpleNamespace( + tail_mode="lean", + _previous_summary="PREVIOUS SUMMARY BODY", + _bound_summary_input=lambda text: text, + ) + stub._summary_template_sections = ContextCompressor._summary_template_sections + stub._build_summary_prompt = ContextCompressor._build_summary_prompt.__get__(stub) + return stub._build_summary_prompt("NEW TURNS", 2000, None, "", True) + + +def test_update_instruction_names_the_emitted_heading(): + prompt = _iterative_prompt() + assert f'Update "{HISTORICAL_TASK_HEADING}"' in prompt + assert '"## Active Task"' not in prompt + + +def test_grounding_prepends_when_the_task_heading_differs(): + """Why the names must agree: a foreign heading yields two task sections.""" + body = "## Active Task\nUser asked: 'do X'\n\n## Goal\nthing\n" + assert _HISTORICAL_TASK_SECTION_RE.search(body) is None + + grounded = ContextCompressor._ground_historical_task_snapshot.__func__( + ContextCompressor, body, [{"role": "user", "content": "do X"}] + ) + headings = [line for line in grounded.splitlines() if line.startswith("## ")] + assert headings.count(HISTORICAL_TASK_HEADING) == 1 + assert "## Active Task" in headings From d295ef02ee511fc6f7da4b3dca2614af602b4e91 Mon Sep 17 00:00:00 2001 From: joaomarcos Date: Thu, 17 Sep 2026 22:12:16 -0300 Subject: [PATCH 20/65] fix(compression): replace alias task headings during snapshot grounding Align the iterative update instruction with HISTORICAL_TASK_HEADING and treat leftover ## Active Task sections as the same snapshot so grounding replaces them instead of prepending a second live-looking task heading. Fixes #114479 --- agent/context_compressor.py | 11 ++- .../test_task_snapshot_heading_contract.py | 73 +++++++++++++++++++ 2 files changed, 81 insertions(+), 3 deletions(-) create mode 100644 tests/agent/test_task_snapshot_heading_contract.py diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 7504cc67c8..0de94ee662 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -1016,7 +1016,12 @@ _PATH_MENTION_RE = re.compile(r"(?:/|~/?|[A-Za-z]:\\)[^\s`'\")\]}<>]+") # MEDIA delivery directives must not reach the summarizer — if one leaks into the summary, the downstream # model may re-emit it as an active directive on the next turn, triggering bogus attachment sends (#14665). _MEDIA_DIRECTIVE_RE = re.compile(r"MEDIA:\S+") -_HISTORICAL_TASK_SECTION_RE = re.compile(rf"(?ms)^{re.escape(HISTORICAL_TASK_HEADING)}\s*\n.*?(?=^## |\Z)") +# Pre-#44454 alias. A summarizer that still emits it must be replaced, not prepended. +_LEGACY_ACTIVE_TASK_HEADING = "## Active Task" +_TASK_SNAPSHOT_HEADINGS = (HISTORICAL_TASK_HEADING, _LEGACY_ACTIVE_TASK_HEADING) +_HISTORICAL_TASK_SECTION_RE = re.compile( + rf"(?ms)^(?:{'|'.join(re.escape(heading) for heading in _TASK_SNAPSHOT_HEADINGS)})\s*\n.*?(?=^## |\Z)" +) def _redact_compaction_text(text: Any) -> str: @@ -3784,8 +3789,8 @@ Write only the summary body. Do not include any preamble or prefix.""" """Reject user attribution when the source transcript has no user.""" if has_user_turn: return - match = re.search(rf"(?ms)^{re.escape(HISTORICAL_TASK_HEADING)}\s*\n(.*?)(?=\n##\s|\Z)", summary) - task_snapshot = match.group(1).strip() if match else "" + match = _HISTORICAL_TASK_SECTION_RE.search(summary) + task_snapshot = match.group(0).split("\n", 1)[-1].strip() if match else "" # The "User asked:" scan can false-positive on quoted tool output; acceptable, since # the RuntimeError only costs one retry on the existing fallback path. if task_snapshot != _NO_USER_TASK_SENTINEL or re.search(r"\bUser\s+asked\s*:", summary, re.IGNORECASE): diff --git a/tests/agent/test_task_snapshot_heading_contract.py b/tests/agent/test_task_snapshot_heading_contract.py new file mode 100644 index 0000000000..e4967db02c --- /dev/null +++ b/tests/agent/test_task_snapshot_heading_contract.py @@ -0,0 +1,73 @@ +"""Task-snapshot heading identity across prompt, template, and grounding. + +The iterative summarizer prompt, the emitted template, and deterministic +grounding must share one heading. A leftover ``## Active Task`` section is +not disclaimed by SUMMARY_PREFIX and reads as live work (#114479 / #44454). +""" + +from types import SimpleNamespace + +from agent.context_compressor import ( + ContextCompressor, + HISTORICAL_TASK_HEADING, +) + +_LEGACY_ACTIVE_TASK_HEADING = "## Active Task" + + +def _iterative_update_prompt() -> str: + stub = SimpleNamespace( + tail_mode="lean", + _previous_summary="PREVIOUS SUMMARY BODY", + _bound_summary_input=lambda text: text, + ) + stub._summary_template_sections = ContextCompressor._summary_template_sections + stub._build_summary_prompt = ContextCompressor._build_summary_prompt.__get__(stub) + return stub._build_summary_prompt("NEW TURNS", 2000, None, "", True) + + +def _headings(text: str) -> list[str]: + return [line for line in text.splitlines() if line.startswith("## ")] + + +def test_iterative_update_instruction_names_canonical_heading(): + prompt = _iterative_update_prompt() + assert f'Update "{HISTORICAL_TASK_HEADING}"' in prompt + assert f'Update "{_LEGACY_ACTIVE_TASK_HEADING}"' not in prompt + assert HISTORICAL_TASK_HEADING in prompt + + +def test_grounding_replaces_legacy_active_task_heading(): + body = ( + f"{_LEGACY_ACTIVE_TASK_HEADING}\n" + "User asked: 'do X'\n\n" + "## Goal\n" + "thing\n" + ) + grounded = ContextCompressor._ground_historical_task_snapshot.__func__( + ContextCompressor, + body, + [{"role": "user", "content": "do X"}], + ) + headings = _headings(grounded) + assert headings.count(HISTORICAL_TASK_HEADING) == 1 + assert _LEGACY_ACTIVE_TASK_HEADING not in headings + assert "## Goal" in headings + + +def test_grounding_keeps_single_canonical_heading(): + body = ( + f"{HISTORICAL_TASK_HEADING}\n" + "User asked: 'stale'\n\n" + "## Goal\n" + "thing\n" + ) + grounded = ContextCompressor._ground_historical_task_snapshot.__func__( + ContextCompressor, + body, + [{"role": "user", "content": "fresh ask"}], + ) + headings = _headings(grounded) + assert headings.count(HISTORICAL_TASK_HEADING) == 1 + assert _LEGACY_ACTIVE_TASK_HEADING not in headings + assert "fresh ask" in grounded From dbf19a6cfa6d777534d48700b6e8ef18d8fa137a Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 00:38:16 -0700 Subject: [PATCH 21/65] fix(compression): collapse duplicate task sections during snapshot grounding Follow-up to the salvaged #114480 (@jonameijers) and #114553 (@JoaoMarcos44). A small summarizer can emit both the canonical "## Historical Task Snapshot" and the legacy "## Active Task" heading in one summary (the 4B model in the issue "duplicates the section set"). With count=1 the grounding pass replaced only the first match and left the second as a live-looking, undisclaimed task section. Replace the first task section with the deterministic snapshot and drop every later one. Tests: fold the two contributor test files into one file with two invariants (prompt names the emitted heading; grounding collapses alias + duplicate sections). The original #114480 test asserted the pre-fix prepend behaviour, which #114553 makes false. Co-authored-by: joaomarcos --- agent/context_compressor.py | 13 +++- .../test_context_compressor_task_heading.py | 55 ++++++++------ .../test_task_snapshot_heading_contract.py | 73 ------------------- 3 files changed, 45 insertions(+), 96 deletions(-) delete mode 100644 tests/agent/test_task_snapshot_heading_contract.py diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 0de94ee662..e2750fc09d 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -3911,7 +3911,18 @@ Write only the summary body. Do not include any preamble or prefix.""" # this regex on the next compaction (deleting every following section). replacement = f"{HISTORICAL_TASK_HEADING}\n{snapshot}\n\n" if _HISTORICAL_TASK_SECTION_RE.search(body): - return _HISTORICAL_TASK_SECTION_RE.sub(lambda _m: replacement, body, count=1).strip() + # Replace the first task section and drop every later one: a summarizer that emits both the + # canonical heading and the legacy alias would otherwise leave a second, undisclaimed task section. + seen = False + + def _collapse(_m: re.Match) -> str: + nonlocal seen + if seen: + return "" + seen = True + return replacement + + return _HISTORICAL_TASK_SECTION_RE.sub(_collapse, body).strip() return f"{replacement}{body}".strip() @classmethod diff --git a/tests/agent/test_context_compressor_task_heading.py b/tests/agent/test_context_compressor_task_heading.py index 247d247716..eacc4c7c82 100644 --- a/tests/agent/test_context_compressor_task_heading.py +++ b/tests/agent/test_context_compressor_task_heading.py @@ -1,16 +1,23 @@ -"""The iterative-update instruction must name the heading the template emits.""" +"""Task-snapshot heading identity across the summarizer prompt, the template, and grounding. + +The iterative-update instruction, the emitted template, and ``_ground_historical_task_snapshot`` must agree on +one heading. A leftover ``## Active Task`` section is not disclaimed by SUMMARY_PREFIX and reads as live work, +so grounding must replace it (and any duplicate task section) rather than prepend a second one. +""" from types import SimpleNamespace -from agent.context_compressor import ( - ContextCompressor, - HISTORICAL_TASK_HEADING, - _HISTORICAL_TASK_SECTION_RE, -) +from agent.context_compressor import ContextCompressor, HISTORICAL_TASK_HEADING + +_LEGACY_ACTIVE_TASK_HEADING = "## Active Task" -def _iterative_prompt() -> str: - """The prompt built when a previous summary exists (the update path).""" +def _headings(text: str) -> list[str]: + return [line for line in text.splitlines() if line.startswith("## ")] + + +def test_update_instruction_names_the_emitted_heading(): + """The prompt built when a previous summary exists (the update path) names HISTORICAL_TASK_HEADING.""" stub = SimpleNamespace( tail_mode="lean", _previous_summary="PREVIOUS SUMMARY BODY", @@ -18,23 +25,27 @@ def _iterative_prompt() -> str: ) stub._summary_template_sections = ContextCompressor._summary_template_sections stub._build_summary_prompt = ContextCompressor._build_summary_prompt.__get__(stub) - return stub._build_summary_prompt("NEW TURNS", 2000, None, "", True) + prompt = stub._build_summary_prompt("NEW TURNS", 2000, None, "", True) - -def test_update_instruction_names_the_emitted_heading(): - prompt = _iterative_prompt() assert f'Update "{HISTORICAL_TASK_HEADING}"' in prompt - assert '"## Active Task"' not in prompt + assert f'"{_LEGACY_ACTIVE_TASK_HEADING}"' not in prompt -def test_grounding_prepends_when_the_task_heading_differs(): - """Why the names must agree: a foreign heading yields two task sections.""" - body = "## Active Task\nUser asked: 'do X'\n\n## Goal\nthing\n" - assert _HISTORICAL_TASK_SECTION_RE.search(body) is None - - grounded = ContextCompressor._ground_historical_task_snapshot.__func__( - ContextCompressor, body, [{"role": "user", "content": "do X"}] +def test_grounding_collapses_alias_and_duplicate_task_sections(): + """A summarizer that emits the legacy alias, or both headings, ends up with exactly one grounded section.""" + body = ( + f"{HISTORICAL_TASK_HEADING}\nUser asked: 'stale canonical'\n\n" + "## Goal\nthing\n\n" + f"{_LEGACY_ACTIVE_TASK_HEADING}\nUser asked: 'stale alias'\n\n" + "## Constraints & Preferences\n- none\n" ) - headings = [line for line in grounded.splitlines() if line.startswith("## ")] + grounded = ContextCompressor._ground_historical_task_snapshot.__func__( + ContextCompressor, body, [{"role": "user", "content": "fresh ask"}] + ) + + headings = _headings(grounded) assert headings.count(HISTORICAL_TASK_HEADING) == 1 - assert "## Active Task" in headings + assert _LEGACY_ACTIVE_TASK_HEADING not in headings + assert headings[1:] == ["## Goal", "## Constraints & Preferences"] + assert "fresh ask" in grounded + assert "stale" not in grounded diff --git a/tests/agent/test_task_snapshot_heading_contract.py b/tests/agent/test_task_snapshot_heading_contract.py deleted file mode 100644 index e4967db02c..0000000000 --- a/tests/agent/test_task_snapshot_heading_contract.py +++ /dev/null @@ -1,73 +0,0 @@ -"""Task-snapshot heading identity across prompt, template, and grounding. - -The iterative summarizer prompt, the emitted template, and deterministic -grounding must share one heading. A leftover ``## Active Task`` section is -not disclaimed by SUMMARY_PREFIX and reads as live work (#114479 / #44454). -""" - -from types import SimpleNamespace - -from agent.context_compressor import ( - ContextCompressor, - HISTORICAL_TASK_HEADING, -) - -_LEGACY_ACTIVE_TASK_HEADING = "## Active Task" - - -def _iterative_update_prompt() -> str: - stub = SimpleNamespace( - tail_mode="lean", - _previous_summary="PREVIOUS SUMMARY BODY", - _bound_summary_input=lambda text: text, - ) - stub._summary_template_sections = ContextCompressor._summary_template_sections - stub._build_summary_prompt = ContextCompressor._build_summary_prompt.__get__(stub) - return stub._build_summary_prompt("NEW TURNS", 2000, None, "", True) - - -def _headings(text: str) -> list[str]: - return [line for line in text.splitlines() if line.startswith("## ")] - - -def test_iterative_update_instruction_names_canonical_heading(): - prompt = _iterative_update_prompt() - assert f'Update "{HISTORICAL_TASK_HEADING}"' in prompt - assert f'Update "{_LEGACY_ACTIVE_TASK_HEADING}"' not in prompt - assert HISTORICAL_TASK_HEADING in prompt - - -def test_grounding_replaces_legacy_active_task_heading(): - body = ( - f"{_LEGACY_ACTIVE_TASK_HEADING}\n" - "User asked: 'do X'\n\n" - "## Goal\n" - "thing\n" - ) - grounded = ContextCompressor._ground_historical_task_snapshot.__func__( - ContextCompressor, - body, - [{"role": "user", "content": "do X"}], - ) - headings = _headings(grounded) - assert headings.count(HISTORICAL_TASK_HEADING) == 1 - assert _LEGACY_ACTIVE_TASK_HEADING not in headings - assert "## Goal" in headings - - -def test_grounding_keeps_single_canonical_heading(): - body = ( - f"{HISTORICAL_TASK_HEADING}\n" - "User asked: 'stale'\n\n" - "## Goal\n" - "thing\n" - ) - grounded = ContextCompressor._ground_historical_task_snapshot.__func__( - ContextCompressor, - body, - [{"role": "user", "content": "fresh ask"}], - ) - headings = _headings(grounded) - assert headings.count(HISTORICAL_TASK_HEADING) == 1 - assert _LEGACY_ACTIVE_TASK_HEADING not in headings - assert "fresh ask" in grounded From de633bfb32d57e17554726b733faf50e4bef31b8 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 00:38:27 -0700 Subject: [PATCH 22/65] chore: map contributor email for @jonameijers --- contributors/emails/openclaw@Jonas-Mac-Studio.local | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/openclaw@Jonas-Mac-Studio.local diff --git a/contributors/emails/openclaw@Jonas-Mac-Studio.local b/contributors/emails/openclaw@Jonas-Mac-Studio.local new file mode 100644 index 0000000000..1966aa8adf --- /dev/null +++ b/contributors/emails/openclaw@Jonas-Mac-Studio.local @@ -0,0 +1 @@ +jonameijers From ffe69fd71101dfe89076063d630b6765f7b56553 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 01:20:50 -0700 Subject: [PATCH 23/65] test(compression): sibling asserts the grounded task heading, not the legacy alias test_summarizer_output_think_block_stripped_before_store fed the summarizer output a "## Active Task" heading and asserted it survived. That only held because grounding could not match the alias and prepended a second task section next to it (#114479). With the alias now normalised the invariant is: canonical heading present, alias absent. --- tests/agent/test_compression_small_ctx_threshold_floor.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/agent/test_compression_small_ctx_threshold_floor.py b/tests/agent/test_compression_small_ctx_threshold_floor.py index 11797c5545..abaf240707 100644 --- a/tests/agent/test_compression_small_ctx_threshold_floor.py +++ b/tests/agent/test_compression_small_ctx_threshold_floor.py @@ -83,7 +83,10 @@ class TestReasoningExcludedFromSummarizer: out = comp._generate_summary([{"role": "user", "content": "hi"}]) assert out is not None assert "OUTPUT_TRACE" not in out - assert "## Active Task" in out + # Snapshot grounding rewrites the legacy "## Active Task" alias into the canonical, grounded + # section instead of prepending a second task section next to it. + assert cc.HISTORICAL_TASK_HEADING in out + assert "## Active Task" not in out # The iterative-update seed must be clean too, or the trace compounds # across every subsequent compaction. assert "OUTPUT_TRACE" not in (comp._previous_summary or "") From f797a23c093da26147ccce13e6604ca692eee6ff Mon Sep 17 00:00:00 2001 From: KoNit-K Date: Fri, 18 Sep 2026 12:14:54 +0800 Subject: [PATCH 24/65] fix(review): cap default background input budget --- agent/background_review.py | 49 ++++++++++++++++--- hermes_cli/config_defaults.py | 9 ++-- .../test_background_review_input_budget.py | 45 +++++++++++++++-- 3 files changed, 89 insertions(+), 14 deletions(-) diff --git a/agent/background_review.py b/agent/background_review.py index 5979b1134b..2d0db4ea5d 100644 --- a/agent/background_review.py +++ b/agent/background_review.py @@ -151,9 +151,12 @@ _REVIEW_MAX_ITERATIONS = 16 # Aggregate INPUT-token budget for one review fork (checked in conversation_loop's # ``_review_input_budget_exhausted``). Request #1 replays the full snapshot as a warm cache read # (both compression gates deferred until the first response); compaction then bounds each -# request, but nothing else caps the SUM across the tool loop. 2x the historical 300k foreground -# trigger. Override via ``auxiliary.background_review.max_input_tokens``; <= 0 disables. -_REVIEW_MAX_INPUT_TOKENS_DEFAULT = 600_000 +# request, but nothing else caps the SUM across the tool loop. The default leaves 25% of the +# review model's context window available and never exceeds the historical cloud-scale ceiling. +# Override via ``auxiliary.background_review.max_input_tokens``; <= 0 disables. +_REVIEW_MAX_INPUT_TOKENS_CAP = 600_000 +_REVIEW_INPUT_CONTEXT_FRACTION = 0.75 +_REVIEW_MAX_INPUT_TOKENS_FALLBACK = 120_000 def _task_block(cfg: Any) -> Dict[str, Any]: @@ -175,13 +178,45 @@ def _background_review_task_config(task_cfg: Optional[Dict[str, Any]] = None) -> return {} -def _review_input_token_budget(task_cfg: Optional[Dict[str, Any]] = None) -> Optional[int]: +def _context_derived_review_input_budget(runtime: Optional[Dict[str, Any]] = None) -> int: + """Return a bounded default based on the active review runtime when known.""" + runtime = runtime if isinstance(runtime, dict) else {} + model = str(runtime.get("model") or "").strip() + if not model: + return _REVIEW_MAX_INPUT_TOKENS_FALLBACK + try: + from agent.model_metadata import get_model_context_length + + context_window = get_model_context_length( + model, + base_url=str(runtime.get("base_url") or ""), + api_key=str(runtime.get("api_key") or ""), + provider=str(runtime.get("provider") or ""), + ) + except Exception: + logger.debug("Background review context-window resolution failed", exc_info=True) + return _REVIEW_MAX_INPUT_TOKENS_FALLBACK + if not isinstance(context_window, int) or context_window <= 0: + return _REVIEW_MAX_INPUT_TOKENS_FALLBACK + return min( + _REVIEW_MAX_INPUT_TOKENS_CAP, + max(1, int(context_window * _REVIEW_INPUT_CONTEXT_FRACTION)), + ) + + +def _review_input_token_budget( + task_cfg: Optional[Dict[str, Any]] = None, + runtime: Optional[Dict[str, Any]] = None, +) -> Optional[int]: """Aggregate input-token budget for one review fork (None = unlimited; <= 0 disables).""" - raw = _background_review_task_config(task_cfg).get("max_input_tokens", _REVIEW_MAX_INPUT_TOKENS_DEFAULT) + task = _background_review_task_config(task_cfg) + if "max_input_tokens" not in task: + return _context_derived_review_input_budget(runtime) + raw = task["max_input_tokens"] try: budget = int(raw) except (TypeError, ValueError): - budget = _REVIEW_MAX_INPUT_TOKENS_DEFAULT + return _context_derived_review_input_budget(runtime) return budget if budget > 0 else None @@ -978,7 +1013,7 @@ def build_cache_parity_fork( _detach_fork_compression(review_agent) # Compaction bounds a single request; this bounds the WHOLE review (checked in # conversation_loop via _review_input_budget_exhausted). - review_agent._review_input_token_budget = _review_input_token_budget(task_cfg) + review_agent._review_input_token_budget = _review_input_token_budget(task_cfg, _rt) return review_agent, _rt, _routed diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 94a1e465bf..80a3bfbafd 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -747,14 +747,15 @@ DEFAULT_CONFIG = { "monitor": _aux(60), # important-mail 0-10 scorer; high-volume, small model fine # Post-turn self-improvement fork (save memory / patch skill). "auto" = main model replaying # the full conversation (warm cache); other models replay a compact digest (~3-5x cheaper). - # enabled=false skips auto spawns (/refine still works). max_input_tokens caps the SUM of - # replayed input tokens over the review loop (iterations capped at 16); the loop stops - # before crossing it. <= 0 = unlimited. + # enabled=false skips auto spawns (/refine still works). An explicit max_input_tokens caps + # the SUM of replayed input tokens over the review loop (iterations capped at 16); the loop + # stops before crossing it. When unset, the runtime derives a budget from the active model + # context window. <= 0 = unlimited. # reasoning_effort is IGNORED while the review stays on the main model: the fork inherits the # conversation's reasoning config verbatim so its request bytes keep the parent's warm # prompt-cache prefix (#30532). Set provider/model below to route the review to another model # if you want a different effort level; a one-time warning says so when the key is set. - "background_review": {"enabled": True, **_aux(120), "max_input_tokens": 600000}, + "background_review": {"enabled": True, **_aux(120)}, # No reasoning_effort on MoA blocks by design — configured PER SLOT in the preset # (moa.presets..reference_models[].reasoning_effort / aggregator.reasoning_effort). "moa_reference": _aux(900, reasoning_effort=False), diff --git a/tests/agent/test_background_review_input_budget.py b/tests/agent/test_background_review_input_budget.py index b0d483f93d..0c556111a7 100644 --- a/tests/agent/test_background_review_input_budget.py +++ b/tests/agent/test_background_review_input_budget.py @@ -224,16 +224,55 @@ def test_review_input_budget_exhausted_predicate_edge_cases(): @pytest.mark.parametrize( ("config_value", "expected"), [ - ({}, 600_000), ({"max_input_tokens": 1_000_000}, 1_000_000), ({"max_input_tokens": 0}, None), ({"max_input_tokens": -5}, None), - ({"max_input_tokens": "not-a-number"}, 600_000), ({"max_input_tokens": "300000"}, 300_000), ], ) def test_review_input_token_budget_resolution(config_value, expected): - """Config parsing: default, override, explicit disable, garbage fallback.""" + """Explicit settings retain their established override and unlimited semantics.""" from agent.background_review import _review_input_token_budget assert _review_input_token_budget(config_value) == expected + + +@pytest.mark.parametrize( + ("context_window", "expected"), + [ + (65_536, 49_152), + (4_096, 3_072), + ], +) +def test_review_input_token_budget_default_tracks_active_context(context_window, expected): + """An unset budget leaves room below the review model's context window.""" + from agent.background_review import _review_input_token_budget + + runtime = {"provider": "lmstudio", "model": "local-model", "base_url": "http://localhost:1234"} + with patch("agent.model_metadata.get_model_context_length", return_value=context_window): + assert _review_input_token_budget({}, runtime) == expected + + +def test_review_input_token_budget_malformed_value_uses_context_derived_default(): + """A malformed explicit value is safe, rather than restoring the old 600k default.""" + from agent.background_review import _review_input_token_budget + + runtime = {"provider": "lmstudio", "model": "local-model", "base_url": "http://localhost:1234"} + with patch("agent.model_metadata.get_model_context_length", return_value=65_536): + assert _review_input_token_budget({"max_input_tokens": "not-a-number"}, runtime) == 49_152 + + +def test_review_input_token_budget_unknown_context_uses_conservative_fallback(): + """Failed context discovery must still bound unattended review work.""" + from agent.background_review import _review_input_token_budget + + runtime = {"provider": "local", "model": "unknown", "base_url": "http://localhost:1234"} + with patch("agent.model_metadata.get_model_context_length", side_effect=RuntimeError("unavailable")): + assert _review_input_token_budget({}, runtime) == 120_000 + + +def test_background_review_config_does_not_freeze_a_fixed_input_budget(): + """The config default must leave the budget resolver access to the active runtime.""" + from hermes_cli.config_defaults import DEFAULT_CONFIG + + assert "max_input_tokens" not in DEFAULT_CONFIG["auxiliary"]["background_review"] From a6ad3cc3ffaf608ef067f0a7fc87473b5a40e486 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 00:54:42 -0700 Subject: [PATCH 25/65] fix(review): derive the default input budget from the fork's resolved context window The salvaged commit (#114651) re-probed the endpoint through get_model_context_length at spawn time. The review fork is a full AIAgent whose context_compressor.context_length is already resolved by the same chain (config override, catalog, persisted provider limit, endpoint probe), so read that instead: no second lookup, and the budget tracks exactly the window the fork's own compaction uses. Tests trimmed to two invariants on the new seam. Docs: `auxiliary.background_review.max_input_tokens` and its derived default in website/docs/user-guide/features/memory.md, including the note that a top-level `background_review:` block is not read (#114645 secondary). Co-authored-by: Artie Fisher --- agent/background_review.py | 48 ++++++------------ .../test_background_review_input_budget.py | 39 ++++----------- tests/agent/test_failed_turn_site_codes.py | 49 +++++++++++++++++++ website/docs/user-guide/features/memory.md | 20 ++++++++ 4 files changed, 94 insertions(+), 62 deletions(-) diff --git a/agent/background_review.py b/agent/background_review.py index 2d0db4ea5d..51bf3c70f2 100644 --- a/agent/background_review.py +++ b/agent/background_review.py @@ -178,45 +178,27 @@ def _background_review_task_config(task_cfg: Optional[Dict[str, Any]] = None) -> return {} -def _context_derived_review_input_budget(runtime: Optional[Dict[str, Any]] = None) -> int: - """Return a bounded default based on the active review runtime when known.""" - runtime = runtime if isinstance(runtime, dict) else {} - model = str(runtime.get("model") or "").strip() - if not model: +def _context_derived_review_input_budget(review_agent: Any = None) -> int: + """Default budget: 75% of the review fork's resolved context window, capped at the historical + 600k ceiling. The fork's ``context_compressor.context_length`` is already resolved by + ``AIAgent.__init__`` (config overrides, catalog, endpoint probe) — no second lookup here. + Unknown window → a conservative fixed fallback so unattended review work stays bounded.""" + context_window = getattr(getattr(review_agent, "context_compressor", None), "context_length", None) + if not isinstance(context_window, int) or isinstance(context_window, bool) or context_window <= 0: return _REVIEW_MAX_INPUT_TOKENS_FALLBACK - try: - from agent.model_metadata import get_model_context_length - - context_window = get_model_context_length( - model, - base_url=str(runtime.get("base_url") or ""), - api_key=str(runtime.get("api_key") or ""), - provider=str(runtime.get("provider") or ""), - ) - except Exception: - logger.debug("Background review context-window resolution failed", exc_info=True) - return _REVIEW_MAX_INPUT_TOKENS_FALLBACK - if not isinstance(context_window, int) or context_window <= 0: - return _REVIEW_MAX_INPUT_TOKENS_FALLBACK - return min( - _REVIEW_MAX_INPUT_TOKENS_CAP, - max(1, int(context_window * _REVIEW_INPUT_CONTEXT_FRACTION)), - ) + return min(_REVIEW_MAX_INPUT_TOKENS_CAP, max(1, int(context_window * _REVIEW_INPUT_CONTEXT_FRACTION))) def _review_input_token_budget( - task_cfg: Optional[Dict[str, Any]] = None, - runtime: Optional[Dict[str, Any]] = None, + task_cfg: Optional[Dict[str, Any]] = None, review_agent: Any = None, ) -> Optional[int]: - """Aggregate input-token budget for one review fork (None = unlimited; <= 0 disables).""" + """Aggregate input-token budget for one review fork (None = unlimited; <= 0 disables). Unset + or malformed ``max_input_tokens`` → derived from ``review_agent``'s context window.""" task = _background_review_task_config(task_cfg) - if "max_input_tokens" not in task: - return _context_derived_review_input_budget(runtime) - raw = task["max_input_tokens"] try: - budget = int(raw) - except (TypeError, ValueError): - return _context_derived_review_input_budget(runtime) + budget = int(task["max_input_tokens"]) + except (KeyError, TypeError, ValueError): + return _context_derived_review_input_budget(review_agent) return budget if budget > 0 else None @@ -1013,7 +995,7 @@ def build_cache_parity_fork( _detach_fork_compression(review_agent) # Compaction bounds a single request; this bounds the WHOLE review (checked in # conversation_loop via _review_input_budget_exhausted). - review_agent._review_input_token_budget = _review_input_token_budget(task_cfg, _rt) + review_agent._review_input_token_budget = _review_input_token_budget(task_cfg, review_agent) return review_agent, _rt, _routed diff --git a/tests/agent/test_background_review_input_budget.py b/tests/agent/test_background_review_input_budget.py index 0c556111a7..846ac93111 100644 --- a/tests/agent/test_background_review_input_budget.py +++ b/tests/agent/test_background_review_input_budget.py @@ -237,38 +237,19 @@ def test_review_input_token_budget_resolution(config_value, expected): assert _review_input_token_budget(config_value) == expected -@pytest.mark.parametrize( - ("context_window", "expected"), - [ - (65_536, 49_152), - (4_096, 3_072), - ], -) -def test_review_input_token_budget_default_tracks_active_context(context_window, expected): - """An unset budget leaves room below the review model's context window.""" +def test_review_input_token_budget_default_tracks_forks_context_window(): + """Unset or malformed ``max_input_tokens`` → 75% of the fork's RESOLVED window (a 65k local + model gets ~49k, not the cloud-scale 600k), capped at 600k; unknown window → 120k fallback.""" from agent.background_review import _review_input_token_budget - runtime = {"provider": "lmstudio", "model": "local-model", "base_url": "http://localhost:1234"} - with patch("agent.model_metadata.get_model_context_length", return_value=context_window): - assert _review_input_token_budget({}, runtime) == expected + def fork(window): + return SimpleNamespace(context_compressor=SimpleNamespace(context_length=window)) - -def test_review_input_token_budget_malformed_value_uses_context_derived_default(): - """A malformed explicit value is safe, rather than restoring the old 600k default.""" - from agent.background_review import _review_input_token_budget - - runtime = {"provider": "lmstudio", "model": "local-model", "base_url": "http://localhost:1234"} - with patch("agent.model_metadata.get_model_context_length", return_value=65_536): - assert _review_input_token_budget({"max_input_tokens": "not-a-number"}, runtime) == 49_152 - - -def test_review_input_token_budget_unknown_context_uses_conservative_fallback(): - """Failed context discovery must still bound unattended review work.""" - from agent.background_review import _review_input_token_budget - - runtime = {"provider": "local", "model": "unknown", "base_url": "http://localhost:1234"} - with patch("agent.model_metadata.get_model_context_length", side_effect=RuntimeError("unavailable")): - assert _review_input_token_budget({}, runtime) == 120_000 + assert _review_input_token_budget({}, fork(65_536)) == 49_152 + assert _review_input_token_budget({"max_input_tokens": "not-a-number"}, fork(65_536)) == 49_152 + assert _review_input_token_budget({}, fork(2_000_000)) == 600_000 + assert _review_input_token_budget({}, fork(None)) == 120_000 + assert _review_input_token_budget({}, None) == 120_000 def test_background_review_config_does_not_freeze_a_fixed_input_budget(): diff --git a/tests/agent/test_failed_turn_site_codes.py b/tests/agent/test_failed_turn_site_codes.py index afeed1cf39..6ce9ce279e 100644 --- a/tests/agent/test_failed_turn_site_codes.py +++ b/tests/agent/test_failed_turn_site_codes.py @@ -40,6 +40,55 @@ def test_overflow_exhaustion_is_non_retryable_context_overflow_with_slash_comman assert build_error_surface_from_result(result)["code"] == "context_overflow" +def _context_rejection(request_tokens: int, window: int = 65_536, error="HTTP 500: Context size has been exceeded."): + """Drive ``_recover_context_length`` with a provider "context exceeded" and a request the rough + estimator prices at ``request_tokens`` against a ``window``-token model (no output cap).""" + from unittest.mock import patch + + from agent.turn_overflow import _recover_context_length + from agent.turn_retry_state import TurnRetryState + + st = _recovery() + st.compression_attempts = 0 + st.agent.max_tokens = None + st.agent.context_compressor = SimpleNamespace(context_length=window) + st.agent.provider, st.agent.base_url, st.agent.tools = "lmstudio", "http://127.0.0.1:1234/v1", None + st.agent._buffer_vprint = st.agent._buffer_diagnostic_status = lambda *a, **k: None + compressed = [] + st.agent._compress_context = lambda msgs, *a, **k: (compressed.append(1) or [{"role": "user", "content": "x"}], None) + with patch("agent.model_metadata.estimate_request_tokens_rough", return_value=request_tokens), \ + patch("agent.model_metadata.estimate_messages_tokens_rough", side_effect=[request_tokens, 10]), \ + patch("agent.turn_overflow.time.sleep", lambda _: None): + verdict = _recover_context_length(st, TurnRetryState(), error) + return verdict, compressed + + +def test_context_rejection_far_below_the_window_is_not_blamed_on_the_conversation(): + """A single-slot local server rejecting a ~3k-token request (another thread held its + context) must not read as "this conversation has grown too long": no compression, transient + + retryable, and the gateway's overflow verdict (drop message / auto-reset) stays off.""" + from gateway.run_turn import is_context_overflow_failure_result + + verdict, compressed = _context_rejection(3_000) + result = verdict.result + assert verdict.action == "return" and not compressed + assert result["failed"] is True and not result.get("compression_exhausted") + assert result["failure_reason"] == "server_error" and result["failure_retryable"] is True + text = result["final_response"] + assert "grown too long" not in text and "/new" not in text + assert "3,000" in text and "65,536" in text and "background" in text and "/retry" in text + assert is_context_overflow_failure_result(result, history_len=2) is False + + +def test_context_rejection_near_the_window_still_compresses(): + """Control: a request that plausibly overflows keeps the compress-and-retry path, and so does + a small local estimate when the SERVER quoted its own count — its measurement wins.""" + verdict, compressed = _context_rejection(60_000) + assert compressed and verdict.action == "break" + verdict, compressed = _context_rejection(3_000, error="prompt is too long: 70000 tokens > 65536 maximum") + assert compressed and verdict.action == "break" + + def test_payload_and_context_overflow_share_one_next_step(): """413 and context-length exhaustion differ in cause text but never in what to do.""" a = _recovery().count_attempt(payload_too_large=True).result["final_response"] diff --git a/website/docs/user-guide/features/memory.md b/website/docs/user-guide/features/memory.md index f15f752960..0e441c2a37 100644 --- a/website/docs/user-guide/features/memory.md +++ b/website/docs/user-guide/features/memory.md @@ -369,6 +369,26 @@ auxiliary: With `enabled: false`, automatic post-turn forks do not spawn; manual `/refine` still works. +### Capping review cost (`max_input_tokens`) + +The review loop replays the conversation on every provider request it makes, +so a single review can multiply input tokens across its tool iterations. +`max_input_tokens` caps the SUM of replayed input tokens for one review; the +loop stops before crossing it. `<= 0` means unlimited. + +```yaml +auxiliary: + background_review: + max_input_tokens: 48000 # <= 0 = unlimited +``` + +When the key is unset, the budget is derived from the review model's resolved +context window: 75% of the window, capped at 600,000 tokens — so it also binds +on small local models (a 65,536-token model gets 49,152), where a fixed +cloud-scale default would never bite. If the window cannot be resolved, a +conservative 120,000-token fallback applies. Note the key lives under +`auxiliary:`; a top-level `background_review:` block is not read. + Fork usage is persisted in `session_model_usage` with `task='background_review'` and a completion line is written to `agent.log` (`Background review complete: thread=bg-review calls=… in=… out=… result=…`). From 5e950506089fdd5749112f38d47cecf2b1168b7f Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 00:54:42 -0700 Subject: [PATCH 26/65] fix(agent): a server context rejection the transcript cannot explain is no longer "conversation too long" MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A single-slot local server (LM Studio, Ollama) returns 500 "Context size has been exceeded." when ANOTHER request — a background review from an earlier session — holds its context. The foreground loop classified that as context_overflow, tried to compress a one-sentence conversation, could not shrink it, and rendered "This conversation has grown too long … /new … /compress" with compression_exhausted=True (gateway auto-reset, user message dropped from the transcript). _recover_context_length now measures first: when the server quoted no count of its own and the local request estimate (+ output reservation) sits under half the known window, the turn ends with distinct copy naming the likely cause (another request on the server / smaller server window), failure_reason=server_error, retryable, no compression_exhausted — so CLI, TUI/Desktop and the gateway all render a transient failure. Servers that quote their own measurement ("233153 tokens > 200000 maximum") and requests near the window keep the compress-and-retry path unchanged. The two buffered "keeping context_length … and compressing" notices drop the trailing clause so they read true on both paths. Fixes #114644 --- agent/turn_failure_copy.py | 11 +++++++++ agent/turn_overflow.py | 44 +++++++++++++++++++++++++++++++---- website/docs/reference/faq.md | 2 ++ 3 files changed, 53 insertions(+), 4 deletions(-) diff --git a/agent/turn_failure_copy.py b/agent/turn_failure_copy.py index ffd7f06bea..e8337db5b6 100644 --- a/agent/turn_failure_copy.py +++ b/agent/turn_failure_copy.py @@ -253,6 +253,17 @@ _ONE_OFF_COPY: Dict[str, str] = { "your settings (compression.enabled). Run /compress to shrink it now, /new to start " "fresh, or pick a model with a bigger context window." ), + # Wording deliberately avoids the overflow phrases gateway/run_turn.py matches on + # (``_CONTEXT_OVERFLOW_ERROR_PHRASES``): this failure is transient, so the user's + # message must stay in the transcript and the session must not be auto-reset. + "server_context_rejection": ( + "The model server rejected this request as too large, but this conversation is only " + "about {tokens:,} tokens — well under the {window:,}-token window Hermes knows for " + "{model} — so shrinking it would not help. Another request on the same server (for " + "example a background memory review from an earlier session) was probably holding its " + "capacity, or the server runs {model} with a smaller window than Hermes assumes. Wait a " + "moment and send /retry; if it keeps happening, check the server's context setting." + ), "stream_dropped_tool_call": ( "The connection to {label} kept dropping while the model was writing a large action, " "so nothing was run. Check your network and send /retry; asking for the file in smaller " diff --git a/agent/turn_overflow.py b/agent/turn_overflow.py index 925a328ccb..e4b0066843 100644 --- a/agent/turn_overflow.py +++ b/agent/turn_overflow.py @@ -10,6 +10,7 @@ estimators that tests patch on the loop module are imported lazily inside the ha from __future__ import annotations import logging +import re import time from dataclasses import dataclass from typing import Any, Dict, List, Optional, Tuple @@ -33,6 +34,11 @@ logger = logging.getLogger("agent.conversation_loop") _RETRY_HINT = " 💡 Try /new to start a fresh conversation, or /compress to retry compression." +# A provider "context exceeded" whose request (incl. the output reservation) is under this share +# of the known window is not explained by the transcript; the rough estimator is ±30%, so a real +# overflow (request ≈ window) never lands below half. +_UNEXPLAINED_REJECTION_FRACTION = 0.5 + _GITHUB_MODELS_HINT = ( " 💡 GitHub Models free tier (models.inference.ai.azure.com) caps every", " request at ~8K tokens. Hermes' system prompt + tool schemas baseline", @@ -88,7 +94,8 @@ class _Recovery(OverflowVerdict): def fail_turn( self, final_response: str, *, notices: tuple = (), log: Optional[tuple] = None, - compression_exhausted: bool = True, **extra: Any, + compression_exhausted: bool = True, reason: str = "context_overflow", + retryable: bool = False, **extra: Any, ) -> OverflowVerdict: """End the turn as failed/partial. ``notices`` flush the buffered retry trace first so the user sees what compression attempts were made.""" @@ -108,7 +115,7 @@ class _Recovery(OverflowVerdict): "error": final_response, "partial": True, "failed": True, - }, "context_overflow", False) + }, reason, retryable) if compression_exhausted: # Reuse the gateway's existing context-recovery contract (#98722, salvaged from #98741). The # bloated transcript remains intact while future input can move to a clean session instead of @@ -346,12 +353,12 @@ def _adopt_provider_context_limit(st: _Recovery, error_msg: str, old_ctx: int) - if is_minimax_provider and "context window exceeds limit (" in error_msg: agent._buffer_vprint( f"Provider reported overflow amount only; " - f"keeping context_length at {old_ctx:,} tokens and compressing." + f"keeping context_length at {old_ctx:,} tokens." ) else: agent._buffer_vprint( f"⚠️ Context length exceeded, but provider did not report a max context length; " - f"keeping context_length at {old_ctx:,} tokens and compressing." + f"keeping context_length at {old_ctx:,} tokens." ) return None @@ -387,6 +394,35 @@ def _recover_context_length(st: _Recovery, _retry: TurnRetryState, error_msg: st new_ctx = _adopt_provider_context_limit(st, error_msg, old_ctx) + # A rejection the transcript cannot explain (#114644): the request sits far below the window + # Hermes knows for this model (after adopting any limit the server reported), so compressing + # would destroy history for nothing. Single-slot local servers reject like this while ANOTHER + # request — a background review from an earlier session — holds their context. Name that, + # keep the turn retryable and transient: no "conversation too long", no gateway auto-reset. + # Only when the server quoted NO measurement of its own: "prompt is too long: 233153 tokens + # > 200000" is the server's count and beats the local estimate. + window = agent.context_compressor.context_length + request_tokens = st.request_tokens() + max(0, int(getattr(agent, "max_tokens", 0) or 0)) + if ( + not re.search(r"\d{4,}", error_msg) + and isinstance(window, int) and window > 0 + and request_tokens < window * _UNEXPLAINED_REJECTION_FRACTION + ): + return st.fail_turn( + site_copy("server_context_rejection", model=agent.model, tokens=request_tokens, window=window), + notices=( + f"❌ The server rejected the request as too large, but it is only ~{request_tokens:,} " + f"tokens against a {window:,}-token window — not compressing.", + " 💡 Wait a moment and /retry; another request on the same server (e.g. a background " + "review) may have been holding its context.", + ), + log=( + "%sProvider context rejection not explained by request size (~%s of %s tokens); " + "skipping compression: %s", agent.log_prefix, f"{request_tokens:,}", f"{window:,}", error_msg[:200], + ), + compression_exhausted=False, reason=FailoverReason.server_error.value, retryable=True, + ) + exhausted = st.count_attempt() if exhausted is not None: return exhausted diff --git a/website/docs/reference/faq.md b/website/docs/reference/faq.md index 62fc20ec60..05d025c63a 100644 --- a/website/docs/reference/faq.md +++ b/website/docs/reference/faq.md @@ -323,6 +323,8 @@ Look at the CLI startup line — it shows the detected context length (e.g., ` **Local servers (llama.cpp, Ollama) that go silent instead of erroring:** when a provider rejects a request as too large, Hermes compacts the conversation and rebuilds the request. Hermes re-measures the *complete* rebuilt request (system prompt + tool schemas + messages) before retrying, and runs further bounded compaction passes if it is still over the threshold. If the request still cannot fit, the turn ends with `Context length exceeded: compression could not reduce the rebuilt request below the safe threshold` rather than sending an oversized request that llama.cpp would silently truncate (`stop processing: n_tokens = 65535, truncated = 1` in the server log). If you hit that message, the fix is almost always the configured `context_length` above: make it match the server's actual `-c` / `--ctx-size`. +**"The model server rejected this request as too large, but this conversation is only about N tokens…":** the server said "context exceeded" without quoting any measurement, while Hermes's own estimate of the request is far below the window it knows for the model — so it does **not** compress or blame the conversation, and the turn stays retryable. On single-slot local servers (LM Studio, Ollama) this is almost always another request holding the server's context at that moment — typically a background memory review from an earlier session (`thread=bg-review` in `logs/agent.log`). Wait a moment and `/retry`. If it recurs with no other Hermes process running, the server is loading the model with a smaller window than Hermes assumes: raise the server's context setting or lower `model.context_length` to match it. + To fix context detection, set it explicitly: ```yaml From 1baae3132a65d38d5b6527cb4202b07f4ea1f166 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 00:55:05 -0700 Subject: [PATCH 27/65] chore: map contributor email for @Shotflame --- contributors/emails/artiefisher123@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/artiefisher123@gmail.com diff --git a/contributors/emails/artiefisher123@gmail.com b/contributors/emails/artiefisher123@gmail.com new file mode 100644 index 0000000000..f7afbbe764 --- /dev/null +++ b/contributors/emails/artiefisher123@gmail.com @@ -0,0 +1 @@ +Shotflame From 9aaeb1f5514fc98973664eb5820cc448901b66db Mon Sep 17 00:00:00 2001 From: thestreamcode Date: Tue, 15 Sep 2026 18:28:24 +0200 Subject: [PATCH 28/65] catalog: add hermes-muse-code (Muse Spark via Muse Code subscription) --- plugin-catalog/hermes-muse-code.yaml | 15 +++++++++++++++ 1 file changed, 15 insertions(+) create mode 100644 plugin-catalog/hermes-muse-code.yaml diff --git a/plugin-catalog/hermes-muse-code.yaml b/plugin-catalog/hermes-muse-code.yaml new file mode 100644 index 0000000000..8be8cf695f --- /dev/null +++ b/plugin-catalog/hermes-muse-code.yaml @@ -0,0 +1,15 @@ +name: hermes-muse-code +repo: https://github.com/TheStreamCode/hermes-muse-code +sha: 07f568d285f97cd856f071222504aaa749847d2b +description: Muse Spark in Hermes billed to the Muse Code monthly login (via omp, no API key). +maintainer: TheStreamCode +tier: community +category: models +requires_hermes: ">=0.21.3" +docs_url: https://github.com/TheStreamCode/hermes-muse-code +platforms: [] +capabilities: + provides_tools: [] + provides_hooks: [] + provides_middleware: [] + requires_env: [] From f3c3c15a85fe2ae5e3d3c28290cd13eb44fcd422 Mon Sep 17 00:00:00 2001 From: Michael Gasperini <163041857+TheStreamCode@users.noreply.github.com> Date: Wed, 16 Sep 2026 20:42:29 +0200 Subject: [PATCH 29/65] catalog: bump hermes-muse-code to v0.2.0 (own device login, no omp) --- plugin-catalog/hermes-muse-code.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/plugin-catalog/hermes-muse-code.yaml b/plugin-catalog/hermes-muse-code.yaml index 8be8cf695f..1dad2b1eb2 100644 --- a/plugin-catalog/hermes-muse-code.yaml +++ b/plugin-catalog/hermes-muse-code.yaml @@ -1,7 +1,7 @@ name: hermes-muse-code repo: https://github.com/TheStreamCode/hermes-muse-code -sha: 07f568d285f97cd856f071222504aaa749847d2b -description: Muse Spark in Hermes billed to the Muse Code monthly login (via omp, no API key). +sha: 1e66d4d923bab9b670846c9005593dc1a71e0d34 +description: Muse Spark in Hermes billed to the Muse Code monthly login (device-code login, no API key). maintainer: TheStreamCode tier: community category: models From da8f0e55753ff4eb6ca2a8011ac8dd4a47d29606 Mon Sep 17 00:00:00 2001 From: Michael Gasperini <163041857+TheStreamCode@users.noreply.github.com> Date: Wed, 16 Sep 2026 20:48:33 +0200 Subject: [PATCH 30/65] catalog: bump hermes-muse-code to 22298cb (poll-tolerant device login) --- plugin-catalog/hermes-muse-code.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/plugin-catalog/hermes-muse-code.yaml b/plugin-catalog/hermes-muse-code.yaml index 1dad2b1eb2..88d3807586 100644 --- a/plugin-catalog/hermes-muse-code.yaml +++ b/plugin-catalog/hermes-muse-code.yaml @@ -1,6 +1,6 @@ name: hermes-muse-code repo: https://github.com/TheStreamCode/hermes-muse-code -sha: 1e66d4d923bab9b670846c9005593dc1a71e0d34 +sha: 22298cbb80d728b52c63babd152ebad1b31767aa description: Muse Spark in Hermes billed to the Muse Code monthly login (device-code login, no API key). maintainer: TheStreamCode tier: community From ffc514d862824776dce2e94978fb30309ebd759b Mon Sep 17 00:00:00 2001 From: Michael Gasperini <163041857+TheStreamCode@users.noreply.github.com> Date: Wed, 16 Sep 2026 20:52:38 +0200 Subject: [PATCH 31/65] catalog: bump hermes-muse-code to e335011 (review fixes) --- plugin-catalog/hermes-muse-code.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/plugin-catalog/hermes-muse-code.yaml b/plugin-catalog/hermes-muse-code.yaml index 88d3807586..ee8b9ee42c 100644 --- a/plugin-catalog/hermes-muse-code.yaml +++ b/plugin-catalog/hermes-muse-code.yaml @@ -1,6 +1,6 @@ name: hermes-muse-code repo: https://github.com/TheStreamCode/hermes-muse-code -sha: 22298cbb80d728b52c63babd152ebad1b31767aa +sha: e33501185e328b5597fd98c6de916495922ec727 description: Muse Spark in Hermes billed to the Muse Code monthly login (device-code login, no API key). maintainer: TheStreamCode tier: community From 1f3cb8bee1bb9d84cfe82e7a66cd94c37afed516 Mon Sep 17 00:00:00 2001 From: teknium Date: Fri, 18 Sep 2026 19:15:48 -0700 Subject: [PATCH 32/65] chore: map contributor email for @TheStreamCode --- contributors/emails/dev@mikesoft.it | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/dev@mikesoft.it diff --git a/contributors/emails/dev@mikesoft.it b/contributors/emails/dev@mikesoft.it new file mode 100644 index 0000000000..dfcbd18b41 --- /dev/null +++ b/contributors/emails/dev@mikesoft.it @@ -0,0 +1 @@ +TheStreamCode From c36681ef681761a644f741ab571e75ff0e8d7b08 Mon Sep 17 00:00:00 2001 From: SmokeDev Date: Fri, 11 Sep 2026 21:01:59 -0700 Subject: [PATCH 33/65] feat(plugin-catalog): add Hermes Talk Signed-off-by: SmokeDev --- plugin-catalog/hermes-talk.yaml | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) create mode 100644 plugin-catalog/hermes-talk.yaml diff --git a/plugin-catalog/hermes-talk.yaml b/plugin-catalog/hermes-talk.yaml new file mode 100644 index 0000000000..49aafb99f0 --- /dev/null +++ b/plugin-catalog/hermes-talk.yaml @@ -0,0 +1,17 @@ +name: hermes-talk +repo: https://github.com/TheSmokeDev/hermes-talk +sha: efebeebfe26f6d1ea9b79dd3e7da80a2f8f418b8 +description: Realtime voice for Hermes through terminal, Discord, and dashboard interfaces. +maintainer: TheSmokeDev +tier: community +docs_url: https://github.com/TheSmokeDev/hermes-talk/blob/efebeebfe26f6d1ea9b79dd3e7da80a2f8f418b8/README.md +capabilities: + provides_tools: [] + provides_hooks: + - on_session_end + - subagent_start + - subagent_stop + - post_tool_call + - pre_approval_request + provides_middleware: [] + requires_env: [] From cf31cccd2d4c9bff8dbe1a1622dfa270f8e2aa89 Mon Sep 17 00:00:00 2001 From: SmokeDev Date: Fri, 11 Sep 2026 21:12:04 -0700 Subject: [PATCH 34/65] fix(plugin-catalog): pin corrected Hermes Talk source Signed-off-by: SmokeDev --- plugin-catalog/hermes-talk.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/plugin-catalog/hermes-talk.yaml b/plugin-catalog/hermes-talk.yaml index 49aafb99f0..5a2d8e206c 100644 --- a/plugin-catalog/hermes-talk.yaml +++ b/plugin-catalog/hermes-talk.yaml @@ -1,10 +1,10 @@ name: hermes-talk repo: https://github.com/TheSmokeDev/hermes-talk -sha: efebeebfe26f6d1ea9b79dd3e7da80a2f8f418b8 +sha: 8aeac6a6e1bf6f95dd3b9ed6ee53389883746253 description: Realtime voice for Hermes through terminal, Discord, and dashboard interfaces. maintainer: TheSmokeDev tier: community -docs_url: https://github.com/TheSmokeDev/hermes-talk/blob/efebeebfe26f6d1ea9b79dd3e7da80a2f8f418b8/README.md +docs_url: https://github.com/TheSmokeDev/hermes-talk/blob/8aeac6a6e1bf6f95dd3b9ed6ee53389883746253/README.md capabilities: provides_tools: [] provides_hooks: From dbb9dcc200c1ceb6910692aac91c184976476b23 Mon Sep 17 00:00:00 2001 From: SmokeDev Date: Mon, 14 Sep 2026 23:42:35 -0700 Subject: [PATCH 35/65] catalog: pin hermes-talk to v0.19.1 (read-only Codex lane) Moves the pin from 8aeac6a6 to 4764f687, the peeled commit of tag v0.19.1. The only source change on the auth surface is #146 (read-only Codex lane, authored by @teknium1), merged as-is. plugin.yaml changed only its version field; hook registrations, tools, middleware and required env are unchanged against the declared capabilities block. Signed-off-by: SmokeDev --- plugin-catalog/hermes-talk.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/plugin-catalog/hermes-talk.yaml b/plugin-catalog/hermes-talk.yaml index 5a2d8e206c..0cde96bcaa 100644 --- a/plugin-catalog/hermes-talk.yaml +++ b/plugin-catalog/hermes-talk.yaml @@ -1,10 +1,10 @@ name: hermes-talk repo: https://github.com/TheSmokeDev/hermes-talk -sha: 8aeac6a6e1bf6f95dd3b9ed6ee53389883746253 +sha: 4764f6876521caeb8cbc5c83244e87cd8390096e description: Realtime voice for Hermes through terminal, Discord, and dashboard interfaces. maintainer: TheSmokeDev tier: community -docs_url: https://github.com/TheSmokeDev/hermes-talk/blob/8aeac6a6e1bf6f95dd3b9ed6ee53389883746253/README.md +docs_url: https://github.com/TheSmokeDev/hermes-talk/blob/4764f6876521caeb8cbc5c83244e87cd8390096e/README.md capabilities: provides_tools: [] provides_hooks: From e8e712e7dc158e43e1df7f855ff51508f23b3418 Mon Sep 17 00:00:00 2001 From: SmokeDev Date: Tue, 15 Sep 2026 08:23:56 -0700 Subject: [PATCH 36/65] feat(plugins): pin hermes-talk catalog entry to v0.19.2 Moves sha and docs_url to a4843d82c0b2c8443830b79f5e0a19b8a78a8303 (v0.19.2, peeled commit). Capabilities block unchanged and re-verified at that commit. Signed-off-by: SmokeDev --- plugin-catalog/hermes-talk.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/plugin-catalog/hermes-talk.yaml b/plugin-catalog/hermes-talk.yaml index 0cde96bcaa..d05ceeb16e 100644 --- a/plugin-catalog/hermes-talk.yaml +++ b/plugin-catalog/hermes-talk.yaml @@ -1,10 +1,10 @@ name: hermes-talk repo: https://github.com/TheSmokeDev/hermes-talk -sha: 4764f6876521caeb8cbc5c83244e87cd8390096e +sha: a4843d82c0b2c8443830b79f5e0a19b8a78a8303 description: Realtime voice for Hermes through terminal, Discord, and dashboard interfaces. maintainer: TheSmokeDev tier: community -docs_url: https://github.com/TheSmokeDev/hermes-talk/blob/4764f6876521caeb8cbc5c83244e87cd8390096e/README.md +docs_url: https://github.com/TheSmokeDev/hermes-talk/blob/a4843d82c0b2c8443830b79f5e0a19b8a78a8303/README.md capabilities: provides_tools: [] provides_hooks: From a3baf5135dcfcf5c93168eaea2fb65d8b7231aad Mon Sep 17 00:00:00 2001 From: TheSmokeDev Date: Fri, 18 Sep 2026 17:20:28 -0700 Subject: [PATCH 37/65] plugin-catalog: bump hermes-talk pin to v0.21.0, add voice category --- plugin-catalog/hermes-talk.yaml | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/plugin-catalog/hermes-talk.yaml b/plugin-catalog/hermes-talk.yaml index d05ceeb16e..7a0b110ea3 100644 --- a/plugin-catalog/hermes-talk.yaml +++ b/plugin-catalog/hermes-talk.yaml @@ -1,10 +1,12 @@ name: hermes-talk repo: https://github.com/TheSmokeDev/hermes-talk -sha: a4843d82c0b2c8443830b79f5e0a19b8a78a8303 +sha: 7219f2fdff353df72f5323b219bfe759991b0365 description: Realtime voice for Hermes through terminal, Discord, and dashboard interfaces. maintainer: TheSmokeDev tier: community -docs_url: https://github.com/TheSmokeDev/hermes-talk/blob/a4843d82c0b2c8443830b79f5e0a19b8a78a8303/README.md +category: voice +version: "0.21.0" +docs_url: https://github.com/TheSmokeDev/hermes-talk/blob/7219f2fdff353df72f5323b219bfe759991b0365/README.md capabilities: provides_tools: [] provides_hooks: From 1af1db281e7274581cb61fdd03c64f822fbebb83 Mon Sep 17 00:00:00 2001 From: teknium Date: Fri, 18 Sep 2026 19:31:23 -0700 Subject: [PATCH 38/65] plugin-catalog: hermes-talk disclosure line for the Codex realtime lane + contributor map --- plugin-catalog/hermes-talk.yaml | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/plugin-catalog/hermes-talk.yaml b/plugin-catalog/hermes-talk.yaml index 7a0b110ea3..c6a5bc8039 100644 --- a/plugin-catalog/hermes-talk.yaml +++ b/plugin-catalog/hermes-talk.yaml @@ -1,7 +1,12 @@ name: hermes-talk repo: https://github.com/TheSmokeDev/hermes-talk sha: 7219f2fdff353df72f5323b219bfe759991b0365 -description: Realtime voice for Hermes through terminal, Discord, and dashboard interfaces. +description: >- + Realtime voice for Hermes through terminal, Discord, dashboard and Desktop. + Live voice mode (TALK_VOICE_MODE=live, TALK_LIVE_AUTH=subscription) connects with your + own Codex OAuth session to the undocumented chatgpt.com/backend-api/codex/realtime/calls + endpoint, which is outside the vendor's published API terms and may change or stop + working without notice. maintainer: TheSmokeDev tier: community category: voice From 0ddba07ad701d55f40212ff624ba85dd8ecda7af Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 19:33:52 -0700 Subject: [PATCH 39/65] fix(ci): report an interpreter crash as CRASHED, not "no tests ran" MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When a per-file pytest subprocess dies by signal (the sqlite cross-thread close in #113186 was a SIGSEGV after every test had passed), faulthandler prints "Fatal Python error: Segmentation fault" and no summary line, so every count parses to 0. The runner filed that under "1 file where no tests ran (collection/import error, ...)" beneath a summary that read "0 failed" and exited 1 — two wrong diagnoses for one real bug, and it was misread as a runner problem twice on main. The runner now detects a signal death or a "Fatal Python error:" banner, prefixes the captured output with the diagnosis (same convention as the timeout path), marks the progress line CRASHED, counts "N files CRASHED" on the summary line, lists the file in its own failure bucket, and no longer trips the "NO TESTS RAN" guard for a crash that ran tests. The flake retry already covers crashes (any non-zero rc), so nothing changes there. --- scripts/run_tests_parallel.py | 63 +++++++++++++++++++++--- tests/scripts/test_run_tests_parallel.py | 35 +++++++++++++ 2 files changed, 92 insertions(+), 6 deletions(-) diff --git a/scripts/run_tests_parallel.py b/scripts/run_tests_parallel.py index 9c461f8882..81424bb764 100755 --- a/scripts/run_tests_parallel.py +++ b/scripts/run_tests_parallel.py @@ -522,6 +522,12 @@ def _run_one_file_once( # (venv without pytest, -k that matches nothing) can't report green. rc = 0 summary = _parse_pytest_summary(output) + crash = _describe_interpreter_crash(rc, output) if rc != 0 else None + if crash: + # Same convention as the timeout path: the diagnosis leads the + # captured output, so the failure dump reads correctly on its own. + summary["crashed"] = 1 + output = f"(interpreter crashed: {crash})\n{output}" subproc_wall = time.monotonic() - subproc_start return file, rc, output, summary, subproc_wall @@ -559,6 +565,35 @@ def _parse_pytest_summary(output: str) -> dict[str, int]: return result +def _describe_interpreter_crash(rc: int, output: str) -> Optional[str]: + """Return a one-line description when the pytest subprocess died instead of exiting. + + A native fault (sqlite stepping a connection another thread closed, + #113186) kills the interpreter mid-file: faulthandler prints ``Fatal + Python error: Segmentation fault`` and the process dies by signal, so + there is no summary line and every count parses to 0. Without this the + file is reported as "no tests ran (collection/import error)" under a + summary that says ``0 failed`` — the wrong diagnosis in both places. + """ + fatal = next( + (line.strip() for line in output.splitlines() if "Fatal Python error:" in line), + None, + ) + if fatal: + # The faulthandler banner is appended to the progress dots of the + # last test; keep only the banner. + fatal = fatal[fatal.index("Fatal Python error:"):] + if rc < 0: + import signal as _signal + + try: + name = _signal.Signals(-rc).name + except ValueError: + name = f"signal {-rc}" + return f"{fatal} ({name})" if fatal else f"killed by {name}" + return fatal + + def _format_file(file: Path, repo_root: Path) -> str: """Render a test-file path for display: strip the repo-root prefix when possible so output reads ``tests/acp_adapter/test_auth.py`` instead of @@ -621,6 +656,8 @@ def _print_progress( parts.append(f"{xf}xf") if xp: parts.append(f"{xp}xp") + if file_summary.get("crashed"): + parts.append("CRASHED") test_str = " ".join(parts) + ", " if parts else "" else: n_tests = test_counts.get(file, 0) @@ -1165,11 +1202,12 @@ def main() -> int: # nothing-ran guard, whereas a file that died before collection reports # nothing at all and must. tests_collected = 0 + files_crashed = 0 lock = threading.Lock() def _on_done(file: Path, started_at: float, fut: "Future[Tuple[Path, int, str, Dict[str, int], float]]") -> None: nonlocal files_done, tests_done, pass_count, fail_count, tests_passed, tests_failed, tests_skipped - nonlocal tests_collected + nonlocal tests_collected, files_crashed n_tests = test_counts.get(file, 0) try: fpath, rc, output, summary, subproc_wall = fut.result() @@ -1194,6 +1232,7 @@ def main() -> int: tests_passed += summary.get("passed", 0) tests_failed += summary.get("failed", 0) tests_skipped += summary.get("skipped", 0) + files_crashed += summary.get("crashed", 0) tests_collected += sum( summary.get(k, 0) for k in ("passed", "failed", "skipped", "errors", "xfailed", "xpassed") @@ -1242,7 +1281,13 @@ def main() -> int: print() pct = min(100, (tests_done / approx_total_tests * 100)) if approx_total_tests else 0 skipped_note = f", {tests_skipped} skipped" if tests_skipped else "" - print(f"=== Summary: {len(files)} files, {tests_passed} tests passed, {tests_failed} failed{skipped_note} ({pct:.0f}% complete) in {elapsed:.1f}s ({args.jobs} workers) ===") + # A crashed interpreter has no failed-test count; say so on the one line + # everyone reads, or "0 failed" + exit 1 looks like a runner bug. + crashed_note = ( + f", {files_crashed} file{'s' if files_crashed != 1 else ''} CRASHED" + if files_crashed else "" + ) + print(f"=== Summary: {len(files)} files, {tests_passed} tests passed, {tests_failed} failed{crashed_note}{skipped_note} ({pct:.0f}% complete) in {elapsed:.1f}s ({args.jobs} workers) ===") # Host-OS gating note: tests marked for another OS were skipped by the # conftest hook, not run. Say so explicitly — a green local run on Linux @@ -1265,7 +1310,7 @@ def main() -> int: # The summary line above reads green at a glance ("0 failed ... 100% # complete"), which has been misread as a successful verification, so say # it plainly AND fail the exit code. - no_tests_ran_at_all = bool(files) and tests_collected == 0 + no_tests_ran_at_all = bool(files) and tests_collected == 0 and not files_crashed if no_tests_ran_at_all: print() print( @@ -1333,11 +1378,17 @@ def main() -> int: print(output.rstrip()) print() # Split: files with actual test failures vs non-zero exit for other reasons - test_fail_files = [(f, s) for f, _o, s in failures if s.get("failed", 0) > 0] - all_passed_but_nonzero = [(f, s) for f, _o, s in failures + crashed_files = [(f, o, s) for f, o, s in failures if s.get("crashed")] + rest = [(f, s) for f, _o, s in failures if not s.get("crashed")] + test_fail_files = [(f, s) for f, s in rest if s.get("failed", 0) > 0] + all_passed_but_nonzero = [(f, s) for f, s in rest if s.get("failed", 0) == 0 and s.get("passed", 0) > 0] - no_tests_ran = [(f, s) for f, _o, s in failures + no_tests_ran = [(f, s) for f, s in rest if s.get("failed", 0) == 0 and s.get("passed", 0) == 0] + if crashed_files: + print(f"=== {len(crashed_files)} file{'s' if len(crashed_files) != 1 else ''} where the interpreter CRASHED mid-run (native fault — a real bug, not a collection error; the tests that did run are not counted) ===") + for file, output, _s in crashed_files: + print(f" {_format_file(file, repo_root)} {output.splitlines()[0]}") if test_fail_files: total_tf = sum(s.get("failed", 0) for _, s in test_fail_files) print(f"=== {len(test_fail_files)} file{'s' if len(test_fail_files) != 1 else ''} with test failures ({total_tf} test{'s' if total_tf != 1 else ''} failed) ===") diff --git a/tests/scripts/test_run_tests_parallel.py b/tests/scripts/test_run_tests_parallel.py index 14450ecfbe..56195b77f0 100644 --- a/tests/scripts/test_run_tests_parallel.py +++ b/tests/scripts/test_run_tests_parallel.py @@ -516,3 +516,38 @@ def test_drive_letter_colon_is_not_a_path_separator(tmp_path: Path) -> None: f"drive letter split off as a phantom root:\n{proc.stdout}" ) assert "Discovered 1 test files" in proc.stdout, proc.stdout + + +@pytest.mark.skipif(sys.platform == "win32", reason="POSIX signal death; Windows has no SIGSEGV exit") +def test_interpreter_crash_is_reported_as_a_crash_not_as_no_tests_ran(tmp_path: Path) -> None: + """A file whose interpreter dies by signal is classified as CRASHED (#113186). + + A native fault after some tests passed leaves no pytest summary line, so + every count parses to 0. The runner used to file that under "no tests ran + (collection/import error)" beneath a summary reading ``0 failed`` — two + wrong diagnoses for one real bug. The crash must be named on the summary + line and in the failure buckets, and the run must still exit non-zero. + """ + probe_dir = tmp_path / "probe" + probe_dir.mkdir() + (probe_dir / "test_probe_crash.py").write_text( + textwrap.dedent( + """ + import os, signal + + def test_before(): + assert True + + def test_crash(): + os.kill(os.getpid(), signal.SIGSEGV) + """ + ) + ) + + proc = _run_runner(probe_dir, "--file-retries", "0") + + assert proc.returncode != 0 + assert "1 file CRASHED" in proc.stdout + assert "SIGSEGV" in proc.stdout + assert "where no tests ran" not in proc.stdout + assert "NO TESTS RAN" not in proc.stdout From 98ad0d8148f3fa11b3f48995ada6ba668027d0db Mon Sep 17 00:00:00 2001 From: Yagna Vudathu Date: Thu, 17 Sep 2026 14:25:53 -0400 Subject: [PATCH 40/65] fix(image-gen): record token-billed OpenRouter calls to session usage OpenRouter token-billed image calls parsed usage into extra only, so session_model_usage never saw them. Record prompt/completion tokens via the ambient aux accounting on success; flat-fee responses without token usage write nothing. Fixes #114324. --- plugins/image_gen/openrouter/__init__.py | 25 +++++++++ .../test_openrouter_compat_provider.py | 53 +++++++++++++++++++ 2 files changed, 78 insertions(+) diff --git a/plugins/image_gen/openrouter/__init__.py b/plugins/image_gen/openrouter/__init__.py index 0d1e6a237c..3a607d1a38 100644 --- a/plugins/image_gen/openrouter/__init__.py +++ b/plugins/image_gen/openrouter/__init__.py @@ -692,6 +692,7 @@ class OpenRouterCompatImageProvider(ImageGenProvider): return _fail(f"Could not save generated image: {exc}", "io_error") if not saved: return _fail(f"{self._display} response carried neither b64_json nor url.", "empty_response") + _record_image_api_usage(body, model_id, provider=self._name) return success_response( image=saved[0], model=model_id, prompt=prompt, aspect_ratio=semantic_aspect, provider=self._name, modality="image" if usable_refs else "text", @@ -802,6 +803,30 @@ class OpenRouterCompatImageProvider(ImageGenProvider): model=model_chain[-1] if model_chain else "", prompt=prompt, aspect_ratio=aspect) +def _record_image_api_usage(body: Any, model_id: str, *, provider: str) -> None: + """Record a token-billed Image API call against the ambient session (#114324). + + Best-effort: accounting must never break image generation. No-ops outside + an agent turn or when the response carries no token usage. + """ + try: + usage = _dict_at(body, "usage") + if not isinstance(usage.get("total_tokens"), int): + return + from agent.aux_accounting import record_aux_usage + from types import SimpleNamespace + + record_aux_usage( + SimpleNamespace(model=model_id, usage=SimpleNamespace( + prompt_tokens=usage.get("prompt_tokens", 0), + completion_tokens=usage.get("completion_tokens", 0), + total_tokens=usage.get("total_tokens", 0), + )), + "image_gen", provider=provider) + except Exception: # noqa: BLE001 + logger.debug("%s: image usage recording failed (non-fatal)", provider, exc_info=True) + + def _build_providers() -> List[OpenRouterCompatImageProvider]: return [ OpenRouterCompatImageProvider( diff --git a/tests/plugins/image_gen/test_openrouter_compat_provider.py b/tests/plugins/image_gen/test_openrouter_compat_provider.py index b3647376b9..5e36083325 100644 --- a/tests/plugins/image_gen/test_openrouter_compat_provider.py +++ b/tests/plugins/image_gen/test_openrouter_compat_provider.py @@ -703,6 +703,59 @@ class TestImageApiSurface: assert result["exact_aspect_ratio"] == "9:16" assert result["image"] == "/tmp/i.png" + def test_token_billed_call_records_session_usage(self): + """#114324: an OpenRouter token-billed call must reach session_model_usage.""" + from agent import aux_accounting + + recorded = [] + + class _DB: + def record_auxiliary_usage(self, *args, **kwargs): + recorded.append(kwargs) + + token = aux_accounting.set_accounting_context(_DB(), "sess-1") + try: + with patch(_RUNTIME, return_value=_runtime_ok()), \ + patch("requests.post", return_value=_mock_image_api_response( + usage={"total_tokens": 1128, "prompt_tokens": 1000, + "completion_tokens": 128})), \ + patch("plugins.image_gen.openrouter.save_b64_image", return_value=Path("/tmp/i.png")): + result = _openrouter_image_api().generate( + prompt="p", aspect_ratio="portrait", model="krea/krea-2-medium" + ) + finally: + aux_accounting.reset_accounting_context(token) + + assert result["success"] is True + assert len(recorded) == 1 + assert recorded[0]["input_tokens"] == 1000 + assert recorded[0]["output_tokens"] == 128 + assert recorded[0]["model"] == "krea/krea-2-medium" + + def test_untokened_call_records_nothing(self): + """No token usage in the body: no session write (flat-fee image models).""" + from agent import aux_accounting + + recorded = [] + + class _DB: + def record_auxiliary_usage(self, *args, **kwargs): + recorded.append(kwargs) + + token = aux_accounting.set_accounting_context(_DB(), "sess-1") + try: + with patch(_RUNTIME, return_value=_runtime_ok()), \ + patch("requests.post", return_value=_mock_image_api_response()), \ + patch("plugins.image_gen.openrouter.save_b64_image", return_value=Path("/tmp/i.png")): + result = _openrouter_image_api().generate( + prompt="p", aspect_ratio="portrait", model="krea/krea-2-medium" + ) + finally: + aux_accounting.reset_accounting_context(token) + + assert result["success"] is True + assert recorded == [] + def test_multiple_images_land_in_additional_images(self): entries = [ {"b64_json": "AA==", "media_type": "image/png"}, From 6babdc96b800beb3726078eced761d9946bb4c2f Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 00:52:21 -0700 Subject: [PATCH 41/65] fix(image-gen): every token-billed image backend records session usage (chat path, Image API, OpenAI gpt-image) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Widen the salvaged Image API hunk (#114340) to the whole class: - plugins/image_gen/_common.py: record_token_usage() — one helper feeding the aux accounting chokepoint (agent.aux_accounting.record_aux_usage) with task "image_generation", the billing provider and the priced model id. Dict and SDK usage objects alike; a body without tokens is a no-op, as is a call outside a turn. - plugins/image_gen/openrouter: the /chat/completions path (the DEFAULT model chain — openai/gpt-5.4-image-2, google/gemini-3-pro-image — is chat-only and token-billed) now records too; the Image API path uses the shared helper (the contributor's local _record_image_api_usage is folded into it) and both pass base_url so pricing resolves the route. Task renamed image_gen -> image_generation to match the other aux task names. - plugins/image_gen/openai: gpt-image bills per text/image token; record the Images API usage block under the API model (gpt-image-2), not the Hermes quality-tier label. - FAL, xAI, Krea, DeepInfra, Meta, openai-codex return no token usage and stay unrecorded (nothing to bill per token). - tests: the contributor's two tests folded into one invariant parametrized over chat / Image API / no-usage control; one OpenAI invariant. - docs: image-generation.md "How It Works Internally" gains the accounting step. Fixes #114324 --- plugins/image_gen/_common.py | 19 ++++++ plugins/image_gen/openai/__init__.py | 4 +- plugins/image_gen/openrouter/__init__.py | 29 +-------- .../plugins/image_gen/test_openai_provider.py | 28 +++++++++ .../test_openrouter_compat_provider.py | 62 ++++++++----------- .../user-guide/features/image-generation.md | 1 + 6 files changed, 80 insertions(+), 63 deletions(-) diff --git a/plugins/image_gen/_common.py b/plugins/image_gen/_common.py index efc8b8f4ad..aea9d21cda 100644 --- a/plugins/image_gen/_common.py +++ b/plugins/image_gen/_common.py @@ -242,6 +242,25 @@ class HttpFailure: response: Any = None +def record_token_usage(usage: Any, *, model: str, provider: str, base_url: Optional[str] = None) -> None: + """Record a token-billed image call against the ambient session as task ``image_generation``. + + Token-metered image models (OpenRouter chat-image and Image API models, OpenAI ``gpt-image``) + bill exactly like a chat completion, so they get a ``session_model_usage`` row through the + same chokepoint as auxiliary calls. ``usage`` is the response's usage block (dict or SDK + object). Per-image backends (FAL, xAI, Krea, ...) return no token usage and never call this; + a body without tokens is a no-op inside ``record_aux_usage``, as is running outside a turn. + """ + if not usage: + return + from types import SimpleNamespace + + from agent.aux_accounting import record_aux_usage + + record_aux_usage( + SimpleNamespace(model=model, usage=usage), "image_generation", provider=provider, base_url=base_url) + + def post_json( url: str, *, headers: Dict[str, str], payload: Dict[str, Any], timeout: Any, label: str, error_message: Callable[[Any, Exception], str] = requests_error_message, diff --git a/plugins/image_gen/openai/__init__.py b/plugins/image_gen/openai/__init__.py index 2c0742ee8a..260412c758 100644 --- a/plugins/image_gen/openai/__init__.py +++ b/plugins/image_gen/openai/__init__.py @@ -14,7 +14,7 @@ from agent.image_gen_provider import DEFAULT_ASPECT_RATIO, resolve_aspect_ratio, from plugins.image_gen._common import ( GPT_IMAGE_2_API_MODEL as API_MODEL, GPT_IMAGE_2_DEFAULT as DEFAULT_MODEL, GPT_IMAGE_2_TIERS, StaticImageGenProvider, collect_source_images, error_factory, import_openai, materialize_image, - openai_importable, prompt_required_error, resolve_static_model, size_for) + openai_importable, prompt_required_error, record_token_usage, resolve_static_model, size_for) logger = logging.getLogger(__name__) @@ -153,6 +153,8 @@ class OpenAIImageGenProvider(StaticImageGenProvider): model=tier_id, prompt=prompt, aspect=aspect, log=logger) if err: return err + # gpt-image bills per text/image token; the tier id is a Hermes label, the API model prices. + record_token_usage(getattr(response, "usage", None), model=meta["api_model"], provider="openai") extra: Dict[str, Any] = {"size": size, "quality": meta["quality"]} if getattr(first, "revised_prompt", None): extra["revised_prompt"] = first.revised_prompt diff --git a/plugins/image_gen/openrouter/__init__.py b/plugins/image_gen/openrouter/__init__.py index 3a607d1a38..b7b4eb9a62 100644 --- a/plugins/image_gen/openrouter/__init__.py +++ b/plugins/image_gen/openrouter/__init__.py @@ -22,7 +22,7 @@ from typing import Any, Dict, List, Optional, Tuple from agent.image_gen_provider import ( DEFAULT_ASPECT_RATIO, ImageGenProvider, error_response, resolve_aspect_ratio, save_b64_image, save_url_image, success_response) -from plugins.image_gen._common import error_factory, load_image_gen_config, post_json +from plugins.image_gen._common import error_factory, load_image_gen_config, post_json, record_token_usage logger = logging.getLogger(__name__) @@ -692,7 +692,7 @@ class OpenRouterCompatImageProvider(ImageGenProvider): return _fail(f"Could not save generated image: {exc}", "io_error") if not saved: return _fail(f"{self._display} response carried neither b64_json nor url.", "empty_response") - _record_image_api_usage(body, model_id, provider=self._name) + record_token_usage(_dict_at(body, "usage"), model=model_id, provider=self._name, base_url=base_url) return success_response( image=saved[0], model=model_id, prompt=prompt, aspect_ratio=semantic_aspect, provider=self._name, modality="image" if usable_refs else "text", @@ -741,6 +741,7 @@ class OpenRouterCompatImageProvider(ImageGenProvider): saved_path = save_url_image(first, prefix=f"{self._name}_gen") except Exception as exc: # noqa: BLE001 return fail(f"Could not save generated image: {exc}", "io_error"), None + record_token_usage(_dict_at(result, "usage"), model=model_id, provider=self._name, base_url=base_url) return success_response( image=str(saved_path), model=model_id, prompt=prompt, aspect_ratio=aspect, provider=self._name, ), None @@ -803,30 +804,6 @@ class OpenRouterCompatImageProvider(ImageGenProvider): model=model_chain[-1] if model_chain else "", prompt=prompt, aspect_ratio=aspect) -def _record_image_api_usage(body: Any, model_id: str, *, provider: str) -> None: - """Record a token-billed Image API call against the ambient session (#114324). - - Best-effort: accounting must never break image generation. No-ops outside - an agent turn or when the response carries no token usage. - """ - try: - usage = _dict_at(body, "usage") - if not isinstance(usage.get("total_tokens"), int): - return - from agent.aux_accounting import record_aux_usage - from types import SimpleNamespace - - record_aux_usage( - SimpleNamespace(model=model_id, usage=SimpleNamespace( - prompt_tokens=usage.get("prompt_tokens", 0), - completion_tokens=usage.get("completion_tokens", 0), - total_tokens=usage.get("total_tokens", 0), - )), - "image_gen", provider=provider) - except Exception: # noqa: BLE001 - logger.debug("%s: image usage recording failed (non-fatal)", provider, exc_info=True) - - def _build_providers() -> List[OpenRouterCompatImageProvider]: return [ OpenRouterCompatImageProvider( diff --git a/tests/plugins/image_gen/test_openai_provider.py b/tests/plugins/image_gen/test_openai_provider.py index 93ad401321..9a315cbb07 100644 --- a/tests/plugins/image_gen/test_openai_provider.py +++ b/tests/plugins/image_gen/test_openai_provider.py @@ -172,6 +172,34 @@ class TestGenerate: # gpt-image-2 rejects response_format — we must NOT send it. assert "response_format" not in call_kwargs + def test_token_usage_reaches_session_accounting(self, provider): + """gpt-image bills per token: the Images API ``usage`` block lands as one + ``image_generation`` row keyed on the API model, not the Hermes tier label.""" + from agent import aux_accounting + + recorded = [] + + class _DB: + def record_auxiliary_usage(self, *args, **kwargs): + recorded.append((args, kwargs)) + + response = _fake_response(b64=_b64_png()) + response.usage = SimpleNamespace(input_tokens=23, output_tokens=1056, total_tokens=1079) + fake_client = MagicMock() + fake_client.images.generate.return_value = response + token = aux_accounting.set_accounting_context(_DB(), "sess-1") + try: + with _patched_openai(fake_client): + result = provider.generate("a cat", aspect_ratio="landscape") + finally: + aux_accounting.reset_accounting_context(token) + + assert result["success"] is True + ((session_id, task), kwargs), = recorded + assert (session_id, task) == ("sess-1", "image_generation") + assert (kwargs["model"], kwargs["billing_provider"]) == ("gpt-image-2", "openai") + assert (kwargs["input_tokens"], kwargs["output_tokens"]) == (23, 1056) + @pytest.mark.parametrize("api_model,quality", [ ("gpt-image-2", quality) for quality in ("low", "medium", "high") ] + [ diff --git a/tests/plugins/image_gen/test_openrouter_compat_provider.py b/tests/plugins/image_gen/test_openrouter_compat_provider.py index 5e36083325..ec6a1aa5a0 100644 --- a/tests/plugins/image_gen/test_openrouter_compat_provider.py +++ b/tests/plugins/image_gen/test_openrouter_compat_provider.py @@ -703,58 +703,48 @@ class TestImageApiSurface: assert result["exact_aspect_ratio"] == "9:16" assert result["image"] == "/tmp/i.png" - def test_token_billed_call_records_session_usage(self): - """#114324: an OpenRouter token-billed call must reach session_model_usage.""" + _USAGE = {"prompt_tokens": 1000, "completion_tokens": 128, "total_tokens": 1128} + + @pytest.mark.parametrize("surface, model, usage", [ + ("chat", "openai/gpt-5.4-image-2", _USAGE), # default chain: token-billed via /chat/completions + ("images", "krea/krea-2-medium", _USAGE), # curated Image API model + ("images", "krea/krea-2-medium", None), # flat-fee body without usage: no write + ]) + def test_token_usage_reaches_session_accounting(self, surface, model, usage): + """A response carrying token usage records one ``image_generation`` row on the ambient + session; a body without usage records nothing.""" from agent import aux_accounting recorded = [] class _DB: def record_auxiliary_usage(self, *args, **kwargs): - recorded.append(kwargs) + recorded.append((args, kwargs)) + if surface == "chat": + response = _mock_chat_response([_PNG_DATA_URI]) + response.json.return_value["usage"] = dict(usage) + else: + response = _mock_image_api_response(usage=usage) token = aux_accounting.set_accounting_context(_DB(), "sess-1") try: with patch(_RUNTIME, return_value=_runtime_ok()), \ - patch("requests.post", return_value=_mock_image_api_response( - usage={"total_tokens": 1128, "prompt_tokens": 1000, - "completion_tokens": 128})), \ + patch("requests.post", return_value=response), \ patch("plugins.image_gen.openrouter.save_b64_image", return_value=Path("/tmp/i.png")): - result = _openrouter_image_api().generate( - prompt="p", aspect_ratio="portrait", model="krea/krea-2-medium" - ) + result = _openrouter_image_api().generate(prompt="p", aspect_ratio="portrait", model=model) finally: aux_accounting.reset_accounting_context(token) assert result["success"] is True - assert len(recorded) == 1 - assert recorded[0]["input_tokens"] == 1000 - assert recorded[0]["output_tokens"] == 128 - assert recorded[0]["model"] == "krea/krea-2-medium" + if usage is None: + assert recorded == [] + return + ((session_id, task), kwargs), = recorded + assert (session_id, task) == ("sess-1", "image_generation") + assert kwargs["model"] == model + assert kwargs["billing_provider"] == "openrouter" + assert (kwargs["input_tokens"], kwargs["output_tokens"]) == (1000, 128) - def test_untokened_call_records_nothing(self): - """No token usage in the body: no session write (flat-fee image models).""" - from agent import aux_accounting - - recorded = [] - - class _DB: - def record_auxiliary_usage(self, *args, **kwargs): - recorded.append(kwargs) - - token = aux_accounting.set_accounting_context(_DB(), "sess-1") - try: - with patch(_RUNTIME, return_value=_runtime_ok()), \ - patch("requests.post", return_value=_mock_image_api_response()), \ - patch("plugins.image_gen.openrouter.save_b64_image", return_value=Path("/tmp/i.png")): - result = _openrouter_image_api().generate( - prompt="p", aspect_ratio="portrait", model="krea/krea-2-medium" - ) - finally: - aux_accounting.reset_accounting_context(token) - - assert result["success"] is True - assert recorded == [] def test_multiple_images_land_in_additional_images(self): entries = [ diff --git a/website/docs/user-guide/features/image-generation.md b/website/docs/user-guide/features/image-generation.md index f184308054..f3390d6bb7 100644 --- a/website/docs/user-guide/features/image-generation.md +++ b/website/docs/user-guide/features/image-generation.md @@ -319,6 +319,7 @@ If upscaling fails (network issue, rate limit), the original image is returned a 3. **Submission** — `_submit_fal_request()` routes via direct FAL credentials or the managed Nous gateway, according to the stored `image_gen.provider` selection. 4. **Upscaling** — runs only when the agent passed `upscale: true`; every model's catalog default is off. 5. **Delivery** — final image URL returned to the agent, which emits a `MEDIA:` tag that platform adapters convert to native media. +6. **Usage accounting** — token-billed image models (OpenRouter chat-image and Image API models such as `google/gemini-3.1-flash-lite-image`, OpenAI `gpt-image`) return real token counts, so each call is recorded in `session_model_usage` as task `image_generation` under the billing provider and model, and shows up in `hermes insights` and the dashboard's Usage analytics alongside other model calls. Per-image backends (FAL, xAI, Krea, ...) return no token usage and are not recorded there. ## Debugging From 255b4fd9da3113919753863c9dc42349bfdc76f5 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 03:29:27 -0700 Subject: [PATCH 42/65] fix(image-gen): record token usage as soon as the billed HTTP 200 lands OpenRouter (_generate_via_chat, _generate_via_image_api) and OpenAI gpt-image called record_token_usage only after image extraction and save succeeded. A token-billed 200 that returned text but no image (the "returned no image" fallback case), an empty `data` list, or a failed save therefore consumed billed tokens that never reached session_model_usage; on the fallback chain only the fallback model's tokens were recorded. Move the call to immediately after a successful post_json / SDK call, before extraction/save, in all three paths. One call per HTTP success, so no double count; failure responses (non-2xx, timeouts) still record nothing because there is no body to read usage from. Tests: the parametrized invariant in test_openrouter_compat_provider gains the no-image-with-usage case for both surfaces (result is `empty_response`, one row still recorded); the OpenAI test is parametrized on has_image the same way. --- plugins/image_gen/openai/__init__.py | 5 ++-- plugins/image_gen/openrouter/__init__.py | 6 +++-- .../plugins/image_gen/test_openai_provider.py | 12 +++++++--- .../test_openrouter_compat_provider.py | 23 +++++++++++-------- 4 files changed, 30 insertions(+), 16 deletions(-) diff --git a/plugins/image_gen/openai/__init__.py b/plugins/image_gen/openai/__init__.py index 260412c758..be013eebbf 100644 --- a/plugins/image_gen/openai/__init__.py +++ b/plugins/image_gen/openai/__init__.py @@ -143,6 +143,9 @@ class OpenAIImageGenProvider(StaticImageGenProvider): logger.debug("OpenAI image %s failed", verb, exc_info=True) return fail(f"OpenAI image {'editing' if is_edit else 'generation'} failed: {exc}", "api_error") + # gpt-image bills per text/image token; the tier id is a Hermes label, the API model prices. + # Recorded before extraction/save: the tokens are billed whether or not an image came back. + record_token_usage(getattr(response, "usage", None), model=meta["api_model"], provider="openai") data = getattr(response, "data", None) or [] if not data: return fail("OpenAI returned no image data", "empty_response") @@ -153,8 +156,6 @@ class OpenAIImageGenProvider(StaticImageGenProvider): model=tier_id, prompt=prompt, aspect=aspect, log=logger) if err: return err - # gpt-image bills per text/image token; the tier id is a Hermes label, the API model prices. - record_token_usage(getattr(response, "usage", None), model=meta["api_model"], provider="openai") extra: Dict[str, Any] = {"size": size, "quality": meta["quality"]} if getattr(first, "revised_prompt", None): extra["revised_prompt"] = first.revised_prompt diff --git a/plugins/image_gen/openrouter/__init__.py b/plugins/image_gen/openrouter/__init__.py index b7b4eb9a62..7b26a7c999 100644 --- a/plugins/image_gen/openrouter/__init__.py +++ b/plugins/image_gen/openrouter/__init__.py @@ -680,6 +680,8 @@ class OpenRouterCompatImageProvider(ImageGenProvider): "model_access", retryable=True) return _fail(failure.error, "api_error", retryable=status in _IMAGE_API_FALLBACK_STATUSES) + # The provider billed these tokens on HTTP 200 whether or not an image came back / saved. + record_token_usage(_dict_at(body, "usage"), model=model_id, provider=self._name, base_url=base_url) entries = [e for e in _list_at(body, "data") if isinstance(e, dict)] if not entries: return _fail( @@ -692,7 +694,6 @@ class OpenRouterCompatImageProvider(ImageGenProvider): return _fail(f"Could not save generated image: {exc}", "io_error") if not saved: return _fail(f"{self._display} response carried neither b64_json nor url.", "empty_response") - record_token_usage(_dict_at(body, "usage"), model=model_id, provider=self._name, base_url=base_url) return success_response( image=saved[0], model=model_id, prompt=prompt, aspect_ratio=semantic_aspect, provider=self._name, modality="image" if usable_refs else "text", @@ -725,6 +726,8 @@ class OpenRouterCompatImageProvider(ImageGenProvider): return fail(hint, "model_access"), "unavailable" return fail(failure.error, "api_error"), None + # The provider billed these tokens on HTTP 200 whether or not an image came back / saved. + record_token_usage(_dict_at(result, "usage"), model=model_id, provider=self._name, base_url=base_url) images = _extract_images(result) if not images: # Text but no image usually means the model didn't honor image output. @@ -741,7 +744,6 @@ class OpenRouterCompatImageProvider(ImageGenProvider): saved_path = save_url_image(first, prefix=f"{self._name}_gen") except Exception as exc: # noqa: BLE001 return fail(f"Could not save generated image: {exc}", "io_error"), None - record_token_usage(_dict_at(result, "usage"), model=model_id, provider=self._name, base_url=base_url) return success_response( image=str(saved_path), model=model_id, prompt=prompt, aspect_ratio=aspect, provider=self._name, ), None diff --git a/tests/plugins/image_gen/test_openai_provider.py b/tests/plugins/image_gen/test_openai_provider.py index 9a315cbb07..2797ebee1e 100644 --- a/tests/plugins/image_gen/test_openai_provider.py +++ b/tests/plugins/image_gen/test_openai_provider.py @@ -172,9 +172,11 @@ class TestGenerate: # gpt-image-2 rejects response_format — we must NOT send it. assert "response_format" not in call_kwargs - def test_token_usage_reaches_session_accounting(self, provider): + @pytest.mark.parametrize("has_image", [True, False]) + def test_token_usage_reaches_session_accounting(self, provider, has_image): """gpt-image bills per token: the Images API ``usage`` block lands as one - ``image_generation`` row keyed on the API model, not the Hermes tier label.""" + ``image_generation`` row keyed on the API model, not the Hermes tier label — also + when the billed HTTP 200 carries no image data.""" from agent import aux_accounting recorded = [] @@ -184,6 +186,8 @@ class TestGenerate: recorded.append((args, kwargs)) response = _fake_response(b64=_b64_png()) + if not has_image: + response.data = [] response.usage = SimpleNamespace(input_tokens=23, output_tokens=1056, total_tokens=1079) fake_client = MagicMock() fake_client.images.generate.return_value = response @@ -194,7 +198,9 @@ class TestGenerate: finally: aux_accounting.reset_accounting_context(token) - assert result["success"] is True + assert result["success"] is has_image + if not has_image: + assert result["error_type"] == "empty_response" ((session_id, task), kwargs), = recorded assert (session_id, task) == ("sess-1", "image_generation") assert (kwargs["model"], kwargs["billing_provider"]) == ("gpt-image-2", "openai") diff --git a/tests/plugins/image_gen/test_openrouter_compat_provider.py b/tests/plugins/image_gen/test_openrouter_compat_provider.py index ec6a1aa5a0..09ad4ffb35 100644 --- a/tests/plugins/image_gen/test_openrouter_compat_provider.py +++ b/tests/plugins/image_gen/test_openrouter_compat_provider.py @@ -705,14 +705,17 @@ class TestImageApiSurface: _USAGE = {"prompt_tokens": 1000, "completion_tokens": 128, "total_tokens": 1128} - @pytest.mark.parametrize("surface, model, usage", [ - ("chat", "openai/gpt-5.4-image-2", _USAGE), # default chain: token-billed via /chat/completions - ("images", "krea/krea-2-medium", _USAGE), # curated Image API model - ("images", "krea/krea-2-medium", None), # flat-fee body without usage: no write + @pytest.mark.parametrize("surface, model, usage, images", [ + ("chat", "openai/gpt-5.4-image-2", _USAGE, True), # default chain: token-billed via /chat/completions + ("images", "krea/krea-2-medium", _USAGE, True), # curated Image API model + ("images", "krea/krea-2-medium", None, True), # flat-fee body without usage: no write + ("chat", "openai/gpt-5.4-image-2", _USAGE, False), # billed HTTP 200 with text but no image + ("images", "krea/krea-2-medium", _USAGE, False), # billed HTTP 200 with empty ``data`` ]) - def test_token_usage_reaches_session_accounting(self, surface, model, usage): + def test_token_usage_reaches_session_accounting(self, surface, model, usage, images): """A response carrying token usage records one ``image_generation`` row on the ambient - session; a body without usage records nothing.""" + session — also when it carries no image (the provider billed the tokens anyway); a body + without usage records nothing.""" from agent import aux_accounting recorded = [] @@ -722,10 +725,10 @@ class TestImageApiSurface: recorded.append((args, kwargs)) if surface == "chat": - response = _mock_chat_response([_PNG_DATA_URI]) + response = _mock_chat_response([_PNG_DATA_URI] if images else []) response.json.return_value["usage"] = dict(usage) else: - response = _mock_image_api_response(usage=usage) + response = _mock_image_api_response([] if not images else None, usage=usage) token = aux_accounting.set_accounting_context(_DB(), "sess-1") try: with patch(_RUNTIME, return_value=_runtime_ok()), \ @@ -735,7 +738,9 @@ class TestImageApiSurface: finally: aux_accounting.reset_accounting_context(token) - assert result["success"] is True + assert result["success"] is images + if not images: + assert result["error_type"] == "empty_response" if usage is None: assert recorded == [] return From 1f6e737ecbcacd47346e4d4b50fa522816498c84 Mon Sep 17 00:00:00 2001 From: Yagna Vudathu Date: Thu, 17 Sep 2026 00:08:25 -0400 Subject: [PATCH 43/65] fix(cli): heal memory-provider bridge packages on every venv repair path The venv-repair and runtime-repair paths reinstalled core, lazy and tool deps but never refreshed memory-provider bridge packages, unlike the git-pull and ZIP paths. Both repair paths now heal them last. Fixes #113741. --- hermes_cli/update_cmd.py | 6 +++++ tests/hermes_cli/test_cmd_update.py | 42 +++++++++++++++++++++++++++++ 2 files changed, 48 insertions(+) diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index ba2c2ce6e1..3aebc4466e 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -651,6 +651,9 @@ def _repair_venv_on_current_checkout( _m()._refresh_active_lazy_features(repair_prefix, env=repair_env, features=active_lazy_features) _m()._restore_active_tool_dependencies(active_tool_dependencies, repair_prefix, env=repair_env) _m()._reapply_plugin_python_dependencies() + # Heal memory-provider bridge packages last. The steps above may have stripped them + # (parity with the git-pull and ZIP paths, #113741). + _m()._refresh_active_memory_provider_dependencies() # Core ``.[all]`` install finished. Clear the generic core breadcrumb before the lazy-refresh phase — # that phase uses its own marker so a later lazy failure cannot be "healed" by clearing the core marker # based on a narrow 7-package import probe (#58004 review). @@ -742,6 +745,9 @@ def _repair_current_checkout( _m()._restore_active_tool_dependencies( active_tool_dependencies, repair_prefix, env=repair_env) _m()._reapply_plugin_python_dependencies() + # Heal memory-provider bridge packages last. The swapped-in venv was built + # from uv.lock alone (parity with the pull/ZIP/venv-repair paths, #113741). + _m()._refresh_active_memory_provider_dependencies() current_checkout_complete = _repair_node_deps_on_current_checkout( _print_verified_update_completion, assume_yes=assume_yes, gateway_mode=gateway_mode, pre_update_snapshot_id=pre_update_snapshot_id, diff --git a/tests/hermes_cli/test_cmd_update.py b/tests/hermes_cli/test_cmd_update.py index b9d38f847f..24fff6438b 100644 --- a/tests/hermes_cli/test_cmd_update.py +++ b/tests/hermes_cli/test_cmd_update.py @@ -351,6 +351,48 @@ class TestRepairCurrentCheckoutRuntimeRepair: assert markers == [] + def test_repair_refreshes_memory_provider_after_tool_restore(self, monkeypatch): + """The venv-repair path must heal memory-provider bridge packages like the + git-pull and ZIP paths do, after the tool-dep restore (#113741).""" + from hermes_cli import main as hm + + calls: list = [] + monkeypatch.setattr( + hm, "_refresh_active_memory_provider_dependencies", lambda: calls.append("memory")) + restored, _, _, _ = self._run(monkeypatch, repaired=True) + assert calls == ["memory"] + assert restored[-1][0] == "tools" + + def test_venv_repair_path_refreshes_memory_provider(self, monkeypatch, tmp_path): + """The unhealthy-venv repair path heals memory-provider bridge packages too + (parity with pull/ZIP, #113741). Drives _repair_venv_on_current_checkout.""" + from hermes_cli import main as hm + + calls: list = [] + monkeypatch.setattr(hm, "_abort_dependency_sync_if_self_locked", lambda *_a, **k: None) + monkeypatch.setattr(update_cmd, "_write_update_incomplete_marker", lambda: None) + monkeypatch.setattr(update_cmd, "_venv_core_imports_healthy", lambda: (True, "ok")) + monkeypatch.setattr(update_cmd, "project_venv_dir", lambda _root: tmp_path / "venv") + monkeypatch.setattr( + update_cmd, "venv_python_path", lambda _d, **k: tmp_path / "venv" / "bin" / "python") + monkeypatch.setattr(update_cmd, "_pip_install_prefix", lambda _uv: (["uv", "pip"], {})) + monkeypatch.setattr( + hm, "_install_python_dependencies_with_optional_fallback", lambda *_a, **k: None) + monkeypatch.setattr(hm, "_refresh_active_lazy_features", lambda *a, **k: True) + monkeypatch.setattr(hm, "_restore_active_tool_dependencies", lambda *a, **k: None) + monkeypatch.setattr( + hm, "_refresh_active_memory_provider_dependencies", lambda: calls.append("memory")) + monkeypatch.setattr(hm, "_clear_update_incomplete_marker", lambda: None) + monkeypatch.setattr(hm, "_is_windows", lambda: False) + monkeypatch.setattr("hermes_cli.managed_uv.ensure_uv", lambda **k: "uv") + + assert update_cmd._repair_venv_on_current_checkout( + assume_yes=True, gateway_mode=False, pre_update_snapshot_id=None, + had_desktop_app_before_update=False, active_lazy_features=[], + active_tool_dependencies=[], _windows_gateway_resume=None, + ) + assert calls == ["memory"] + class TestCmdUpdateBranchFallback: """cmd_update falls back to main when current branch has no remote counterpart.""" From 6c16233b0a27a292e7572b70cfdc83623112ebd2 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 00:40:07 -0700 Subject: [PATCH 44/65] fix(update): refresh memory-provider deps in the same order as the pull path Both repair paths now run the memory-provider refresh between the tool-dep restore and the plugin-dep reapply, exactly like _sync_python_dependencies_after_pull and the ZIP path, so the three routes stay interchangeable. --- hermes_cli/update_cmd.py | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index 3aebc4466e..966987d7a4 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -650,10 +650,10 @@ def _repair_venv_on_current_checkout( _m()._install_python_dependencies_with_optional_fallback(repair_prefix, env=repair_env, group="all") _m()._refresh_active_lazy_features(repair_prefix, env=repair_env, features=active_lazy_features) _m()._restore_active_tool_dependencies(active_tool_dependencies, repair_prefix, env=repair_env) - _m()._reapply_plugin_python_dependencies() - # Heal memory-provider bridge packages last. The steps above may have stripped them - # (parity with the git-pull and ZIP paths, #113741). + # Same order as the pull and ZIP paths: the ``[all]`` reinstall above may have stripped the + # active memory provider's bridge packages (hindsight-embed, torch, ...). _m()._refresh_active_memory_provider_dependencies() + _m()._reapply_plugin_python_dependencies() # Core ``.[all]`` install finished. Clear the generic core breadcrumb before the lazy-refresh phase — # that phase uses its own marker so a later lazy failure cannot be "healed" by clearing the core marker # based on a narrow 7-package import probe (#58004 review). @@ -744,10 +744,9 @@ def _repair_current_checkout( "to finish import-based venv repair.") _m()._restore_active_tool_dependencies( active_tool_dependencies, repair_prefix, env=repair_env) - _m()._reapply_plugin_python_dependencies() - # Heal memory-provider bridge packages last. The swapped-in venv was built - # from uv.lock alone (parity with the pull/ZIP/venv-repair paths, #113741). + # Same order as the pull path: the swapped-in venv was built from uv.lock alone. _m()._refresh_active_memory_provider_dependencies() + _m()._reapply_plugin_python_dependencies() current_checkout_complete = _repair_node_deps_on_current_checkout( _print_verified_update_completion, assume_yes=assume_yes, gateway_mode=gateway_mode, pre_update_snapshot_id=pre_update_snapshot_id, From e56ec8c9f7a275e9b6892627e941498155d44bf7 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 19:16:27 -0700 Subject: [PATCH 45/65] fix(chronos): a 403 invalid_client from NAS hands cron fires to the built-in ticker (#97494) NAS maps the agent-cron bearer to a provisioned instance via an `agent:*` client or the `hermes-cli-vps` bootstrap session (hermes-portal `server/agent-cron/instance-auth.ts`). A container whose auth.json holds a plain `hermes-cli` user login is refused with 403 invalid_client on every arm, re-arm and list, for the life of that credential - and the re-login users try first (`hermes auth logout nous` + device code) replaces the bootstrap session, making it permanent. Chronos previously logged one bare warning per job and left the jobs with no trigger at all: they only ran through the misfire sweep, minutes late. NasCronClientError now carries the HTTP status and the OAuth `error` code. On an identity rejection the provider logs ONE warning that names the real remedy (restore the hosted credential from the Nous Portal; re-login cannot fix it), stops calling NAS, and starts the built-in ticker with the gateway's own adapters/loop so scheduled jobs keep firing on time. Transient 5xx/transport failures keep retrying on the next reconcile. Supersedes #97566 (@wesleysimplicio), whose runtime-credential swap resolves to the same bearer on main (`agent_key` is the access token) and so could not change the 403. --- plugins/cron_providers/chronos/__init__.py | 51 +++++++++++++- plugins/cron_providers/chronos/_nas_client.py | 24 ++++++- tests/cron/test_ticker_startup_survival.py | 14 ++++ tests/plugins/test_chronos_cron.py | 66 +++++++++++++++++++ .../chronos-managed-cron-contract.md | 14 ++++ 5 files changed, 165 insertions(+), 4 deletions(-) diff --git a/plugins/cron_providers/chronos/__init__.py b/plugins/cron_providers/chronos/__init__.py index 0af25f24bc..9a846aa73d 100644 --- a/plugins/cron_providers/chronos/__init__.py +++ b/plugins/cron_providers/chronos/__init__.py @@ -15,6 +15,8 @@ from typing import Any, Dict from cron.scheduler_provider import CronScheduler +from ._nas_client import NasCronClientError + logger = logging.getLogger("cron.chronos") @@ -35,6 +37,12 @@ class ChronosCronScheduler(CronScheduler): self._armed: Dict[str, str] = {} self._lock = threading.Lock() self._client = None # lazily constructed (no network in is_available) + # Set when NAS answered 403 invalid_client: the Nous token in auth.json is not this + # instance's provisioned identity, so every arm would fail the same way for the life of + # the process. Once set, NAS is left alone and the built-in ticker fires jobs (#97494). + self._identity_rejected = False + self._stop_event = None + self._ticker_kwargs: Dict[str, Any] = {} @property def name(self) -> str: @@ -66,6 +74,10 @@ class ChronosCronScheduler(CronScheduler): def start(self, stop_event, *, adapters=None, loop=None, interval=60): """Arm all enabled jobs via NAS, then RETURN — no loop, no periodic wake (scale-to-zero).""" + # Kept so a later identity rejection (boot or mid-life re-arm) can hand this process's + # fires to the built-in ticker with the gateway's own adapters/loop. + self._ticker_kwargs = {"adapters": adapters, "loop": loop, "interval": interval} + self._stop_event = stop_event # A new lifecycle can't prove what an interrupted process did: classify unknown, never requeue. self.recover_interrupted() self._reconcile_logged(logger.warning, "start()") @@ -74,15 +86,23 @@ class ChronosCronScheduler(CronScheduler): pass def on_jobs_changed(self) -> None: - self._reconcile_logged(logger.debug, "on_jobs_changed") + if not self._identity_rejected: + self._reconcile_logged(logger.debug, "on_jobs_changed") def register_job(self, job: Dict[str, Any]) -> None: """Arm the first one-shot for a new job; may raise so creation can report it.""" - self._arm_one_shot(job) + try: + self._arm_one_shot(job) + except NasCronClientError as e: + if not e.identity_rejected: + raise + self._note_identity_rejected() # the job is stored; the ticker fires it def _arm_one_shot(self, job: Dict[str, Any]) -> None: """Arm one one-shot at next_run_at (agent computes the time; NAS executes). dedup_key=(job_id, fire_at) makes re-arming the same fire a no-op.""" + if self._identity_rejected: + return # the built-in ticker owns this process's fires; NAS would 403 again job_id = job["id"] fire_at = job.get("next_run_at") if not fire_at: @@ -93,11 +113,38 @@ class ChronosCronScheduler(CronScheduler): with self._lock: self._armed[job_id] = fire_at + def _note_identity_rejected(self) -> None: + """403 invalid_client is deterministic: NAS maps the bearer to a provisioned instance via an + ``agent:*`` client or the hosted bootstrap session, and a plain ``hermes auth`` login is + neither — re-logging in cannot fix it, which is what users try first (#97494). Without + NAS the jobs have no trigger at all (the misfire sweep runs them ``misfire_grace_minutes`` + late), so the built-in ticker takes over this process's fires.""" + with self._lock: + if self._identity_rejected: + return + self._identity_rejected = True + logger.warning( + "Chronos: NAS rejected this agent's Nous credential for agent-cron (403 invalid_client). " + "The Nous token in auth.json is not this instance's provisioned identity (an agent:* client " + "or the hosted bootstrap session), so no job can be armed. A normal `hermes auth` re-login " + "cannot fix this; the hosted credential has to be restored from the Nous Portal. Falling back " + "to the built-in cron ticker for this process so scheduled jobs keep firing on time.") + if self._stop_event is None: + return # start() never ran (e.g. a CLI `hermes cron add`); nothing to tick here + from agent.memory_provider import spawn_context_thread + from cron.scheduler_provider import InProcessCronScheduler + spawn_context_thread( + InProcessCronScheduler().start, name="cron-scheduler-chronos-fallback", + args=(self._stop_event,), kwargs=self._ticker_kwargs).start() + def _arm_logged(self, job: Dict[str, Any], what: str) -> None: """Best-effort arm: log a warning instead of raising (reconcile/fire must not die).""" try: self._arm_one_shot(job) except Exception as e: + if isinstance(e, NasCronClientError) and e.identity_rejected: + self._note_identity_rejected() + return logger.warning("Chronos failed to %s: %s", what, e) def _cancel(self, job_id: str) -> None: diff --git a/plugins/cron_providers/chronos/_nas_client.py b/plugins/cron_providers/chronos/_nas_client.py index 6777a271d5..2ef7843b5f 100644 --- a/plugins/cron_providers/chronos/_nas_client.py +++ b/plugins/cron_providers/chronos/_nas_client.py @@ -16,7 +16,22 @@ _LIST_PATH = "/api/agent-cron/list" class NasCronClientError(RuntimeError): - """Raised when a NAS agent-cron call fails (non-2xx or transport error).""" + """Raised when a NAS agent-cron call fails (non-2xx or transport error). + + ``status`` is the HTTP status (None on transport error) and ``error_code`` the OAuth-style + ``error`` field of a JSON error body (``invalid_client`` marks a deterministic identity + rejection the provider can act on, unlike a transient 5xx). + """ + + def __init__(self, message: str, *, status: int | None = None, error_code: str = "") -> None: + super().__init__(message) + self.status = status + self.error_code = error_code + + @property + def identity_rejected(self) -> bool: + """NAS refused the bearer as not belonging to a provisioned agent (never transient).""" + return self.status == 403 and self.error_code == "invalid_client" class NasCronClient: @@ -41,7 +56,12 @@ class NasCronClient: except Exception as e: raise NasCronClientError(f"{method} {path} failed: {e}") from e if resp.status_code // 100 != 2: - raise NasCronClientError(f"{method} {path} returned {resp.status_code}: {resp.text[:200]}") + error_code = "" + with contextlib.suppress(Exception): + error_code = str((resp.json() or {}).get("error") or "") + raise NasCronClientError( + f"{method} {path} returned {resp.status_code}: {resp.text[:200]}", + status=resp.status_code, error_code=error_code) with contextlib.suppress(Exception): return resp.json() if resp.content else {} return {} diff --git a/tests/cron/test_ticker_startup_survival.py b/tests/cron/test_ticker_startup_survival.py index 4884741228..ffb05da7d8 100644 --- a/tests/cron/test_ticker_startup_survival.py +++ b/tests/cron/test_ticker_startup_survival.py @@ -105,3 +105,17 @@ def test_housekeeping_restarts_a_dead_ticker(monkeypatch): stop.set() gateway_run._start_gateway_housekeeping(_OneTick(), interval=0, cron_thread=ticker) assert len(starts) == 2, "a stopped ticker must not be respawned" + + +def test_supervisor_leaves_a_returning_external_provider_alone(): + """An external provider's start() (Chronos) arms remote one-shots and RETURNS by design; the + supervisor must not read that as a dead ticker and re-run start() every housekeeping tick + (each rerun re-reconciles against NAS and logs a spurious "died without a stop request").""" + from cron.scheduler_thread import SupervisedTickerThread + + starts = [] + stop = threading.Event() + ticker = SupervisedTickerThread(lambda stop_event: starts.append(1), args=(stop,), stop_event=stop) + ticker.start() + _wait_until(lambda: not ticker.is_alive()) + assert ticker.restart_if_dead() is False and starts == [1] and ticker.restarts == 0 diff --git a/tests/plugins/test_chronos_cron.py b/tests/plugins/test_chronos_cron.py index 97486632bf..cb63781b97 100644 --- a/tests/plugins/test_chronos_cron.py +++ b/tests/plugins/test_chronos_cron.py @@ -100,6 +100,72 @@ def test_register_job_propagates_provision_failure(chronos): # -- reconcile ---------------------------------------------------------------- +def test_identity_rejection_hands_fires_to_the_builtin_ticker(temp_home, chronos, monkeypatch, caplog): + """Regression for #97494: NAS answering 403 invalid_client is a deterministic identity rejection + (the auth.json token is not this instance's provisioned agent), not a transient. The provider + must say so ONCE with the remedy, stop calling NAS, and start the built-in ticker so jobs still + fire on time instead of only via the late misfire sweep.""" + import threading + + from plugins.cron_providers.chronos._nas_client import NasCronClientError + + prov, fake = chronos + calls = [] + + def rejected(**kw): + calls.append(kw["job_id"]) + raise NasCronClientError( + "POST /api/agent-cron/provision returned 403: invalid_client", + status=403, error_code="invalid_client") + + fake.provision = rejected + jobs = [ + {"id": "a", "enabled": True, "next_run_at": "2026-06-18T12:00:00+00:00", "state": "scheduled"}, + {"id": "b", "enabled": True, "next_run_at": "2026-06-18T12:05:00+00:00", "state": "scheduled"}, + ] + monkeypatch.setattr("cron.jobs.load_jobs", lambda: jobs) + monkeypatch.setattr("cron.jobs.get_job", lambda jid: next(j for j in jobs if j["id"] == jid)) + monkeypatch.setattr("cron.executions.recover_interrupted_executions", lambda: 0) + ticker_started = threading.Event() + monkeypatch.setattr( + "cron.scheduler_provider.InProcessCronScheduler.start", + lambda self, stop_event, **kw: ticker_started.set()) + + stop = threading.Event() + with caplog.at_level("WARNING", logger="cron.chronos"): + prov.start(stop, adapters={"x": 1}, loop=None, interval=7) + assert ticker_started.wait(2.0), "the built-in ticker must take over this process's fires" + assert calls == ["a"], "after the first rejection NAS is left alone" + identity_msgs = [r.message for r in caplog.records if "re-login" in r.message] + assert len(identity_msgs) == 1 and "built-in cron ticker" in identity_msgs[0] + + # Job creation and re-arms no longer reach NAS (and no longer fail the create). + prov.register_job({"id": "c", "next_run_at": "2026-06-18T12:10:00+00:00"}) + prov.on_jobs_changed() + assert calls == ["a"] + + +def test_transient_provision_failure_does_not_degrade(temp_home, chronos, monkeypatch): + """A 5xx / transport error is retried on the next reconcile; only 403 invalid_client degrades.""" + from plugins.cron_providers.chronos._nas_client import NasCronClientError + + prov, fake = chronos + attempts = [] + + def flaky(**kw): + attempts.append(kw["job_id"]) + raise NasCronClientError("POST /api/agent-cron/provision returned 502: upstream", status=502) + + fake.provision = flaky + jobs = [{"id": "a", "enabled": True, "next_run_at": "2026-06-18T12:00:00+00:00", "state": "scheduled"}] + monkeypatch.setattr("cron.jobs.load_jobs", lambda: jobs) + monkeypatch.setattr("cron.jobs.get_job", lambda jid: jobs[0]) + + prov.reconcile() + prov.reconcile() + assert attempts == ["a", "a"] and prov._identity_rejected is False + + def test_reconcile_arms_all_enabled(temp_home, chronos, monkeypatch): prov, fake = chronos jobs = [ diff --git a/website/docs/developer-guide/chronos-managed-cron-contract.md b/website/docs/developer-guide/chronos-managed-cron-contract.md index fe331fea11..c7d4c34f40 100644 --- a/website/docs/developer-guide/chronos-managed-cron-contract.md +++ b/website/docs/developer-guide/chronos-managed-cron-contract.md @@ -219,6 +219,20 @@ If `callback_url` / `portal_url` is blank or the agent has no Nous login, `is_available()` returns False and the resolver falls back to the built-in in-process ticker — cron never loses its trigger. +**Identity rejection at runtime (`403 invalid_client`).** `is_available()` is +config-only, so it cannot tell whether the stored Nous token is the identity NAS +maps to a provisioned instance (hop 1 above). When `provision` answers 403 +`invalid_client` — the token in `auth.json` is a plain `hermes-cli` user login +rather than the `hermes-cli-vps` bootstrap session or an `agent:*` client — the +rejection is deterministic for the life of that credential: every arm, re-arm +and `list` would fail the same way, and a `hermes auth` re-login makes it +permanent (it *replaces* the bootstrap session; only NAS can re-mint one). The +provider therefore logs ONE warning naming that remedy, stops calling NAS, and +starts the built-in ticker for the rest of the process so jobs keep firing on +time instead of only through the late misfire sweep +(`cron.misfire_grace_minutes`). Transient failures (5xx, transport) do not +degrade; the next reconcile retries them. + ## Escape hatch (not default) The inbound `/api/cron/fire` verifier is pluggable (`get_fire_verifier()`). If From c6e3a77577a6759458333440b570171a1487c733 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 19:16:28 -0700 Subject: [PATCH 46/65] fix(cron): the ticker supervisor respawns only a ticker that crashed, not one that returned SupervisedTickerThread (#111010) treated any ended thread as dead. An external provider's start() (Chronos) arms remote one-shots and returns by design, so every housekeeping tick logged "Cron ticker thread died without a stop request; restarting" at ERROR and re-ran start() - a fresh recover_interrupted + NAS list/arm reconcile once a minute on every hosted instance. Track whether the target escaped with an exception and respawn only then; the built-in ticker returns normally only on stop_event. --- cron/scheduler_thread.py | 17 ++++++++++++++--- 1 file changed, 14 insertions(+), 3 deletions(-) diff --git a/cron/scheduler_thread.py b/cron/scheduler_thread.py index 2552842909..85f54e5bb7 100644 --- a/cron/scheduler_thread.py +++ b/cron/scheduler_thread.py @@ -23,11 +23,22 @@ class SupervisedTickerThread: stop_event: threading.Event, name: str = "cron-scheduler") -> None: self._target, self._args, self._kwargs = target, args, dict(kwargs or {}) self._stop_event, self._name = stop_event, name + # An external provider's start() (Chronos) arms remote one-shots and RETURNS by design; + # only a target that escaped with an exception is a dead ticker worth respawning. + self._crashed = False self._thread = self._spawn() self.restarts = 0 + def _run(self) -> None: + try: + self._target(*self._args, **self._kwargs) + except BaseException: + self._crashed = True + raise + def _spawn(self) -> threading.Thread: - return threading.Thread(target=self._target, args=self._args, kwargs=self._kwargs, daemon=True, name=self._name) + self._crashed = False + return threading.Thread(target=self._run, daemon=True, name=self._name) def start(self) -> None: self._thread.start() @@ -39,8 +50,8 @@ class SupervisedTickerThread: self._thread.join(timeout) def restart_if_dead(self) -> bool: - """Respawn the ticker when it ended without ``stop_event``; True when a restart happened.""" - if self._stop_event.is_set() or self._thread.is_alive(): + """Respawn the ticker when it crashed without ``stop_event``; True when a restart happened.""" + if self._stop_event.is_set() or self._thread.is_alive() or not self._crashed: return False self.restarts += 1 logger.error( From c20432fffa511b6290707e2b4bbe79ad362ba11d Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 19:35:32 -0700 Subject: [PATCH 47/65] fix(kanban): worker exits EX_CONFIG on a terminal provider error MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A dispatcher-spawned worker whose turn failed on a provider error that no retry can heal (credential rejected: auth / auth_permanent, model_not_found, ssl_cert_verification) now exits KANBAN_TERMINAL_PROVIDER_EXIT_CODE (78, BSD EX_CONFIG) instead of a plain 1, so the dispatcher can tell "the provider will reject every further spawn" from "this attempt crashed". The classification is the agent's own FailoverReason verdict already on the turn result (failure_reason) — no error-text regex. billing stays in the transient set: credit comes back, a revoked key does not. Shared by both one-shot paths (-q and -Q) and gated on HERMES_KANBAN_TASK, so a person's `hermes chat -q` keeps exiting 1 on the same error. Part of #114587 --- cli.py | 22 ++++++++++++++++--- hermes_cli/kanban_db.py | 5 +++++ .../test_single_query_exit_contract.py | 12 +++++++++- 3 files changed, 35 insertions(+), 4 deletions(-) diff --git a/cli.py b/cli.py index ffc124df4b..3af89c3d03 100644 --- a/cli.py +++ b/cli.py @@ -4119,6 +4119,15 @@ _TRANSIENT_PROVIDER_REASONS = frozenset({ "rate_limit", "upstream_rate_limit", "billing", "overloaded", "server_error", "timeout", }) +# ``failure_reason`` values a retry can never heal: the credential was rejected, the model does +# not exist for this account, or the TLS chain is broken. A Kanban worker exits +# ``KANBAN_TERMINAL_PROVIDER_EXIT_CODE`` so the dispatcher parks the card after ONE spawn with +# the provider's words as the reason, instead of re-spawning into the same wall until +# ``kanban.failure_limit`` is spent. ``billing`` stays transient: credit comes back. +_TERMINAL_PROVIDER_REASONS = frozenset({ + "auth", "auth_permanent", "model_not_found", "ssl_cert_verification", +}) + def _single_query_exit_code(result) -> int: """Map a one-shot turn result onto a process exit code, for both `-q` and `-Q`. @@ -4129,6 +4138,8 @@ def _single_query_exit_code(result) -> int: failed purely on a provider rate-limit / billing wall exits ``KANBAN_RATE_LIMIT_EXIT_CODE`` (EX_TEMPFAIL): the dispatcher books that run ``rate_limited`` and requeues the task WITHOUT counting a failure, so a quota window or a provider outage cannot trip the breaker. + One that failed on a terminal provider error (credential revoked, model gone) exits + ``KANBAN_TERMINAL_PROVIDER_EXIT_CODE`` (EX_CONFIG): the dispatcher blocks the card at once. """ if not isinstance(result, dict): return 1 @@ -4136,9 +4147,14 @@ def _single_query_exit_code(result) -> int: return 130 if not (result.get("failed") or result.get("partial") or result.get("completed") is False): return 0 - if os.environ.get("HERMES_KANBAN_TASK") and result.get("failure_reason") in _TRANSIENT_PROVIDER_REASONS: - from hermes_cli.kanban_db import KANBAN_RATE_LIMIT_EXIT_CODE - return KANBAN_RATE_LIMIT_EXIT_CODE + if os.environ.get("HERMES_KANBAN_TASK"): + reason = result.get("failure_reason") + if reason in _TRANSIENT_PROVIDER_REASONS: + from hermes_cli.kanban_db import KANBAN_RATE_LIMIT_EXIT_CODE + return KANBAN_RATE_LIMIT_EXIT_CODE + if reason in _TERMINAL_PROVIDER_REASONS: + from hermes_cli.kanban_db import KANBAN_TERMINAL_PROVIDER_EXIT_CODE + return KANBAN_TERMINAL_PROVIDER_EXIT_CODE return 1 diff --git a/hermes_cli/kanban_db.py b/hermes_cli/kanban_db.py index 38396ade79..248bd25e2f 100644 --- a/hermes_cli/kanban_db.py +++ b/hermes_cli/kanban_db.py @@ -305,6 +305,11 @@ DEFAULT_CRASH_GRACE_SECONDS = 30 # breaker must never trip on a throttle). 75 == BSD EX_TEMPFAIL. KANBAN_RATE_LIMIT_EXIT_CODE = 75 +# Worker exit "provider rejected the configuration": credential revoked (401/403), model gone +# (404), TLS chain broken — a retry cannot fix it, so the dispatcher parks the card blocked on +# the FIRST occurrence instead of spending ``failure_limit`` identical spawns. 78 == BSD EX_CONFIG. +KANBAN_TERMINAL_PROVIDER_EXIT_CODE = 78 + def _resolve_crash_grace_seconds() -> int: """``HERMES_KANBAN_CRASH_GRACE_SECONDS`` (0 = immediate, for tests) else default.""" diff --git a/tests/hermes_cli/test_single_query_exit_contract.py b/tests/hermes_cli/test_single_query_exit_contract.py index cef28eea59..79ac9748d5 100644 --- a/tests/hermes_cli/test_single_query_exit_contract.py +++ b/tests/hermes_cli/test_single_query_exit_contract.py @@ -13,7 +13,7 @@ from types import SimpleNamespace import pytest import cli -from hermes_cli.kanban_db import KANBAN_RATE_LIMIT_EXIT_CODE +from hermes_cli.kanban_db import KANBAN_RATE_LIMIT_EXIT_CODE, KANBAN_TERMINAL_PROVIDER_EXIT_CODE @pytest.fixture(autouse=True) @@ -53,6 +53,16 @@ def test_dispatcher_spawned_worker_signals_a_provider_outage_not_a_protocol_viol assert code == KANBAN_RATE_LIMIT_EXIT_CODE +@pytest.mark.parametrize("reason", ["auth", "auth_permanent", "model_not_found", "ssl_cert_verification"]) +def test_dispatcher_spawned_worker_signals_a_terminal_provider_error(monkeypatch, reason): + """A revoked credential / missing model cannot be retried into working: the worker says so + with EX_CONFIG so the dispatcher parks the card after one spawn. A person's run keeps 1.""" + monkeypatch.setenv("HERMES_KANBAN_TASK", "t_abc123") + assert _run_non_quiet(monkeypatch, {"failed": True, "failure_reason": reason}) == KANBAN_TERMINAL_PROVIDER_EXIT_CODE + monkeypatch.delenv("HERMES_KANBAN_TASK") + assert _run_non_quiet(monkeypatch, {"failed": True, "failure_reason": reason}) == 1 + + @pytest.mark.parametrize( ("turn_result", "expected"), [ From 2b94b0d40face903b7447ebc82f04ab8eb39d91a Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 19:35:33 -0700 Subject: [PATCH 48/65] fix(kanban): dispatcher blocks a card on the first terminal provider error MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Before, a worker killed by a revoked credential or a missing model was booked as an ordinary crash and re-spawned into the identical failure until kanban.failure_limit / max_retries was spent — burning worker slots and the retry budget on something a retry cannot fix (#114587). Now KANBAN_TERMINAL_PROVIDER_EXIT_CODE (78) is its own exit kind, `terminal_provider`: `_classify_dead_worker_exit` books the run crashed with the provider's words appended, and `_account_crashes` force-trips the breaker on that first death (sticky, so recompute_ready does not resume it before the operator fixes the provider). Same booking in the implementation and the review lane — the review worker dies through the same sweep. Transient failures (429 / 5xx / timeout) keep the existing rate_limited requeue and the consecutive_failures budget. `hermes kanban show` / the dashboard diagnostics now fire for that trip below the repeated-failure threshold ("Provider rejected this profile's credential or model — blocked after one attempt") with the fix path. No new columns, no separate review-lane counter, no new config: option B of #114587. The terminal-vs-transient split was proposed in #114589 by @TFOjojo; its regex-on-error-text classifier is replaced by the worker's own FailoverReason verdict. Part of #114587 Co-authored-by: TFOjojo <279183633+TFOjojo@users.noreply.github.com> --- cron/AGENTS.md | 5 +- hermes_cli/kanban_db_dispatch.py | 55 +++++++++++++++++----- hermes_cli/kanban_diagnostics.py | 28 ++++++++++- tests/hermes_cli/test_kanban_db.py | 44 +++++++++++++++++ website/docs/user-guide/features/kanban.md | 19 ++++++-- 5 files changed, 133 insertions(+), 18 deletions(-) diff --git a/cron/AGENTS.md b/cron/AGENTS.md index ec38f3dcdb..cddd8e5fc6 100644 --- a/cron/AGENTS.md +++ b/cron/AGENTS.md @@ -75,7 +75,10 @@ zero outside a kanban task (footprint ladder rung 3). Isolation: **board** is the hard boundary — workers get `HERMES_KANBAN_BOARD` pinned in their env and cannot see other boards; **tenant** is a soft namespace within a board (workspace-path + memory-key isolation, one fleet serving several businesses). After `kanban.failure_limit` consecutive -non-success attempts on a task (default 2) the dispatcher auto-blocks it to stop spin loops. +non-success attempts on a task (default 2) the dispatcher auto-blocks it to stop spin loops; a +worker exit of `KANBAN_TERMINAL_PROVIDER_EXIT_CODE` (78 — credential revoked, model gone; the +worker's own `failure_reason` classification via `cli._TERMINAL_PROVIDER_REASONS`) trips it on +the first attempt, sticky, because no retry can heal it (#114587). Process-identity note: `kanban --preserve-cache` contains "serve" — never classify processes by argv substring (root). Worker liveness is `(worker_pid, worker_started_at)` — the start-time fingerprint (`gateway.status.get_process_start_time`) recorded at claim time — never bare PID existence, or a diff --git a/hermes_cli/kanban_db_dispatch.py b/hermes_cli/kanban_db_dispatch.py index 5f30da9ef0..1bd402d613 100644 --- a/hermes_cli/kanban_db_dispatch.py +++ b/hermes_cli/kanban_db_dispatch.py @@ -246,6 +246,8 @@ def _exit_code_kind(code: int) -> "tuple[str, int]": return ("clean_exit", 0) if code == _kb.KANBAN_RATE_LIMIT_EXIT_CODE: return ("rate_limited", code) + if code == _kb.KANBAN_TERMINAL_PROVIDER_EXIT_CODE: + return ("terminal_provider", code) return ("nonzero_exit", code) @@ -1016,6 +1018,9 @@ class _DeadWorker: event_payload: dict protocol_violation: bool = False rate_limited: bool = False + terminal_provider: bool = False + """``KANBAN_TERMINAL_PROVIDER_EXIT_CODE``: the provider rejected the worker's + credential/model — trips the breaker on this first occurrence.""" @property def run_outcome(self) -> str: @@ -1084,6 +1089,18 @@ def _classify_dead_worker_exit( {"pid": pid, "claimer": claimer, "exit_code": code}, rate_limited=True, ) + if kind == "terminal_provider": + # The worker classified its own provider failure as unhealable (credential + # revoked, model gone): every further spawn would hit the same wall, so + # ``_account_crashes`` trips the breaker now instead of after ``failure_limit``. + return _DeadWorker( + kind, code, + f"pid {pid} exited on a terminal provider error (exit {code}): the provider rejected " + "this profile's credential or model — fix the configuration, then unblock.", + "crashed", + {"pid": pid, "claimer": claimer, "exit_kind": kind, "exit_code": code, "terminal_provider": True}, + terminal_provider=True, + ) if kind == "nonzero_exit": error_text = f"pid {pid} exited with code {code}" elif kind == "signaled": @@ -1103,9 +1120,9 @@ class _CrashSweep: crashed: list[str] = field(default_factory=list) rate_limited: list[str] = field(default_factory=list) - # ``(task_id, pid, claimer, protocol_violation, error_text)``: accounted - # after the txn via ``_record_task_failure`` (needs its own write_txn). - crash_details: list[tuple[str, int, str, bool, str]] = field(default_factory=list) + # ``(task_id, pid, claimer, dead_worker)``: accounted after the txn via + # ``_record_task_failure`` (needs its own write_txn). + crash_details: list[tuple[str, int, str, _DeadWorker]] = field(default_factory=list) # Worker-exit observer payloads, fired only after every reclaim/accounting # txn has committed. exited_hook_payloads: list[dict] = field(default_factory=list) @@ -1177,9 +1194,7 @@ def _reclaim_dead_workers(conn: sqlite3.Connection, board: Optional[str] = None) sweep.rate_limited.append(row["id"]) else: sweep.crashed.append(row["id"]) - sweep.crash_details.append( - (row["id"], pid, row["claim_lock"], dead.protocol_violation, dead.error_text) - ) + sweep.crash_details.append((row["id"], pid, row["claim_lock"], dead)) return sweep @@ -1188,16 +1203,18 @@ def _account_crashes(conn: sqlite3.Connection, crash_details: list) -> list[str] Protocol violations get a BOUNDED violation-only budget independent of ``consecutive_failures`` (per-task ``max_retries`` takes precedence); - systemic same-error crashes (>= 3 identical fingerprints this tick) trip - immediately. + systemic same-error crashes (>= 3 identical fingerprints this tick) and + terminal provider errors (credential revoked, model gone — a retry cannot + heal them) trip immediately. """ auto_blocked: list[str] = [] fp_counts: dict[str, int] = {} - for _, _, _, _, err_text in crash_details: - fp = _error_fingerprint(err_text) + for _, _, _, dead in crash_details: + fp = _error_fingerprint(dead.error_text) fp_counts[fp] = fp_counts.get(fp, 0) + 1 - for tid, pid, claimer, protocol_violation, error_text in crash_details: - if protocol_violation: + for tid, pid, claimer, dead in crash_details: + error_text = dead.error_text + if dead.protocol_violation: streak = _protocol_violation_streak(conn, tid) trow = conn.execute("SELECT max_retries FROM tasks WHERE id = ?", (tid,)).fetchone() if trow is None: @@ -1227,6 +1244,20 @@ def _account_crashes(conn: sqlite3.Connection, crash_details: list) -> list[str] "protocol_violation_limit": violation_limit, }, ) + elif dead.terminal_provider: + # A retry cannot heal a revoked credential or a missing model, so + # the whole ``failure_limit`` budget would be spent on identical + # failures. ``force_trip`` blocks now, sticky: ``recompute_ready`` + # must not auto-resume it before the operator fixes the provider. + tripped = _record_task_failure( + conn, tid, + error=error_text, + outcome="crashed", + force_trip=True, + release_claim=False, + end_run=False, + event_payload_extra={"pid": pid, "claimer": claimer, "terminal_provider": True}, + ) else: is_systemic = fp_counts.get(_error_fingerprint(error_text), 0) >= 3 extra = {"pid": pid, "claimer": claimer} diff --git a/hermes_cli/kanban_diagnostics.py b/hermes_cli/kanban_diagnostics.py index 9e255ab9f1..de123d9c7a 100644 --- a/hermes_cli/kanban_diagnostics.py +++ b/hermes_cli/kanban_diagnostics.py @@ -112,6 +112,18 @@ def _latest_event_ts(events: Iterable[Any], kinds: set[str]) -> int: return max([0, *(_event_ts(ev) for ev in events if _event_kind(ev) in kinds)]) +def _latest_gave_up_is_terminal_provider(events: Iterable[Any]) -> bool: + """True when the most recent breaker trip was a terminal provider error (credential + revoked, model gone) and nothing has resumed the task since.""" + for ev in reversed(list(events)): + kind = _event_kind(ev) + if kind == "gave_up": + return bool(_parse_payload(ev).get("terminal_provider")) + if kind in {"unblocked", "promoted", "completed", "claimed"}: + return False + return False + + def _cli_hint(label: str, command: str, *, suggested: bool = False) -> DiagnosticAction: return DiagnosticAction(kind="cli_hint", label=label, payload={"command": command}, suggested=suggested) @@ -369,8 +381,12 @@ def _rule_repeated_failures(task, events, runs, now, cfg) -> list[Diagnostic]: threshold = _positive_int(_failure_threshold(cfg), 3) failure_limit = _positive_int(cfg.get("failure_limit"), threshold) failures = _first_field(task, "consecutive_failures", "spawn_failures", 0) - if failures is None or failures < threshold: + # A terminal provider error (credential revoked, model gone) blocks the card after ONE + # attempt, below any threshold; it still needs an operator, so diagnose it now. + terminal_trip = _latest_gave_up_is_terminal_provider(events) + if not terminal_trip and (failures is None or failures < threshold): return [] + failures = failures or 0 last_err = _first_field(task, "last_failure_error", "last_spawn_error") assignee = _task_field(task, "assignee") @@ -397,7 +413,15 @@ def _rule_repeated_failures(task, events, runs, now, cfg) -> list[Diagnostic]: severity = "critical" if failures >= threshold * 2 else "error" err_snippet = _error_snippet(last_err) outcome_label = _OUTCOME_LABELS.get(most_recent_outcome or "", "failure") - if err_snippet: + if terminal_trip: + title = "Provider rejected this profile's credential or model — blocked after one attempt" + detail = ( + f"The worker's provider call failed with an error a retry cannot fix (revoked or invalid " + f"API key, model not found), so the dispatcher blocked the task instead of spending the " + f"{failure_limit}-attempt retry budget on it. Full last error:\n\n{err_snippet}\n\n" + f"Fix the assignee profile's provider credentials/model, then unblock the task." + ) + elif err_snippet: title = f"Agent {outcome_label} x{failures}: {err_snippet.splitlines()[0][:160]}" detail = ( f"This task has failed {failures} times in a row (most recent: {outcome_label}). Full " diff --git a/tests/hermes_cli/test_kanban_db.py b/tests/hermes_cli/test_kanban_db.py index 4dd3739162..f344984dbe 100644 --- a/tests/hermes_cli/test_kanban_db.py +++ b/tests/hermes_cli/test_kanban_db.py @@ -387,6 +387,50 @@ def test_rate_limit_exit_requeues_without_counting_failure( assert "crashed" not in outcomes +@pytest.mark.parametrize("lane", ["ready", "review"]) +def test_terminal_provider_exit_blocks_after_one_attempt_in_either_lane(kanban_home, monkeypatch, lane): + """A worker that exits ``KANBAN_TERMINAL_PROVIDER_EXIT_CODE`` (credential revoked, model + gone) parks the card ``blocked`` on the FIRST death — well below ``failure_limit`` and the + per-task ``max_retries`` — with the provider error as the reason, sticky against + ``recompute_ready``. Same booking for the implementation and the review lane (#114587).""" + import hermes_cli.kanban_db as _kb + from hermes_cli import kanban_db_dispatch as _kbd + + monkeypatch.setattr(_kb, "_pid_alive", lambda _pid: False) + monkeypatch.setenv("HERMES_KANBAN_CRASH_GRACE_SECONDS", "0") + + with kbc.connect() as conn: + host = _kb._claimer_id().split(":", 1)[0] + tid = kb.create_task(conn, title="terminal", assignee="a", max_retries=5) + claimed = kb.claim_task(conn, tid, claimer=f"{host}:w0") + if lane == "review": + assert kb.request_review(conn, tid, summary="done", reviewer="r", + expected_run_id=claimed.current_run_id) + assert kb.claim_review_task(conn, tid, claimer=f"{host}:r0") is not None + pid = 71000 + conn.execute("UPDATE tasks SET worker_pid=? WHERE id=?", (pid, tid)) + conn.commit() + _kbd._record_worker_exit(pid, _exited_status(_kb.KANBAN_TERMINAL_PROVIDER_EXIT_CODE)) + + crashed = kbd.detect_crashed_workers(conn) + assert tid in crashed + assert tid in getattr(_kbd.detect_crashed_workers, "_last_auto_blocked", []) + + task = kb.get_task(conn, tid) + assert task.status == "blocked" + assert task.consecutive_failures == 1 # one spawn, not failure_limit / max_retries of them + assert "terminal provider error" in (task.last_failure_error or "") + gave_up = conn.execute( + "SELECT payload FROM task_events WHERE task_id=? AND kind='gave_up'", (tid,), + ).fetchone() + assert json.loads(gave_up["payload"])["terminal_provider"] is True + + # Sticky: the breaker did not reach its counter limit, yet the card must stay parked + # until an operator fixes the provider and unblocks it. + kb.recompute_ready(conn) + assert kb.get_task(conn, tid).status == "blocked" + + def test_respawn_guard_defers_rate_limited_within_cooldown( diff --git a/website/docs/user-guide/features/kanban.md b/website/docs/user-guide/features/kanban.md index 99fd8c8c03..5fc2d4c3af 100644 --- a/website/docs/user-guide/features/kanban.md +++ b/website/docs/user-guide/features/kanban.md @@ -561,11 +561,24 @@ is part of the worker protocol. If the worker process exits with status 0 while the task is still `running`, the dispatcher treats that as a protocol violation and emits a `protocol_violation` event. A dispatcher-spawned worker whose turn failed -therefore exits non-zero: `1` for an ordinary failure, and `75` +therefore exits non-zero: `1` for an ordinary failure, `75` (`EX_TEMPFAIL`) when the provider was rate-limited, overloaded, returning 5xx or timing out, or the account hit a billing/quota wall — the dispatcher records that run as `rate_limited` and requeues the task without counting a failure, so a quota window is never -booked as a protocol violation. The worker also writes its exit code as the +booked as a protocol violation — and `78` (`EX_CONFIG`) when the provider +rejected something a retry cannot fix: the profile's credential (401/403, +revoked or invalid key), the model (404 / model not found) or the TLS chain. +That **terminal provider error** trips the circuit breaker on the first +occurrence: the dispatcher records the run as `crashed` with +`exit_kind: terminal_provider`, emits `gave_up` with `terminal_provider: true` +and parks the card `blocked` (sticky — `recompute_ready` will not auto-resume +it) with the provider's own words in `last_failure_error`, instead of +re-spawning into the same wall until `kanban.failure_limit` / `max_retries` +is spent. Both lanes get the same booking: a reviewer worker that dies on a +revoked key parks the card exactly like an implementer. `hermes kanban show` +surfaces it as *Provider rejected this profile's credential or model — blocked +after one attempt*; fix the assignee profile's provider (`hermes -p +auth` / `setup`), then `hermes kanban unblock `. The worker also writes its exit code as the last line of its own log (`[kanban-worker-exit] rc=`), so a per-tick `hermes kanban dispatch` process — which never reaped the worker and cannot read its exit status — books the same death the same way the gateway-embedded @@ -1407,7 +1420,7 @@ Every transition appends a row to `task_events`. Each row carries an optional `r | `respawn_guarded` | `{reason}` | Dispatcher refused to re-spawn this ready task this tick. Reasons: `infrastructure_cooldown` (the host refused the last spawn — no restart-safe systemd scope — and the cooldown has not elapsed; never counted against the card), `rate_limit_cooldown` (the last run hit a quota wall; same cooldown, never counted), `blocker_auth` (last failure was a quota/auth/429 error — wait for the rate window to reset), `recent_success` (a completed run happened in the last hour — wait for review before re-running), `active_pr` (a GitHub PR URL appears in a recent comment — a prior worker already opened a PR). The task stays in `ready`; the next tick gets another chance to spawn. If the underlying condition persists, the normal `consecutive_failures` circuit breaker will auto-block via `gave_up` after `failure_limit` failures. | | `spawn_failed` | `{error, failures}` | One spawn attempt failed (missing PATH, workspace unmountable, …). Counter increments; task returns to `ready` for retry. | | `protocol_violation` | `{pid, claimer, exit_code, protocol_violation, worker_output?}` | Worker exited successfully while the task was still `running`, usually because it answered without a terminal board call (`kanban_complete`, `kanban_request_review` or `kanban_block`). Emitted on every violation (the payload's `protocol_violation: true` marker is copied into the run metadata and feeds the violation-only retry budget). Below the budget — up to `_PROTOCOL_VIOLATION_FAILURE_LIMIT` (default 3) *consecutive* violations, per-task `max_retries` overriding — the task simply returns to `ready` for another attempt; when the streak reaches the bound the dispatcher also emits `gave_up` and auto-blocks. `worker_output` carries the worker's own last printed text (usually its explanation of why it stopped), also folded into `last_failure_error` and shown to the retry worker as the prior-attempt error. | -| `gave_up` | `{failures, effective_limit, limit_source, error}` | Circuit breaker fired after N consecutive non-successful attempts. Task auto-blocks with the last error. The effective limit resolves as task `max_retries`, then dispatcher `failure_limit` / `kanban.failure_limit`, then the built-in default. | +| `gave_up` | `{failures, effective_limit, limit_source, error, terminal_provider?}` | Circuit breaker fired after N consecutive non-successful attempts. Task auto-blocks with the last error. The effective limit resolves as task `max_retries`, then dispatcher `failure_limit` / `kanban.failure_limit`, then the built-in default. `terminal_provider: true` means the worker exited `78` on a provider error a retry cannot fix (credential revoked, model gone) and the breaker fired on that first attempt, sticky, regardless of the limit. | `hermes kanban tail ` shows these for a single task. `hermes kanban watch` streams them board-wide. From b71359066cbd8987b4b06ca7ea3bb3c556df41ea Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 20:06:02 -0700 Subject: [PATCH 49/65] fix: /stop halts background subagents and returns their partial results as interrupted completions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A turn's hard interrupt fans out only to `_active_children`; background delegate_task units are detached from the parent at dispatch (`_dispatch_background` / honor_parent_interrupt=False), so /stop left them running to completion and their result arrived minutes later as a parked wake. Every stop surface now also calls `tools.async_delegation.interrupt_for_session` for the session's units: - gateway: `_interrupt_and_clear_session` (busy /stop, /new share it — /new already did this in `_handle_reset_command`, the second request is idempotent) and the idle `_handle_stop_command` tail, which replied "No active task to stop." while a background child was running; it now stops them and replies "Stopped". - tui_gateway `session.interrupt` (Desktop Stop / TUI stop) — own UI sid + spawner id only, so a viewer tab never kills gateway work. - acp_adapter `cancel`. - CLI `/stop` already used `interrupt_all` (process-wide); unchanged. The stop recurses: depth>0 delegations are always synchronous (`_model_background_value`), so the child's hard interrupt reaches its workers through its own `_active_children` fan-out, and each level's interrupted partial result rolls up as that child's completion. An interrupted child's entry now carries what it actually had: the loop's `final_response` is the "Operation interrupted." placeholder (also appended as the closing assistant row), so `_build_result_entry` takes the child's last real assistant text as `summary` and keeps the placeholder as `error`. The unit finalizes normally and re-enters at once as its completion notice (status=interrupted, "Partial output: ...", "Subagent Task Interrupted" on TUI/Desktop) instead of the chat waiting for the child's budget to run out. Docs: delegate_task description, tools/AGENTS.md, delegation.md, gateway-session-lifecycle.md. Part of #114456 --- acp_adapter/server.py | 4 + gateway/run_agent_cache.py | 7 ++ gateway/slash_commands.py | 8 +- .../test_stop_ends_background_delegations.py | 80 +++++++++++++++++++ ...est_delegate_interrupted_partial_output.py | 28 +++++++ tools/AGENTS.md | 7 +- tools/delegate_tool.py | 8 +- tools/delegate_tool_child_run.py | 12 +++ tui_gateway/session_lifecycle.py | 8 ++ .../gateway-session-lifecycle.md | 11 +++ .../docs/user-guide/features/delegation.md | 8 +- 11 files changed, 170 insertions(+), 11 deletions(-) create mode 100644 tests/gateway/test_stop_ends_background_delegations.py create mode 100644 tests/tools/test_delegate_interrupted_partial_output.py diff --git a/acp_adapter/server.py b/acp_adapter/server.py index d9a5f0beaa..719766c1da 100644 --- a/acp_adapter/server.py +++ b/acp_adapter/server.py @@ -638,6 +638,10 @@ class HermesACPAgent(SlashCommandsMixin, acp.Agent): try: if state.agent: request_hard_interrupt(state.agent) + # Background delegations are detached from the turn's fan-out; they end with the cancel. + from tools.async_delegation import interrupt_for_session + interrupt_for_session(parent_session_id=str(getattr(state.agent, "session_id", "") or ""), + reason="acp_cancel") except Exception: logger.debug("Failed to interrupt ACP session %s", session_id, exc_info=True) logger.info("Cancelled session %s", session_id) diff --git a/gateway/run_agent_cache.py b/gateway/run_agent_cache.py index 462a50ee12..19a7e0b086 100644 --- a/gateway/run_agent_cache.py +++ b/gateway/run_agent_cache.py @@ -481,6 +481,13 @@ class GatewayAgentCacheMixin: session_key, interrupt_reason=interrupt_reason, invalidation_reason=invalidation_reason, ) from gateway.run import _AGENT_PENDING_SENTINEL + # The turn's hard interrupt reaches only its in-turn children; background delegations were + # detached at dispatch and would otherwise run to completion and wake the session later. + # Each interrupted unit still returns as a completion (status=interrupted, partial output). + from tools.async_delegation import interrupt_for_session + interrupt_for_session( + session_key=session_key, reason=invalidation_reason, + parent_session_id=str(getattr(running_agent, "session_id", "") or "")) if running_agent and running_agent is not _AGENT_PENDING_SENTINEL: # Plugins holding a per-turn external resource (an outbound RPC blocked on a tool result # the loop will never consume) learn the turn is gone. Fires for /stop and the /new diff --git a/gateway/slash_commands.py b/gateway/slash_commands.py index 37c0acca3e..1b0dd0309e 100644 --- a/gateway/slash_commands.py +++ b/gateway/slash_commands.py @@ -458,7 +458,13 @@ class GatewaySlashCommandsMixin( reason, session_key, len(fallback_keys), ", ".join(fallback_keys)) return EphemeralReply(t("gateway.stop.stopped")) - # No running agent anywhere for this scope. A platform status indicator can still be stuck — + # No running agent anywhere for this scope. Background delegations the session dispatched in an + # earlier turn still count as "active": stop them; each returns as an interrupted completion. + from tools.async_delegation import interrupt_for_session + if interrupt_for_session(session_key=session_key, reason="stop_command", + parent_session_id=str(getattr(session_entry, "session_id", "") or "")): + return EphemeralReply(t("gateway.stop.stopped")) + # A platform status indicator can still be stuck — # e.g. Slack's persistent assistant.threads.setStatus survives a gateway restart or a turn # that died without a final send. # Best-effort clear so /stop always dismisses a phantom "is thinking...". See #32295. diff --git a/tests/gateway/test_stop_ends_background_delegations.py b/tests/gateway/test_stop_ends_background_delegations.py new file mode 100644 index 0000000000..17ba5010af --- /dev/null +++ b/tests/gateway/test_stop_ends_background_delegations.py @@ -0,0 +1,80 @@ +"""Invariant: ``/stop`` ends the session's BACKGROUND delegate_task children too — whether the session is +mid-turn (busy fast path) or idle after the dispatching turn already ended — instead of only the in-turn +children, leaving the detached unit to run to completion and wake the chat later. See #114456. + +Real ``GatewayRunner`` + real ``BasePlatformAdapter`` subclass; the background unit is a real +``tools.async_delegation`` registry record whose ``interrupt_fn`` is the observable. +""" +from unittest.mock import MagicMock + +import pytest + +import tools.async_delegation as ad +from gateway.config import GatewayConfig, Platform, PlatformConfig +from gateway.platforms.base import BasePlatformAdapter, SendResult +from gateway.platforms.event import MessageEvent, MessageType +from gateway.run import GatewayRunner +from gateway.session import SessionSource + + +class _Adapter(BasePlatformAdapter): + def __init__(self): + super().__init__(PlatformConfig(enabled=True, token="x"), Platform.TELEGRAM) + + @property + def name(self): + return "telegram" + + async def connect(self, *, is_reconnect=False): + return True + + async def disconnect(self): + pass + + async def send(self, chat_id, content, reply_to=None, metadata=None): + return SendResult(success=True) + + async def get_chat_info(self, chat_id): + return {"id": chat_id, "type": "private"} + + +@pytest.fixture(autouse=True) +def _reset_async_delegation(): + ad._reset_for_tests() + yield + ad._reset_for_tests() + + +def _seed_unit(session_key: str, parent_session_id: str = "") -> MagicMock: + """A live background unit as ``_dispatch_background`` registers it (running, routed by session_key).""" + fn = MagicMock() + with ad._records_lock: + ad._records["deleg_bg1"] = {"delegation_id": "deleg_bg1", "status": "running", "session_key": session_key, + "origin_ui_session_id": "", "parent_session_id": parent_session_id, "interrupt_fn": fn} + return fn + + +@pytest.mark.asyncio +@pytest.mark.parametrize("session_state", ["idle", "busy"]) +async def test_stop_ends_background_delegations_of_the_session(monkeypatch, session_state): + monkeypatch.setenv("TELEGRAM_ALLOWED_USERS", "u1") + adapter = _Adapter() + runner = GatewayRunner(config=GatewayConfig(platforms={Platform.TELEGRAM: PlatformConfig(enabled=True, token="x")})) + runner.adapters = {Platform.TELEGRAM: adapter} + source = SessionSource(platform=Platform.TELEGRAM, chat_id="c1", chat_type="dm", user_id="u1", user_name="tester") + key = adapter._event_session_key(MessageEvent(text="", message_type=MessageType.TEXT, source=source)) + stop_fn = _seed_unit(key) + other_fn = MagicMock() + with ad._records_lock: + ad._records["deleg_other"] = {"delegation_id": "deleg_other", "status": "running", "session_key": "agent:main:telegram:dm:c2", + "origin_ui_session_id": "", "parent_session_id": "", "interrupt_fn": other_fn} + + if session_state == "busy": + await runner._interrupt_and_clear_session(key, source, interrupt_reason="stop", invalidation_reason="stop_command") + else: + reply = await runner._handle_stop_command(MessageEvent(text="/stop", message_type=MessageType.TEXT, source=source)) + # The chat is told something WAS stopped, not "No active task to stop." + assert "Stopped" in str(getattr(reply, "text", reply)) + + stop_fn.assert_called_once() + other_fn.assert_not_called() # another chat's background work is untouched diff --git a/tests/tools/test_delegate_interrupted_partial_output.py b/tests/tools/test_delegate_interrupted_partial_output.py new file mode 100644 index 0000000000..6eb573949e --- /dev/null +++ b/tests/tools/test_delegate_interrupted_partial_output.py @@ -0,0 +1,28 @@ +"""Invariant: a child stopped mid-work reports what it HAD so far. The loop's ``final_response`` for an +interrupted turn is the placeholder ``"Operation interrupted."`` (also appended as the closing assistant row); +the parent-visible entry must carry the child's last real assistant text as ``summary`` and keep the +placeholder as ``error`` — never a partial result that reads as an empty stop. See #114456. +""" +from types import SimpleNamespace + +from tools.delegate_tool_child_run import _SchemaOutcome, _build_result_entry + + +def test_interrupted_child_entry_carries_its_partial_output(): + messages = [ + {"role": "user", "content": "kickoff"}, + {"role": "assistant", "content": [{"type": "text", "text": "Audited 3 of 7 modules; two findings so far."}], + "tool_calls": [{"id": "c1", "type": "function", "function": {"name": "terminal", "arguments": "{}"}}]}, + {"role": "tool", "tool_call_id": "c1", "content": "[Command interrupted]"}, + {"role": "assistant", "content": "Operation interrupted."}, + ] + result = {"final_response": "Operation interrupted.", "messages": messages, "api_calls": 2, + "completed": False, "interrupted": True} + child = SimpleNamespace(model="m", session_estimated_cost_usd=0.0, session_cost_status="unknown", + session_prompt_tokens=1, session_completion_tokens=1, _delegate_role="leaf") + + entry = _build_result_entry(child, result, 0, 12.5, _SchemaOutcome(None, None, [], 0)) + + assert entry["status"] == entry["exit_reason"] == "interrupted" + assert entry["summary"] == "Audited 3 of 7 modules; two findings so far." + assert entry["error"] == "Operation interrupted." diff --git a/tools/AGENTS.md b/tools/AGENTS.md index 75a973a354..588342f6fd 100644 --- a/tools/AGENTS.md +++ b/tools/AGENTS.md @@ -140,7 +140,12 @@ processes are killed at its teardown and their notices are suppressed in the par (`process_registry.transfer_ownership`) so the completion routes and reaps by the new owner; un-handed leftovers land on the result as `orphaned_processes`, exited-but-never-read notify processes as `unread_completions` (`_ChildRun.account_background_processes`, before `cleanup` kills them). **Child kernels:** a child's `execute_code` kernels are keyed `::child::`, pinned against the `max_session_kernels` LRU cap while the child runs and disposed by `cleanup` (`code_kernel.shutdown_kernels_for_delegated_child`) — never let a finished child's kernel squat the cap. **output_schema:** a miss after the one retry keeps `status: completed` with the raw text in `summary` plus `schema_valid: false` / `schema_errors` / `schema_note` — never discard a child's result. **Durability:** background delegation is process-local; work that must survive restart uses `cronjob` or -`terminal(background=True, notify_on_complete=True)`. API: `website/docs/developer-guide/subagent-lifecycle-api.md`. +`terminal(background=True, notify_on_complete=True)`. **Stop fan-out:** a turn's hard interrupt reaches only +`_active_children`; background units are detached at dispatch, so every stop surface (gateway `/stop` busy AND +idle paths, TUI `session.interrupt`, ACP `cancel`, CLI `/stop` via `interrupt_all`) calls +`async_delegation.interrupt_for_session` too. Depth>0 delegations are always synchronous +(`_model_background_value`), so the stop recurses through the child's own `_active_children` fan-out; an +interrupted entry's `summary` is the child's last real assistant text (`_build_result_entry`). API: `website/docs/developer-guide/subagent-lifecycle-api.md`. ## Tests diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index 5634fc5116..0a04dc68b1 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -545,16 +545,14 @@ _DESCRIPTION_HEAD = ( "as a new message when subagents finish ({delivery}). Background results are delivered only " "BETWEEN your turns: finish whatever does not depend on them, then give a one-line status and END YOUR TURN. Never " "wait or poll on transcripts, artifact files, or CI for a child. " - "While children run, `action` (list/steer/stop) controls them live — steer when a transcript shows a " - "child drifting.\n\n" - "USE FOR: reasoning-heavy subtasks, work that would flood your context with intermediate data, or independent " - "parallel workstreams.\n" + "While children run, `action` (list/steer/stop) controls them live.\n\n" + "USE FOR: reasoning-heavy subtasks, work that would flood your context, or independent parallel workstreams.\n" "DO NOT USE FOR (use these instead):\n" "- Mechanical multi-step work with no reasoning needed -> execute_code\n" "- A single tool call -> call the tool directly\n" "- Tasks needing user interaction -> subagents cannot ask questions\n" "- Durable work that must survive this session -> cronjob or terminal(background=True, notify=True); /stop, /new, " - "or process exit discards running subagents.\n\n" + "or process exit halts running subagents (whole tree); each returns an 'interrupted' completion with partial output.\n\n" "RULES:\n" "- Children know nothing of this conversation: pass everything needed via 'context', including any required " "output language, tone, or style (e.g. \"respond in Chinese\").\n" diff --git a/tools/delegate_tool_child_run.py b/tools/delegate_tool_child_run.py index fa0f56cddc..a2b7f72a61 100644 --- a/tools/delegate_tool_child_run.py +++ b/tools/delegate_tool_child_run.py @@ -503,8 +503,18 @@ def _build_result_entry( # "(empty)" is run_agent's give-up sentinel after repeated empty LLM # responses (usually a transport bug) — a failure, not a success. usable_summary = bool(summary) and summary.strip() != "(empty)" + interrupt_note = "" if result.get("interrupted", False): status, exit_reason = "interrupted", "interrupted" + # The loop's final_response is a placeholder here ("Operation interrupted…", also appended as the closing + # assistant row); the completion must carry what the child actually had so far — its last real assistant + # text — and keep the placeholder as the error. + from agent.message_content import flatten_message_text + placeholders = {"", summary.strip(), "Operation interrupted."} + partial = next((t for m in reversed(result.get("messages") or []) if m.get("role") == "assistant" + and (t := flatten_message_text(m.get("content")).strip()) not in placeholders), "") + if partial: + interrupt_note, summary = summary.strip(), partial elif result.get("failed") or result.get("error"): # The loop returns the error text as final_response, which would otherwise read as "completed". Never report a # provider rejection as "max_iterations" — that is only truthful for real budget exhaustion. @@ -554,6 +564,8 @@ def _build_result_entry( _failure_reason = result.get("failure_reason") if isinstance(_failure_reason, str) and _failure_reason: entry["failure_reason"] = _failure_reason + elif interrupt_note: + entry["error"] = interrupt_note # Schema-validation outcome — emitted ONLY when a schema was requested, so # legacy (schema-less) payloads keep their exact shape. diff --git a/tui_gateway/session_lifecycle.py b/tui_gateway/session_lifecycle.py index fed9a956f3..a200af8991 100644 --- a/tui_gateway/session_lifecycle.py +++ b/tui_gateway/session_lifecycle.py @@ -621,6 +621,14 @@ def _interrupt_session_turn(sid: str, session: dict, *, request_id: str | None = if should_interrupt: from agent.interrupt_compat import request_hard_interrupt request_hard_interrupt(session.get("agent")) + # Background delegations are detached from the turn's interrupt fan-out; a stop ends them too + # (own UI sid + spawner id only — a viewer tab must not kill gateway work). Each returns as an + # interrupted completion with its partial output. + with contextlib.suppress(Exception): + from tools.async_delegation import interrupt_for_session + interrupt_for_session( + origin_ui_session_id=_lifecycle_own_sid(session, sid), reason="user_stop", + parent_session_id=str(getattr(session.get("agent"), "session_id", "") or "")) if not run_thread_alive: with session["history_lock"]: if session.get("running"): diff --git a/website/docs/developer-guide/gateway-session-lifecycle.md b/website/docs/developer-guide/gateway-session-lifecycle.md index 21c78aa6a0..2df83f49cc 100644 --- a/website/docs/developer-guide/gateway-session-lifecycle.md +++ b/website/docs/developer-guide/gateway-session-lifecycle.md @@ -463,6 +463,17 @@ post-command drain starts it right away instead of the session idling until the message. Whether a wake pinned to a session that `/new` just closed may still run is decided at processing time (`_resolve_async_delegation_session`, fail-closed). +Both commands also end the session's **background delegations** (`tools.async_delegation. +interrupt_for_session`, selected by routing key and by the spawner's durable session id): +`_interrupt_and_clear_session` fans the stop out for the busy path, and `_handle_stop_command` +does the same for an idle session whose dispatching turn already ended (replying "Stopped" rather +than "No active task to stop"). The turn's own hard interrupt never reaches those units — they are +detached from `_active_children` at dispatch — so without the fan-out they run to completion and wake +the chat minutes later. Each stopped unit still finalizes normally and re-enters as its completion +notice with `status="interrupted"` and the child's partial output. `/new` and `/reset` already did this +in `_handle_reset_command`; the shared helper's earlier call is idempotent there (a hard interrupt +requested twice is one stop). + ### FIFO Invariant Each `/queue` invocation produces exactly one full agent turn, in FIFO order, with no diff --git a/website/docs/user-guide/features/delegation.md b/website/docs/user-guide/features/delegation.md index 93a793abdc..fc7330a322 100644 --- a/website/docs/user-guide/features/delegation.md +++ b/website/docs/user-guide/features/delegation.md @@ -212,7 +212,7 @@ The dispatch handle lists each unit (`units[].delegation_id`, `group`, `task_ind - **Thread pool:** Uses `ThreadPoolExecutor` with the configured concurrency limit as max workers - **Progress display:** In CLI mode, a tree-view shows tool calls from each subagent in real-time with per-task completion lines. In gateway mode, progress is batched and relayed to the parent's progress callback. CLI and TUI completion notices use task-first titles such as `Subagent Task Completed: Review changes`; multi-task groups use the group name and task count. Unsuccessful or incomplete work gets a corresponding status label. These compact notices do not replace the full results delivered to the parent agent. - **Result ordering:** Within a unit, results are sorted by task index to match input order regardless of completion order; `TASK i/N` labels index the whole call -- **Cancellation:** Follow-up messages do not cancel a top-level background batch. `/stop` or closing/resetting the owning session cancels its active children. Synchronous orchestrator children still follow their parent's interrupt state +- **Cancellation:** Follow-up messages do not cancel a top-level background batch. `/stop` (gateway `/stop`, CLI `/stop`, the Desktop/TUI Stop button, an ACP cancel) or closing/resetting the owning session ends its background children and every synchronous descendant beneath them; each stopped child still returns as a completion with `status="interrupted"` and its partial output Synchronous single-task delegation from an orchestrator runs directly without thread pool overhead. @@ -548,11 +548,11 @@ delegate_task( :::warning Background completion durability is not durable execution Top-level model-facing `delegate_task` calls run in the background automatically where the session supports later delivery. Hermes returns a handle immediately, and the result re-enters the conversation after the child or batch finishes. Orchestrator subagents wait for their workers in the current turn because they must synthesize those results before returning. Stateless request/response endpoints fall back to synchronous execution when they cannot deliver a detached result later. -- Normal follow-up messages do not cancel background children. `/stop` cancels running background delegations, and closing or resetting the owning session discards its active children. +- Normal follow-up messages do not cancel background children. `/stop` on any surface (gateway `/stop` — also when the session is idle after the dispatching turn ended — CLI `/stop`, the Desktop/TUI Stop button, an ACP cancel) ends the session's running background delegations, and closing or resetting the owning session does the same. - Explicit session close/reset interrupts that session's background children. Closing a TUI viewer of a gateway-owned session does not kill the gateway's work. - A Hermes process restart does **not** resume a running child. Its attempt becomes `unknown` because Hermes cannot prove which side effects happened. - A child that completed before restart but whose result was not delivered is restored and routed back through the owning session's normal checks. -- Cancelled children return a structured result (`status="interrupted"`, `exit_reason="interrupted"`), but because the parent was interrupted too, that result often never makes it into a user-visible reply. +- Stopped children return a structured result (`status="interrupted"`, `exit_reason="interrupted"`) whose `summary` is the last text the child produced before the stop (the interrupt placeholder moves to `error`). The stop recurses down the spawn tree — an orchestrator child's synchronous workers are interrupted first and their partial results roll up into the child's own interrupted result — and each background unit's result re-enters the conversation right away as its normal completion notice (`Subagent Task Interrupted: …`), so nothing waits for the child to exhaust its budget. For **durable execution** that must survive session closure or process restart, use: @@ -566,7 +566,7 @@ For **durable execution** that must survive session closure or process restart, - Subagents inherit the parent's enabled toolsets; the model cannot select or widen them per call - **Nested delegation is opt-in** — only `role="orchestrator"` children can delegate further, and only when `max_spawn_depth` is raised from its default of 1 (flat). Disable globally with `orchestrator_enabled: false`. - Leaf subagents **cannot** call: `delegate_task`, `clarify`, `memory`, `send_message`, `cronjob`. Orchestrator subagents retain `delegate_task` but keep the other blocks. Both roles retain `execute_code` (programmatic tool calling) so children can batch mechanical work instead of burning reasoning iterations. -- **Cancellation follows ownership** — `/stop` or closing/resetting the owning session cancels its background children; synchronous descendants under orchestrators follow their parent's interrupt state +- **Cancellation follows ownership** — `/stop` or closing/resetting the owning session ends its background children and the synchronous descendants beneath them; each returns an interrupted completion carrying its partial output - Only the final summary enters the parent's context, keeping token usage efficient - Subagents inherit the parent's **API key, provider configuration, and credential pool** (enabling key rotation on rate limits) From c91d955ff781457beab929c709ca81b198b5b98e Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Fri, 18 Sep 2026 02:46:44 +0800 Subject: [PATCH 50/65] fix(desktop): mark the truncated head of a bot group-chat turn delta --- .../hermes-bots/group-round-members.ts | 5 +-- .../plugins/hermes-bots/group-round-prompt.ts | 19 +++++++- .../plugins/hermes-bots/group-rounds.test.ts | 43 +++++++++++++++++++ 3 files changed, 63 insertions(+), 4 deletions(-) diff --git a/apps/desktop/src/plugins/hermes-bots/group-round-members.ts b/apps/desktop/src/plugins/hermes-bots/group-round-members.ts index 9b9d348690..f8b2664dfe 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-round-members.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-round-members.ts @@ -3,7 +3,6 @@ import { recordGroupActivity } from './group-activity' import { $groupChats, appendGroupChatEntry, - GROUP_CHAT_HISTORY_LIMIT, GROUP_CHAT_MAX_CONTINUATIONS, GROUP_CHAT_MAX_MESSAGES, groupThreadOf, @@ -12,7 +11,7 @@ import { } from './group-chat' import type { GroupChatRoom } from './group-chat' import { groupMemberKey } from './group-membership' -import { buildGroupChatTurnPrompt, formatGroupChatLine } from './group-round-prompt' +import { buildGroupChatTurnPrompt, formatGroupDeltaLines } from './group-round-prompt' import { isGroupPassText, runGroupChatMemberTurn } from './group-turns' import type { Attachment, GroupMember, GroupMessage } from './types' @@ -94,7 +93,7 @@ function prepareGroupRoundMember(context: GroupRoundMemberContext, member: Group groupName: context.group, members, viewer: member, - deltaLines: delta.slice(-GROUP_CHAT_HISTORY_LIMIT).map((e: GroupMessage) => formatGroupChatLine(e, member, context.group)) + deltaLines: formatGroupDeltaLines(delta, member, context.group) }) // Images riding this delta (user attachments — member entries don't diff --git a/apps/desktop/src/plugins/hermes-bots/group-round-prompt.ts b/apps/desktop/src/plugins/hermes-bots/group-round-prompt.ts index 7694facea8..e4c68a6a31 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-round-prompt.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-round-prompt.ts @@ -1,5 +1,5 @@ import { botMentionTag } from './data' -import { groupSpeakerLabel } from './group-chat' +import { GROUP_CHAT_HISTORY_LIMIT, groupSpeakerLabel } from './group-chat' import { groupMemberKey } from './group-membership' import type { GroupMember, GroupMessage, GroupMessageAuthor } from './types' @@ -50,6 +50,23 @@ export function formatGroupChatLine(entry: GroupMessage, viewer: GroupChatLineVi return `${groupSpeakerLabel(entry.from.name, group)}${suffix}${source}: ${relabelMemberControlFrames(entry.text)}${attached}` } +/** #114341: a member's turn renders only the last GROUP_CHAT_HISTORY_LIMIT + * delta lines while the watermark commit advances past the whole tail, so + * the head of an over-long delta is never delivered on any later turn + * either. Mark the cut — without it a member has no way to know its view + * of the room is partial (typically missing the very user instruction + * that started the exchange). */ +export function formatGroupDeltaLines(delta: GroupMessage[], viewer: GroupChatLineViewer, group?: null | string) { + const omitted = delta.length - GROUP_CHAT_HISTORY_LIMIT + const lines = delta.slice(-GROUP_CHAT_HISTORY_LIMIT).map(entry => formatGroupChatLine(entry, viewer, group)) + + if (omitted > 0) { + lines.unshift(`… ${omitted} earlier room message${omitted === 1 ? '' : 's'} omitted since your last turn`) + } + + return lines +} + function viewerNameOf(viewer: GroupChatLineViewer): string { return typeof viewer === 'string' ? viewer : viewer?.name || '' } diff --git a/apps/desktop/src/plugins/hermes-bots/group-rounds.test.ts b/apps/desktop/src/plugins/hermes-bots/group-rounds.test.ts index 36d273a62a..86afa7f3eb 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-rounds.test.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-rounds.test.ts @@ -625,6 +625,49 @@ describe('turn prompt', () => { expect(prompt).toMatch(/never thin out real content/i) expect(prompt).toMatch(/Keep chatter short/i) }) + + // #114341: the rendered window is capped at GROUP_CHAT_HISTORY_LIMIT while + // the watermark commit advances past the whole tail — the head of an + // over-long delta is never delivered on any later turn either, so the cut + // has to be visible in the prompt. + it('marks the head of an over-long delta as omitted', async () => { + await loadRoom() + const { formatGroupDeltaLines } = await import('./group-round-prompt') + const { GROUP_CHAT_HISTORY_LIMIT } = await import('./group-chat') + + const delta: GroupMessage[] = Array.from({ length: GROUP_CHAT_HISTORY_LIMIT + 6 }, (_, i) => ({ + at: i, + from: { kind: 'user', name: 'User' }, + id: `m${i}`, + text: `line ${i}` + })) + + const lines = formatGroupDeltaLines(delta, 'builder', null) + + expect(lines).toHaveLength(GROUP_CHAT_HISTORY_LIMIT + 1) + expect(lines[0]).toBe('… 6 earlier room messages omitted since your last turn') + expect(lines[1]).toContain('line 6') + expect(lines.at(-1)).toContain(`line ${GROUP_CHAT_HISTORY_LIMIT + 5}`) + }) + + it('adds no omission marker when the delta fits the window', async () => { + await loadRoom() + const { formatGroupDeltaLines } = await import('./group-round-prompt') + const { GROUP_CHAT_HISTORY_LIMIT } = await import('./group-chat') + + const delta: GroupMessage[] = Array.from({ length: GROUP_CHAT_HISTORY_LIMIT }, (_, i) => ({ + at: i, + from: { kind: 'user', name: 'User' }, + id: `m${i}`, + text: `line ${i}` + })) + + const lines = formatGroupDeltaLines(delta, 'builder', null) + + expect(lines).toHaveLength(GROUP_CHAT_HISTORY_LIMIT) + expect(lines.every(line => !line.includes('omitted'))).toBe(true) + expect(lines[0]).toContain('line 0') + }) }) describe('attachments', () => { From 0a3fce884647b7872bcd1bc28e66042e12ee78c6 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 01:43:22 -0700 Subject: [PATCH 51/65] test(desktop): pin the omitted-head marker on the real group turn path The salvaged tests exercised the formatGroupDeltaLines helper in isolation, so a call-site regression (a bare slice reappearing in prepareGroupRoundMember) would have stayed green. Drive the room through runGroupChatRounds instead and assert on the prompt the gateway actually received: over-long delta -> marker naming the dropped count and the dropped head absent; delta that fits -> no marker. The helper-level pair is dropped so the file keeps two invariant tests for this fix. --- .../plugins/hermes-bots/group-rounds.test.ts | 88 ++++++++++--------- 1 file changed, 45 insertions(+), 43 deletions(-) diff --git a/apps/desktop/src/plugins/hermes-bots/group-rounds.test.ts b/apps/desktop/src/plugins/hermes-bots/group-rounds.test.ts index 86afa7f3eb..a12a1fbd9d 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-rounds.test.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-rounds.test.ts @@ -393,6 +393,51 @@ describe('per-member delta', () => { expect(room.gateway.calls.at(-1)?.prompt).toContain('unseen-99') }) + // #114341: the turn renders only the last GROUP_CHAT_HISTORY_LIMIT entries + // of the delta while the watermark advances past the whole tail, so the + // head is never delivered later either. The cut must be visible to the + // member (naming how many entries it did not see); a delta that fits + // carries no marker. + it('names the omitted head of an over-long delta in the turn prompt', async () => { + const room = await loadRoom({ turn: () => '(pass)' }) + const members = [MEMBERS[0]] + const thread = room.rounds.sendToGroupChat('Head', members, 'seen-0')! + await settle(room, 'Head') + const limit = room.chat.GROUP_CHAT_HISTORY_LIMIT + + for (let i = 1; i <= limit + 5; i++) { + room.chat.appendGroupChatEntry('Head', { kind: 'user', name: 'You' }, `unseen-${i}`, thread) + } + + const seen = room.chat.$groupChats.get().Head.watermarks[`${thread}::research`] || 0 + const omitted = log(room, 'Head').slice(seen).length - limit + expect(omitted).toBeGreaterThan(0) + + await room.rounds.runGroupChatRounds('Head', members, thread) + const prompt = room.gateway.calls.at(-1)?.prompt || '' + + expect(prompt).toMatch(new RegExp(`${omitted} earlier room messages omitted`)) + expect(prompt).toContain(`unseen-${limit + 5}`) + expect(prompt).not.toContain(`unseen-${omitted}\n`) + }) + + it('adds no omission marker when the delta fits the window', async () => { + const room = await loadRoom({ turn: () => '(pass)' }) + const members = [MEMBERS[0]] + const thread = room.rounds.sendToGroupChat('Fits', members, 'seen-0')! + await settle(room, 'Fits') + + for (let i = 1; i < room.chat.GROUP_CHAT_HISTORY_LIMIT; i++) { + room.chat.appendGroupChatEntry('Fits', { kind: 'user', name: 'You' }, `unseen-${i}`, thread) + } + + await room.rounds.runGroupChatRounds('Fits', members, thread) + const prompt = room.gateway.calls.at(-1)?.prompt || '' + + expect(prompt).toContain('unseen-1') + expect(prompt).not.toMatch(/omitted/) + }) + it('feeds a second send only the NEW messages', async () => { const room = await loadRoom() const member: GroupMember[] = [{ name: 'research', title: '' }] @@ -625,49 +670,6 @@ describe('turn prompt', () => { expect(prompt).toMatch(/never thin out real content/i) expect(prompt).toMatch(/Keep chatter short/i) }) - - // #114341: the rendered window is capped at GROUP_CHAT_HISTORY_LIMIT while - // the watermark commit advances past the whole tail — the head of an - // over-long delta is never delivered on any later turn either, so the cut - // has to be visible in the prompt. - it('marks the head of an over-long delta as omitted', async () => { - await loadRoom() - const { formatGroupDeltaLines } = await import('./group-round-prompt') - const { GROUP_CHAT_HISTORY_LIMIT } = await import('./group-chat') - - const delta: GroupMessage[] = Array.from({ length: GROUP_CHAT_HISTORY_LIMIT + 6 }, (_, i) => ({ - at: i, - from: { kind: 'user', name: 'User' }, - id: `m${i}`, - text: `line ${i}` - })) - - const lines = formatGroupDeltaLines(delta, 'builder', null) - - expect(lines).toHaveLength(GROUP_CHAT_HISTORY_LIMIT + 1) - expect(lines[0]).toBe('… 6 earlier room messages omitted since your last turn') - expect(lines[1]).toContain('line 6') - expect(lines.at(-1)).toContain(`line ${GROUP_CHAT_HISTORY_LIMIT + 5}`) - }) - - it('adds no omission marker when the delta fits the window', async () => { - await loadRoom() - const { formatGroupDeltaLines } = await import('./group-round-prompt') - const { GROUP_CHAT_HISTORY_LIMIT } = await import('./group-chat') - - const delta: GroupMessage[] = Array.from({ length: GROUP_CHAT_HISTORY_LIMIT }, (_, i) => ({ - at: i, - from: { kind: 'user', name: 'User' }, - id: `m${i}`, - text: `line ${i}` - })) - - const lines = formatGroupDeltaLines(delta, 'builder', null) - - expect(lines).toHaveLength(GROUP_CHAT_HISTORY_LIMIT) - expect(lines.every(line => !line.includes('omitted'))).toBe(true) - expect(lines[0]).toContain('line 0') - }) }) describe('attachments', () => { From d32dacf855646a45b948347d106a6b7f2baf7140 Mon Sep 17 00:00:00 2001 From: 686f6c61 Date: Fri, 28 Aug 2026 23:42:47 +0200 Subject: [PATCH 52/65] fix(desktop): mark truncated group-chat sync payloads The ui_meta projection sliced messages at 1200 chars with no signal that the body continued. Stamp truncated and keep the marker inside the existing per-message budget. --- .../plugins/hermes-bots/group-chat.test.ts | 16 +++++ .../src/plugins/hermes-bots/group-chat.ts | 69 +++++++++++++------ apps/desktop/src/plugins/hermes-bots/types.ts | 2 + 3 files changed, 67 insertions(+), 20 deletions(-) diff --git a/apps/desktop/src/plugins/hermes-bots/group-chat.test.ts b/apps/desktop/src/plugins/hermes-bots/group-chat.test.ts index 71feaa6a58..4bf41b58ec 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-chat.test.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-chat.test.ts @@ -502,6 +502,22 @@ describe('gateway mirror', () => { expect(log.length).toBeLessThanOrEqual(16) expect(log.at(-1)?.text).toMatch(/^99:/) expect(log.at(-1)?.text.length).toBeLessThanOrEqual(1200) + expect(log.at(-1)?.truncated).toBe(true) + expect(log.at(-1)?.text).toContain('[truncated]') + expect(log.at(-1)?.text).not.toContain(long) + }) + + it('marks truncated sync text instead of silently slicing it', async () => { + const { chat } = await loadRoom() + const long = `plan:${'x'.repeat(2000)}` + const compacted = chat.compactGroupChatSyncText(long) + + expect(compacted.truncated).toBe(true) + expect(compacted.text).toContain('[truncated]') + expect(compacted.text.length).toBeLessThanOrEqual(1200) + expect(compacted.text.startsWith('plan:')).toBe(true) + expect(compacted.text).not.toBe(long) + expect(chat.compactGroupChatSyncText('short').truncated).toBeUndefined() }) it('preserves threads and budgets escaped Unicode', async () => { diff --git a/apps/desktop/src/plugins/hermes-bots/group-chat.ts b/apps/desktop/src/plugins/hermes-bots/group-chat.ts index 8acb8508b8..06053a130f 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-chat.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-chat.ts @@ -52,6 +52,7 @@ const GROUP_CHAT_SYNC_META_KEY = 'hermes-bots-groups' const GROUP_CHAT_SYNC_MAX_BYTES = 48000 const GROUP_CHAT_SYNC_MESSAGES = 16 const GROUP_CHAT_SYNC_TEXT_CHARS = 1200 +const GROUP_CHAT_SYNC_TRUNCATION_MARK = '… [truncated]' const GROUP_CHAT_SYNC_IMAGE_CHARS = 24000 let groupChatSyncTimer: ReturnType | null = null @@ -91,6 +92,25 @@ const groupChatSyncRetryTimers = new Map>( const groupChatSyncRetryCounts = new Map() export let groupChatSyncDisposed = false +/** Cut one sync-projection line to the per-message budget and mark the cut. + * Receivers used to see a silent mid-sentence slice with no signal that the + * body continued. Keep the mark inside the same char budget so CJK/envelope + * accounting does not grow. */ +export function compactGroupChatSyncText(text: string, limit = GROUP_CHAT_SYNC_TEXT_CHARS) { + const raw = String(text || '') + + if (raw.length <= limit) { + return { text: raw } + } + + const budget = Math.max(0, limit - GROUP_CHAT_SYNC_TRUNCATION_MARK.length) + + return { + text: `${raw.slice(0, budget)}${GROUP_CHAT_SYNC_TRUNCATION_MARK}`, + truncated: true as const + } +} + /** Conservative byte count for the gateway's ensure_ascii JSON encoding. * Python also inserts separator spaces, so reserve one extra byte per JS * structural separator on top of escaped Unicode code-point widths. */ @@ -216,29 +236,38 @@ export function groupChatSyncSnapshot( } for (const [name, room] of ranked) { - const log: GroupMessage[] = room.log.slice(-GROUP_CHAT_SYNC_MESSAGES).map(entry => ({ - ...(entry?.id - ? { - id: String(entry.id).slice(0, 160) - } - : {}), - from: { - kind: entry?.from?.kind === 'member' ? 'member' : 'user', - name: String(entry?.from?.name || (entry?.from?.kind === 'member' ? 'Bot' : 'You')).slice(0, 128), - ...(entry?.from?.source + const log: GroupMessage[] = room.log.slice(-GROUP_CHAT_SYNC_MESSAGES).map(entry => { + const compacted = compactGroupChatSyncText(String(entry?.text || '')) + + return { + ...(entry?.id ? { - source: String(entry.from.source).slice(0, 128) + id: String(entry.id).slice(0, 160) + } + : {}), + from: { + kind: entry?.from?.kind === 'member' ? 'member' : 'user', + name: String(entry?.from?.name || (entry?.from?.kind === 'member' ? 'Bot' : 'You')).slice(0, 128), + ...(entry?.from?.source + ? { + source: String(entry.from.source).slice(0, 128) + } + : {}) + }, + text: compacted.text, + at: Number(entry?.at || 0), + ...(entry?.thread + ? { + thread: String(entry.thread).slice(0, 128) + } + : {}), + ...(compacted.truncated + ? { + truncated: true } : {}) - }, - text: String(entry?.text || '').slice(0, GROUP_CHAT_SYNC_TEXT_CHARS), - at: Number(entry?.at || 0), - ...(entry?.thread - ? { - thread: String(entry.thread).slice(0, 128) - } - : {}) - })) + } + }) const compact: GroupChatSyncRoom = { name: String(name).slice(0, 64), diff --git a/apps/desktop/src/plugins/hermes-bots/types.ts b/apps/desktop/src/plugins/hermes-bots/types.ts index df397e1cb6..c5eff2606d 100644 --- a/apps/desktop/src/plugins/hermes-bots/types.ts +++ b/apps/desktop/src/plugins/hermes-bots/types.ts @@ -159,6 +159,8 @@ export interface GroupMessage { text: string /** Messages predating threading carry the sentinel thread `'legacy'`. */ thread?: string + /** Set on the ui_meta projection when `text` was cut to the sync budget. */ + truncated?: boolean } export interface GroupHold { From d99226015dccf0716a6c3efc7d77aca753007b80 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 01:48:33 -0700 Subject: [PATCH 53/65] fix(desktop): count the head entries the group-chat mirror drops The ui_meta mirror of a room is head-trimmed to 16 messages and then to the 48 KB envelope, silently. It is the only on-disk copy of the room, so an agent that consults it to check what the user said reads a bounded window with no sign that it is bounded (#114341, secondary). Stamp `omitted: N` on the projected room whenever entries fall off the head, recomputed after every byte-budget shift so the envelope bound stays honest, and carry the larger writer's count through the publish merge as a lower bound. Local room state never receives the field: the merge-back into $groupChats builds from explicit fields. --- .../plugins/hermes-bots/group-chat.test.ts | 24 +++++++++++++++++++ .../src/plugins/hermes-bots/group-chat.ts | 24 +++++++++++++++++++ 2 files changed, 48 insertions(+) diff --git a/apps/desktop/src/plugins/hermes-bots/group-chat.test.ts b/apps/desktop/src/plugins/hermes-bots/group-chat.test.ts index 4bf41b58ec..8d5ae2269c 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-chat.test.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-chat.test.ts @@ -538,6 +538,30 @@ describe('gateway mirror', () => { expect(snapshot.rooms['name:Unicode'].log.at(-1)?.thread).toBe('thread-15') }) + // #114341: the mirror is the only on-disk copy of a room. A head trim — + // by message count or by the byte budget — must say how many earlier + // entries it dropped, or a reader concludes the user never said it. + it('counts the head entries the mirror does not carry', async () => { + const { chat } = await loadRoom() + const entry = (index: number, text: string) => ({ at: index, from: { kind: 'user', name: 'You' }, text }) + + const snapshot = chat.groupChatSyncSnapshot({ + Fits: { log: Array.from({ length: 3 }, (_, index) => entry(index, `short ${index}`)) }, + ByCount: { log: Array.from({ length: 40 }, (_, index) => entry(index, `m${index}`)) }, + ByBytes: { log: Array.from({ length: 16 }, (_, index) => entry(index, `${index} ${'🧠'.repeat(1200)}`)) } + } as unknown as Record) + + const byCount = snapshot.rooms['name:ByCount'] + const byBytes = snapshot.rooms['name:ByBytes'] + + expect(snapshot.rooms['name:Fits'].omitted).toBeUndefined() + expect(byCount.log).toHaveLength(16) + expect(byCount.omitted).toBe(24) + expect(byBytes.log.length).toBeLessThan(16) + expect(byBytes.omitted).toBe(16 - byBytes.log.length) + expect(chat.groupChatGatewayJsonSize(snapshot)).toBeLessThanOrEqual(48000) + }) + it('omits empty runtime rooms', async () => { const { chat } = await loadRoom() diff --git a/apps/desktop/src/plugins/hermes-bots/group-chat.ts b/apps/desktop/src/plugins/hermes-bots/group-chat.ts index 06053a130f..4458e667f6 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-chat.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-chat.ts @@ -63,6 +63,9 @@ interface GroupChatSyncRoom { log: GroupMessage[] members?: GroupMember[] name?: string + /** At least this many earlier room entries exist that the projection does + * not carry (head-trimmed to the message/byte budget). */ + omitted?: number revision?: number roomId?: string } @@ -111,6 +114,17 @@ export function compactGroupChatSyncText(text: string, limit = GROUP_CHAT_SYNC_T } } +/** #114341: the ui_meta mirror is the only on-disk copy of a room, so a + * head-trimmed log must say how many earlier entries it does not carry — + * a bare slice reads as "the user never said it". */ +function noteGroupChatSyncOmitted(room: GroupChatSyncRoom, total: number) { + const omitted = total - room.log.length + + if (omitted > 0) { + room.omitted = omitted + } +} + /** Conservative byte count for the gateway's ensure_ascii JSON encoding. * Python also inserts separator spaces, so reserve one extra byte per JS * structural separator on top of escaped Unicode code-point widths. */ @@ -315,9 +329,11 @@ export function groupChatSyncSnapshot( const key = groupChatRoomKey(name, room) rooms[key] = compact + noteGroupChatSyncOmitted(compact, room.log.length) while (compact.log.length > 1 && groupChatGatewayJsonSize(envelope) > GROUP_CHAT_SYNC_MAX_BYTES) { compact.log.shift() + noteGroupChatSyncOmitted(compact, room.log.length) } if (compact.image && groupChatGatewayJsonSize(envelope) > GROUP_CHAT_SYNC_MAX_BYTES) { @@ -447,6 +463,8 @@ export function mergeGroupChatSyncSnapshots( } const remoteRevision = Math.max(0, Number(remoteRoom?.revision || 0)) + // Either writer's head trim is a lower bound on what the union still lacks. + const omitted = Math.max(Number(remoteRoom?.omitted || 0), Number(localRoom?.omitted || 0)) const localRevision = changed.has(key) ? Math.max(0, Number(writeRevision || 0)) @@ -502,6 +520,11 @@ export function mergeGroupChatSyncSnapshots( }), members, revision: Math.max(remoteRevision, localRevision), + ...(omitted > 0 + ? { + omitted + } + : {}), ...(typeof image === 'string' && image ? { image @@ -560,6 +583,7 @@ function groupChatSyncEnvelope( for (const [key, room] of ranked) { while ((room.log?.length || 0) > 1 && groupChatGatewayJsonSize(envelope) > GROUP_CHAT_SYNC_MAX_BYTES) { room.log.shift() + room.omitted = (room.omitted || 0) + 1 } if (room.image && groupChatGatewayJsonSize(envelope) > GROUP_CHAT_SYNC_MAX_BYTES) { From 6c7f693473f3872f830aff6e2d579a6c6fd41118 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 19:30:56 -0700 Subject: [PATCH 54/65] feat(auth): opt out of borrowing Codex CLI / Claude Code logins (auth.adopt_external_logins) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Hermes adopts and refreshes the Codex CLI (~/.codex/auth.json) and Claude Code (~/.claude/.credentials.json) logins automatically whenever its own login is missing or its refresh is rejected. Both providers hand out single-use, rotating refresh tokens, so after an adoption two programs hold one token family and whichever refreshes first logs the other out (#113023). #113816 made the dead logins visible; this adds the switch the reporter asked for. - `auth.adopt_external_logins` (config.yaml, default true — nothing changes for current users). When false: - `read_claude_code_credentials()` — the only reader of the borrowed Claude Code login — returns None, so the resolver fallback, the expired-token refresh, the pool seed/sync and the auxiliary 401 refresher never touch the file; the pool prunes a `claude_code` row an earlier adopting process persisted. - `_recover_codex_tokens_from_cli` returns None for both automatic recovery paths (rejected refresh, half-empty singleton); the real AuthError is surfaced instead. The interactive import offer in `hermes auth add openai-codex` still asks first and is unaffected. - One INFO line per process the first time adoption would have happened; `hermes auth list` / `hermes auth status anthropic|openai-codex` print the same line so the missing borrowed row is explained. - Docs: security.md "Borrowed CLI logins" section; providers.md cross-links. Live (temp HERMES_HOME + CLAUDE_CONFIG_DIR + CODEX_HOME, loopback logging token endpoint): main with the key set to false still POSTed the Claude Code refresh, rewrote the file's refresh token, seeded a claude_code pool row and adopted the Codex CLI pair; on this branch the false arm makes zero Anthropic refresh POSTs, leaves both external files byte-identical, seeds no row and prints the notice, while the default arm is byte-for-byte today's behaviour. --- agent/anthropic_credentials.py | 12 ++- agent/credential_pool.py | 13 ++- agent/credential_sources.py | 33 ++++++++ hermes_cli/auth_codex.py | 8 +- hermes_cli/auth_commands.py | 13 +++ hermes_cli/config_defaults.py | 8 ++ hermes_cli/web_server_config.py | 9 ++ .../test_anthropic_external_login_optout.py | 82 +++++++++++++++++++ tests/hermes_cli/test_auth_codex_self_heal.py | 31 +++++++ website/docs/integrations/providers.md | 7 +- website/docs/user-guide/security.md | 11 +++ 11 files changed, 217 insertions(+), 10 deletions(-) create mode 100644 tests/agent/test_anthropic_external_login_optout.py diff --git a/agent/anthropic_credentials.py b/agent/anthropic_credentials.py index 5deb9ff10b..b03aee1b86 100644 --- a/agent/anthropic_credentials.py +++ b/agent/anthropic_credentials.py @@ -229,8 +229,8 @@ def _read_claude_code_credentials_from_keychain() -> Optional[Dict[str, Any]]: def claude_code_credentials_path() -> Path: """Claude Code's shared OAuth file; every profile reads/writes this same path. Honours ``CLAUDE_CONFIG_DIR`` - like the Claude CLI itself (blank = unset, as in ``hermes_cli.foreign_sessions``), so pointing it at an - empty directory opts a Hermes process out of borrowing the login.""" + like the Claude CLI itself (blank = unset, as in ``hermes_cli.foreign_sessions``). The supported opt-out of + borrowing the login is ``auth.adopt_external_logins: false`` in config.yaml.""" override = os.environ.get("CLAUDE_CONFIG_DIR", "").strip() root = Path(override).expanduser() if override else Path.home() / ".claude" return root / ".credentials.json" @@ -244,7 +244,13 @@ def _read_claude_code_credentials_from_file() -> Optional[Dict[str, Any]]: def read_claude_code_credentials() -> Optional[Dict[str, Any]]: """Read refreshable Claude Code OAuth credentials (Keychain and/or file). When both exist: prefer the only non-expired one (Claude Code 2.1.x refreshes one source but not the other), else the later ``expiresAt`` so a - refresh uses the freshest refreshToken. ~/.claude.json primaryApiKey is deliberately excluded.""" + refresh uses the freshest refreshToken. ~/.claude.json primaryApiKey is deliberately excluded. + + This is the only reader of the borrowed login, so ``auth.adopt_external_logins: false`` is enforced here: + every resolver, pool seed/sync and 401 refresher then sees "no Claude Code login" and never touches the file.""" + from agent.credential_sources import adopt_external_logins_enabled + if not adopt_external_logins_enabled(): + return None kc_creds = _read_claude_code_credentials_from_keychain() file_creds = _read_claude_code_credentials_from_file() if not (kc_creds and file_creds): diff --git a/agent/credential_pool.py b/agent/credential_pool.py index 458c8c62aa..306eb1323e 100644 --- a/agent/credential_pool.py +++ b/agent/credential_pool.py @@ -2182,11 +2182,16 @@ def _seed_anthropic_singletons(seed: _Seeder) -> None: read_claude_code_credentials, read_hermes_oauth_credentials, ) + from agent.credential_sources import adopt_external_logins_enabled - for source_name, creds in ( - ("hermes_pkce", read_hermes_oauth_credentials()), - ("claude_code", read_claude_code_credentials()), - ): + sources = [("hermes_pkce", read_hermes_oauth_credentials())] + if adopt_external_logins_enabled(): + sources.append(("claude_code", read_claude_code_credentials())) + else: + # Singleton-seeded rows are otherwise never pruned; the opt-out must also drop the row an + # earlier (adopting) process persisted, or it keeps rotating a login Hermes no longer reads. + seed.changed |= _retain_sources_not_in(seed.entries, {"claude_code"}) + for source_name, creds in sources: if creds and creds.get("accessToken"): seed.upsert(source_name, { "auth_type": AUTH_TYPE_OAUTH, diff --git a/agent/credential_sources.py b/agent/credential_sources.py index ba90e312f1..7a8ab1644c 100644 --- a/agent/credential_sources.py +++ b/agent/credential_sources.py @@ -7,14 +7,47 @@ Readers live in ``agent.credential_pool``; what is unified here is **removal**: dispatcher suppresses ``(provider, source_id)`` in auth.json so the seeding branch skips the upsert. Adding a source: wire a reader branch in ``_seed_from_*``, gate it behind ``is_source_suppressed``, register a step here. + +Also home to the one policy switch over *borrowed* CLI logins (Codex CLI's +``~/.codex/auth.json``, Claude Code's ``~/.claude/.credentials.json``): +``auth.adopt_external_logins`` in config.yaml. """ from __future__ import annotations +import logging import os from dataclasses import dataclass, field from typing import Callable, List, Optional +logger = logging.getLogger(__name__) + +EXTERNAL_LOGINS_NOT_ADOPTED_NOTICE = ( + "External CLI logins (Codex CLI, Claude Code) are not adopted: auth.adopt_external_logins is false. " + "Hermes uses only its own logins; run `hermes auth add ` to add one." +) +_notice_logged = False + + +def adopt_external_logins_enabled() -> bool: + """``auth.adopt_external_logins`` (default True). + + Codex and Claude OAuth refresh tokens are single-use and rotate, so once Hermes borrows a CLI's + token pair the two programs hold one token family and whichever refreshes first logs the other + out. When the user opts out, Hermes never reads or refreshes those files and says so once per + process (INFO) the first time it would have.""" + global _notice_logged + try: + from hermes_cli.config import load_config_readonly + auth_cfg = (load_config_readonly() or {}).get("auth") + except Exception: + return True + enabled = not isinstance(auth_cfg, dict) or bool(auth_cfg.get("adopt_external_logins", True)) + if not enabled and not _notice_logged: + _notice_logged = True + logger.info(EXTERNAL_LOGINS_NOT_ADOPTED_NOTICE) + return enabled + @dataclass class RemovalResult: diff --git a/hermes_cli/auth_codex.py b/hermes_cli/auth_codex.py index f4c029462b..db55aefe80 100644 --- a/hermes_cli/auth_codex.py +++ b/hermes_cli/auth_codex.py @@ -176,8 +176,14 @@ def _save_codex_tokens(tokens: Dict[str, str], last_refresh: str = None, label: def _recover_codex_tokens_from_cli(reason: str) -> Optional[Dict[str, str]]: - """Adopt a valid Codex CLI token pair into Hermes auth, if available.""" + """Adopt a valid Codex CLI token pair into Hermes auth, if available. + + Automatic adoption only; the interactive import offer in ``_login_openai_codex`` asks first and is + not subject to ``auth.adopt_external_logins``.""" + from agent.credential_sources import adopt_external_logins_enabled from hermes_cli.auth import _import_codex_cli_tokens, _save_codex_tokens + if not adopt_external_logins_enabled(): + return None imported = _import_codex_cli_tokens() # Require BOTH tokens before adopting: persisting a payload without a usable refresh_token # would only break the next refresh cycle. diff --git a/hermes_cli/auth_commands.py b/hermes_cli/auth_commands.py index 7a295f5952..09f520982a 100644 --- a/hermes_cli/auth_commands.py +++ b/hermes_cli/auth_commands.py @@ -28,6 +28,8 @@ _OAUTH_CAPABLE_PROVIDERS = {"anthropic", "nous", "openai-codex", "xai-oauth", "q # ...and default to it when ``--type`` is omitted. OpenRouter stays API-key-first: the documented # ``hermes auth add openrouter --api-key sk-or-...`` must keep working with no ``--type``. _OAUTH_DEFAULT_PROVIDERS = _OAUTH_CAPABLE_PROVIDERS - {"openrouter"} +# Providers whose sibling CLI login Hermes may borrow (``auth.adopt_external_logins``). +EXTERNAL_LOGIN_PROVIDERS = {"anthropic", "openai-codex"} def _get_custom_provider_entries() -> list[dict]: @@ -491,6 +493,15 @@ def auth_list_command(args) -> None: ) print(row.rstrip()) print() + if not provider_filter or provider_filter in EXTERNAL_LOGIN_PROVIDERS: + _print_external_login_notice() + + +def _print_external_login_notice() -> None: + """One line telling the user why no Codex CLI / Claude Code login shows up when adoption is off.""" + from agent.credential_sources import EXTERNAL_LOGINS_NOT_ADOPTED_NOTICE, adopt_external_logins_enabled + if not adopt_external_logins_enabled(): + print(EXTERNAL_LOGINS_NOT_ADOPTED_NOTICE) def auth_remove_command(args) -> None: @@ -607,6 +618,8 @@ def auth_status_command(args) -> None: if not status.get("logged_in"): reason = status.get("error") print(f"{provider}: logged out" + (f" ({reason})" if reason else "")) + if provider in EXTERNAL_LOGIN_PROVIDERS: + _print_external_login_notice() return print(f"{provider}: logged in") for key in ("auth_type", "client_id", "redirect_uri", "scope", "expires_at", "api_base_url"): diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 80a3bfbafd..5f0970c393 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1644,6 +1644,14 @@ DEFAULT_CONFIG = { # Custom personalities: {"name": "system prompt"} or {"name": {"description", "system_prompt", # "tone", "style"}}. "personalities": {}, + "auth": { # Login policy (credentials themselves live in auth.json / .env). + # Borrow and refresh the Codex CLI (~/.codex/auth.json) and Claude Code (~/.claude/.credentials.json) + # logins automatically when Hermes has no usable login of its own. Their refresh tokens are single-use + # and rotate, so two programs on one login can log each other out; set false to make Hermes use only + # its own logins (`hermes auth add `). `hermes auth add openai-codex` still offers the import + # interactively. + "adopt_external_logins": True, + }, "security": { # Security: pre-exec scanning via tirith plus related guards. "allow_private_urls": False, # allow requests to private/internal IPs (OpenWrt, VPNs) # CIDR blocks a local TUN proxy answers DNS with (Mihomo/Clash fake-ip, Surge enhanced). diff --git a/hermes_cli/web_server_config.py b/hermes_cli/web_server_config.py index b5fd93f441..b502ab914f 100644 --- a/hermes_cli/web_server_config.py +++ b/hermes_cli/web_server_config.py @@ -101,6 +101,14 @@ _SCHEMA_OVERRIDES: Dict[str, Dict[str, Any]] = { "description": "Refuse Docker sandboxes when egress is enabled but not configured/running", "category": "security", }, + "auth.adopt_external_logins": { + "type": "boolean", + "description": ( + "Borrow and refresh the Codex CLI / Claude Code logins when Hermes has no usable login of its own. " + "Off: Hermes uses only its own logins (`hermes auth add `)." + ), + "category": "security", + }, "tts.provider": _select( "Text-to-speech provider", "edge", "elevenlabs", "openai", "xai", "minimax", "mistral", "gemini", "neutts", "kittentts", "piper", @@ -197,6 +205,7 @@ _CATEGORY_MERGE: Dict[str, str] = { "session": "general", "nous": "agent", "connections": "agent", + "auth": "security", } diff --git a/tests/agent/test_anthropic_external_login_optout.py b/tests/agent/test_anthropic_external_login_optout.py new file mode 100644 index 0000000000..72a3fafa1b --- /dev/null +++ b/tests/agent/test_anthropic_external_login_optout.py @@ -0,0 +1,82 @@ +"""``auth.adopt_external_logins: false`` keeps Hermes off the Claude Code login (#113023). + +Claude Code's OAuth refresh token is single-use: once Hermes borrows and refreshes it, Hermes and +Claude Code hold one token family and whichever refreshes first logs the other out. With the opt-out +set, Hermes must neither read nor refresh ``~/.claude/.credentials.json``, must drop the pool row an +earlier adopting process persisted, and must say so in ``hermes auth list``. +""" +from __future__ import annotations + +import io +import json +import time +from contextlib import redirect_stdout +from types import SimpleNamespace +from unittest import mock + +import urllib.request + +from agent import anthropic_credentials as ac +from agent import credential_sources +from agent.credential_pool import load_pool +from hermes_cli.auth_commands import auth_list_command + + +def _write_config(hermes_home, adopt): + body = "model:\n provider: anthropic\n model: claude-sonnet-4-5\n" + if adopt is not None: + body += f"auth:\n adopt_external_logins: {'true' if adopt else 'false'}\n# arm {adopt}\n" + (hermes_home / "config.yaml").write_text(body) + + +def test_opt_out_never_reads_or_refreshes_claude_code_login(tmp_path, monkeypatch): + hermes_home, claude_dir = tmp_path / "hermes", tmp_path / "claude" + hermes_home.mkdir() + claude_dir.mkdir() + (hermes_home / ".env").write_text("") + monkeypatch.setenv("HERMES_HOME", str(hermes_home)) + monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(claude_dir)) + monkeypatch.setattr(credential_sources, "_notice_logged", False, raising=False) + cred_file = claude_dir / ".credentials.json" + cred_file.write_text(json.dumps({"claudeAiOauth": { + "accessToken": "sk-ant-oat01-expired", "refreshToken": "sk-ant-ort01-cc", + "expiresAt": int(time.time() * 1000) - 3_600_000, "scopes": ["user:inference"]}})) + original_bytes = cred_file.read_bytes() + + posts: list = [] + + def fake_urlopen(req, timeout=None): + if "oauth/token" in req.full_url: # the Anthropic token endpoint; other seeders probe other hosts + posts.append(req.full_url) + body = json.dumps({"access_token": "sk-ant-oat01-fresh", "refresh_token": "sk-ant-ort01-fresh", + "expires_in": 3600}).encode() + resp = mock.MagicMock() + resp.read.return_value = body + resp.__enter__.return_value = resp + return resp + + monkeypatch.setattr(urllib.request, "urlopen", fake_urlopen) + + # Arm A (default): today's behaviour — the borrowed login is seeded into the pool. + _write_config(hermes_home, adopt=None) + assert [e.source for e in load_pool("anthropic").entries()] == ["claude_code"] + + # Arm B (opt-out): no read, no refresh POST, the persisted row is dropped, the status line explains why. + _write_config(hermes_home, adopt=False) + assert ac.resolve_anthropic_token() is None + assert posts == [] + assert cred_file.read_bytes() == original_bytes + assert [e.source for e in load_pool("anthropic").entries()] == [] + out = io.StringIO() + with redirect_stdout(out): + auth_list_command(SimpleNamespace(provider=None)) + assert credential_sources.EXTERNAL_LOGINS_NOT_ADOPTED_NOTICE in out.getvalue() + + # Arm A again: flipping back adopts (and refreshes) exactly as before. + _write_config(hermes_home, adopt=True) + assert ac.resolve_anthropic_token() == "sk-ant-oat01-fresh" + assert len(posts) == 1 + out = io.StringIO() + with redirect_stdout(out): + auth_list_command(SimpleNamespace(provider=None)) + assert credential_sources.EXTERNAL_LOGINS_NOT_ADOPTED_NOTICE not in out.getvalue() diff --git a/tests/hermes_cli/test_auth_codex_self_heal.py b/tests/hermes_cli/test_auth_codex_self_heal.py index 4c92cde2f5..fa6182d7cb 100644 --- a/tests/hermes_cli/test_auth_codex_self_heal.py +++ b/tests/hermes_cli/test_auth_codex_self_heal.py @@ -96,3 +96,34 @@ def test_self_heals_missing_singleton_access_token_from_codex_cli(tmp_path, monk assert tokens["refresh_token"] == "fresh-refresh" + + +def test_opt_out_never_adopts_codex_cli_login(tmp_path, monkeypatch): + """``auth.adopt_external_logins: false`` (#113023): the Codex CLI pair is a single-use refresh-token + family the user did not hand to Hermes. Both automatic recovery paths must leave it (and Hermes' own + auth.json) untouched and surface the real error instead.""" + hermes_home = tmp_path / "hermes" + codex_home = tmp_path / "codex" + hermes_home.mkdir() + codex_home.mkdir() + (hermes_home / "config.yaml").write_text("auth:\n adopt_external_logins: false\n") + hermes_auth = {"version": 1, "providers": {"openai-codex": { + "tokens": {"refresh_token": "stale-refresh"}, "auth_mode": "chatgpt"}}} + (hermes_home / "auth.json").write_text(json.dumps(hermes_auth)) + (codex_home / "auth.json").write_text(json.dumps({ + "tokens": {"access_token": "fresh-access", "refresh_token": "fresh-refresh"}})) + monkeypatch.setenv("HERMES_HOME", str(hermes_home)) + monkeypatch.setenv("CODEX_HOME", str(codex_home)) + + with pytest.raises(AuthError) as info: + resolve_codex_runtime_credentials() + assert info.value.code == "codex_auth_missing_access_token" + + def _rejected(*_a, **_k): + raise AuthError("bad", provider="openai-codex", code="invalid_grant", relogin_required=True) + + monkeypatch.setattr(auth_codex, "refresh_codex_oauth_pure", _rejected) + with pytest.raises(AuthError) as info: + _refresh_codex_auth_tokens(dict(STALE), 5.0) + assert info.value.relogin_required # surfaced, not papered over with the CLI pair + assert json.loads((hermes_home / "auth.json").read_text()) == hermes_auth diff --git a/website/docs/integrations/providers.md b/website/docs/integrations/providers.md index da7ef399bb..0632b7ad35 100644 --- a/website/docs/integrations/providers.md +++ b/website/docs/integrations/providers.md @@ -91,7 +91,7 @@ Don't have a subscription yet? Get one at [portal.nousresearch.com/manage-subscr :::info Codex Note -The OpenAI Codex provider authenticates via device code (open a URL, enter a code). Hermes stores the resulting credentials in its own auth store under `~/.hermes/auth.json` and can import existing Codex CLI credentials from `~/.codex/auth.json` when present. No Codex CLI installation is required. +The OpenAI Codex provider authenticates via device code (open a URL, enter a code). Hermes stores the resulting credentials in its own auth store under `~/.hermes/auth.json` and can import existing Codex CLI credentials from `~/.codex/auth.json` when present. No Codex CLI installation is required. Automatic adoption of the Codex CLI login (when Hermes' own refresh fails) is controlled by `auth.adopt_external_logins` — see [Borrowed CLI logins](/docs/user-guide/security#borrowed-cli-logins). If a token refresh fails with a terminal error (HTTP 4xx, `invalid_grant`, revoked grant, etc.), Hermes marks the refresh token as dead and stops replaying it so you don't see a flood of identical auth failures. The next request surfaces a typed re-auth message instead. Run `hermes auth add openai-codex` (or `hermes model` → **ChatGPT or Codex Subscription**) to start a fresh device-code login; the quarantine clears on the next successful exchange. @@ -163,7 +163,10 @@ Use Claude models directly through the Anthropic API — no OpenRouter proxy nee When no explicit environment credential is selected, Hermes-owned OAuth grants in the credential pool take precedence over a borrowed Claude Code login. The -borrowed login remains the fallback when no owned OAuth grant is available. +borrowed login remains the fallback when no owned OAuth grant is available — +unless `auth.adopt_external_logins: false` is set, in which case Hermes never +reads or refreshes Claude Code's credentials (see +[Borrowed CLI logins](/docs/user-guide/security#borrowed-cli-logins)). Auxiliary authentication recovery refreshes the credential used by the failed request, not an unrelated ambient login; rotating a borrowed login can otherwise invalidate its owner's refresh token. diff --git a/website/docs/user-guide/security.md b/website/docs/user-guide/security.md index d1e8526d53..9b1b77b138 100644 --- a/website/docs/user-guide/security.md +++ b/website/docs/user-guide/security.md @@ -665,6 +665,17 @@ terminal: Paths are relative to `~/.hermes/`. Files are mounted to `/root/.hermes/` inside the container. This list is read by `tools/credential_files.py` (`terminal.credential_files`) — it lives under the `terminal:` block but is loaded by the credential-files module, not the core terminal backend, so it isn't part of the bundled `DEFAULT_CONFIG` snapshot. +### Borrowed CLI logins (Codex CLI, Claude Code) {#borrowed-cli-logins} + +When Hermes has no usable login of its own for `openai-codex` or `anthropic`, it can borrow the Codex CLI's `~/.codex/auth.json` and Claude Code's `~/.claude/.credentials.json` (or Keychain entry) and refresh them on your behalf. Both use single-use, rotating refresh tokens: once two programs hold one token family, whichever refreshes first invalidates the other's copy, which shows up as "I logged in once in the terminal and Hermes keeps failing" (or the reverse). If you run those CLIs alongside Hermes, give Hermes its own login and turn adoption off: + +```yaml +auth: + adopt_external_logins: false # default: true +``` + +With the switch off Hermes never reads or refreshes those files: the `claude_code` credential-pool row disappears, `hermes auth list` prints one line saying so, and the log carries one INFO line per process. Only automatic adoption is affected — `hermes auth add openai-codex` still asks before importing an existing Codex CLI login. Add your own logins with `hermes auth add anthropic` / `hermes auth add openai-codex`. + ### What Each Sandbox Filters | Sandbox | Default Filter | Passthrough Override | From 0cd13aaf74145cc2b46b175e78f7220dd23b7690 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 20:09:24 -0700 Subject: [PATCH 55/65] docs: relative link to the borrowed-logins section (route-style links fail the docs check) --- website/docs/integrations/providers.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/website/docs/integrations/providers.md b/website/docs/integrations/providers.md index 0632b7ad35..785dbb8256 100644 --- a/website/docs/integrations/providers.md +++ b/website/docs/integrations/providers.md @@ -91,7 +91,7 @@ Don't have a subscription yet? Get one at [portal.nousresearch.com/manage-subscr :::info Codex Note -The OpenAI Codex provider authenticates via device code (open a URL, enter a code). Hermes stores the resulting credentials in its own auth store under `~/.hermes/auth.json` and can import existing Codex CLI credentials from `~/.codex/auth.json` when present. No Codex CLI installation is required. Automatic adoption of the Codex CLI login (when Hermes' own refresh fails) is controlled by `auth.adopt_external_logins` — see [Borrowed CLI logins](/docs/user-guide/security#borrowed-cli-logins). +The OpenAI Codex provider authenticates via device code (open a URL, enter a code). Hermes stores the resulting credentials in its own auth store under `~/.hermes/auth.json` and can import existing Codex CLI credentials from `~/.codex/auth.json` when present. No Codex CLI installation is required. Automatic adoption of the Codex CLI login (when Hermes' own refresh fails) is controlled by `auth.adopt_external_logins` — see [Borrowed CLI logins](../user-guide/security.md#borrowed-cli-logins). If a token refresh fails with a terminal error (HTTP 4xx, `invalid_grant`, revoked grant, etc.), Hermes marks the refresh token as dead and stops replaying it so you don't see a flood of identical auth failures. The next request surfaces a typed re-auth message instead. Run `hermes auth add openai-codex` (or `hermes model` → **ChatGPT or Codex Subscription**) to start a fresh device-code login; the quarantine clears on the next successful exchange. @@ -166,7 +166,7 @@ in the credential pool take precedence over a borrowed Claude Code login. The borrowed login remains the fallback when no owned OAuth grant is available — unless `auth.adopt_external_logins: false` is set, in which case Hermes never reads or refreshes Claude Code's credentials (see -[Borrowed CLI logins](/docs/user-guide/security#borrowed-cli-logins)). +[Borrowed CLI logins](../user-guide/security.md#borrowed-cli-logins)). Auxiliary authentication recovery refreshes the credential used by the failed request, not an unrelated ambient login; rotating a borrowed login can otherwise invalidate its owner's refresh token. From 0c3ebefb5e02150026f30c555bac58628787e511 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 19:46:16 -0700 Subject: [PATCH 56/65] fix(desktop): /reasoning hide|show gates Thinking blocks immediately MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `display.show_reasoning` is a client-side display decision since 40f2368875a, and the Desktop transcript honors it through `$showReasoning` (#115335). The composer path did not: `/reasoning` was marked `desktop="advanced"` in the Python slash registry, so the Desktop refused it as "not available", while a gateway-routed `/reasoning hide` would only have written config.yaml and left the atom waiting for the next config refresh. Route `/reasoning` to the gateway's `config.set key=reasoning` — the Ink TUI's path (ui-tui/src/app/slash/commands/session.ts) — and mirror the answer: `hide`/`show` flip `$showReasoning` on the spot, effort levels leave the gate alone and get session scope (a throwaway slash-worker CLI could never pin the live session's effort). No arg reports the current effort and display, like Ink. Drops the registry's `advanced` marker and regenerates the desktop dump. Live (real `hermes serve`, scratch HERMES_HOME, real WebSocket): config.set key=reasoning value=hide -> {value: hide}, config.yaml show_reasoning=False; show -> True; high -> effort only, display untouched. Fixes #111761 --- .../session/hooks/use-prompt-actions/slash.ts | 41 ++++++++++++++ .../desktop/src/lib/desktop-slash-commands.ts | 7 +++ .../src/lib/desktop-slash-registry.json | 1 - apps/desktop/src/lib/reasoning-slash.test.ts | 36 +++++++++++++ apps/desktop/src/lib/reasoning-slash.ts | 53 +++++++++++++++++++ hermes_cli/commands.py | 3 +- website/docs/user-guide/desktop.md | 2 +- 7 files changed, 139 insertions(+), 4 deletions(-) create mode 100644 apps/desktop/src/lib/reasoning-slash.test.ts create mode 100644 apps/desktop/src/lib/reasoning-slash.ts diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/slash.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/slash.ts index db87364455..876ea7d9f3 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/slash.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/slash.ts @@ -17,6 +17,7 @@ import { resolveDesktopCommand } from '@/lib/desktop-slash-commands' import { isMissingRpcMethod } from '@/lib/gateway-rpc' +import { applyReasoningSlashResult, reasoningSlashParams } from '@/lib/reasoning-slash' import { setSessionYolo } from '@/lib/yolo-session' import { openCommandPalettePage } from '@/store/command-palette' import { setComposerDraft } from '@/store/composer' @@ -771,6 +772,46 @@ export function useSlashCommand(deps: SlashCommandDeps) { compressInFlightRef.current.delete(sessionId) } }, + // /reasoning runs the gateway's `config.set key=reasoning` — the Ink + // TUI's path. Through slash.exec the display words only reached + // config.yaml and the Thinking gate ($showReasoning) waited for the + // next config refresh; the effort level was set on a throwaway CLI. + reasoning: async ctx => { + const resolved = await withSlashOutput(ctx) + + if (!resolved) { + return + } + + const { render: renderSlashOutput, sessionId } = resolved + const params = reasoningSlashParams(ctx.arg, sessionId) + + try { + if (!params) { + const current = await requestGateway<{ display?: string; value?: string }>('config.get', { + key: 'reasoning', + session_id: sessionId + }) + + renderSlashOutput(`reasoning: ${current.value || 'medium'} · display ${current.display || 'hide'}`) + + return + } + + const result = await requestGateway<{ value?: string }>('config.set', params) + + applyReasoningSlashResult(result.value) + renderSlashOutput(`reasoning: ${result.value || params.value}`) + } catch (err) { + if (isMissingRpcMethod(err)) { + await runExec(ctx) + + return + } + + renderSlashOutput(`error: ${err instanceof Error ? err.message : String(err)}`) + } + }, // /yolo maps to the status-bar YOLO control — a per-session approval // bypass, same scope as the TUI's Shift+Tab. With no session yet we arm // it locally; the session-create path applies it on the first message. diff --git a/apps/desktop/src/lib/desktop-slash-commands.ts b/apps/desktop/src/lib/desktop-slash-commands.ts index f1decc521d..b3e3668c43 100644 --- a/apps/desktop/src/lib/desktop-slash-commands.ts +++ b/apps/desktop/src/lib/desktop-slash-commands.ts @@ -66,6 +66,7 @@ export type DesktopActionId = | 'new' | 'pet' | 'profile' + | 'reasoning' | 'skin' | 'stop' | 'title' @@ -188,6 +189,12 @@ const DESKTOP_COMMAND_SPECS: readonly DesktopCommandSpec[] = [ surface: action('branch') }, { name: '/yolo', description: 'Toggle YOLO — auto-approve dangerous commands', surface: action('yolo') }, + { + name: '/reasoning', + description: 'Reasoning effort or display [ [--global]|show|hide|full|clamp]', + surface: action('reasoning'), + argumentMode: 'options' + }, { name: '/wake', description: 'Control the desktop wake-word listener [on|off|status]', diff --git a/apps/desktop/src/lib/desktop-slash-registry.json b/apps/desktop/src/lib/desktop-slash-registry.json index d8a05d279d..91917627df 100644 --- a/apps/desktop/src/lib/desktop-slash-registry.json +++ b/apps/desktop/src/lib/desktop-slash-registry.json @@ -22,7 +22,6 @@ "/platforms": "terminal", "/plugins": "terminal", "/quit": "terminal", - "/reasoning": "advanced", "/redraw": "terminal", "/reload": "terminal", "/reload-mcp": "advanced", diff --git a/apps/desktop/src/lib/reasoning-slash.test.ts b/apps/desktop/src/lib/reasoning-slash.test.ts new file mode 100644 index 0000000000..755e6944ea --- /dev/null +++ b/apps/desktop/src/lib/reasoning-slash.test.ts @@ -0,0 +1,36 @@ +import { describe, expect, it } from 'vitest' + +import { resolveDesktopCommand } from '@/lib/desktop-slash-commands' +import { applyReasoningSlashResult, reasoningSlashParams } from '@/lib/reasoning-slash' +import { $showReasoning, setShowReasoningFromConfig } from '@/store/reasoning-disclosure' + +// `/reasoning hide` typed into the Desktop composer must gate Thinking blocks +// in the transcript immediately (#111761, #49664): the renderer mirrors the +// gateway's `config.set key=reasoning` answer instead of waiting for the next +// config refresh. +describe('/reasoning slash command', () => { + it('runs as a desktop action instead of the slash worker', () => { + expect(resolveDesktopCommand('/reasoning')?.surface.kind).toBe('action') + }) + + it('builds the gateway config.set payload, honoring the scope flags', () => { + expect(reasoningSlashParams('hide', 's1')).toEqual({ key: 'reasoning', session_id: 's1', value: 'hide' }) + expect(reasoningSlashParams('high --global', 's1')).toEqual({ + key: 'reasoning', + scope: 'global', + session_id: 's1', + value: 'high' + }) + expect(reasoningSlashParams(' ', 's1')).toBeNull() + }) + + it('flips the Thinking gate on hide/show and leaves it alone for effort levels', () => { + setShowReasoningFromConfig(true) + applyReasoningSlashResult('hide') + expect($showReasoning.get()).toBe(false) + applyReasoningSlashResult('high') + expect($showReasoning.get()).toBe(false) + applyReasoningSlashResult('show') + expect($showReasoning.get()).toBe(true) + }) +}) diff --git a/apps/desktop/src/lib/reasoning-slash.ts b/apps/desktop/src/lib/reasoning-slash.ts new file mode 100644 index 0000000000..71f8c5ac49 --- /dev/null +++ b/apps/desktop/src/lib/reasoning-slash.ts @@ -0,0 +1,53 @@ +import { setShowReasoningFromConfig } from '@/store/reasoning-disclosure' + +// `/reasoning` typed into the composer — the Ink TUI's path +// (ui-tui/src/app/slash/commands/session.ts). The gateway's `config.set +// key=reasoning` handles both the display words (show/hide/full/clamp) and the +// effort levels; through `slash.exec` the words only reached config.yaml and +// the transcript kept rendering Thinking until the next config refresh. + +const GLOBAL_FLAGS = new Set(['--global', '-g', 'global']) +const SESSION_FLAGS = new Set(['--session', '-s', 'session']) + +export type ReasoningSlashParams = { + key: 'reasoning' + scope?: 'global' | 'session' + session_id: string + value: string +} + +/** Build the `config.set` params for `/reasoning `; `null` when there is nothing to set. */ +export function reasoningSlashParams(arg: string, sessionId: string): null | ReasoningSlashParams { + let scope: ReasoningSlashParams['scope'] + const values: string[] = [] + + for (const part of arg.trim().split(/\s+/).filter(Boolean)) { + const flag = part.toLowerCase() + + if (GLOBAL_FLAGS.has(flag)) { + scope = 'global' + } else if (SESSION_FLAGS.has(flag)) { + scope = 'session' + } else { + values.push(part) + } + } + + if (!values.length) { + return null + } + + return { key: 'reasoning', session_id: sessionId, value: values.join(' '), ...(scope ? { scope } : {}) } +} + +/** + * Mirror the gateway's answer into the renderer: `hide`/`show` are the + * `display.show_reasoning` words; effort levels leave the display gate alone. + */ +export function applyReasoningSlashResult(value: unknown): void { + if (value === 'hide') { + setShowReasoningFromConfig(false) + } else if (value === 'show') { + setShowReasoningFromConfig(true) + } +} diff --git a/hermes_cli/commands.py b/hermes_cli/commands.py index e87d9b4699..47fd759bb8 100644 --- a/hermes_cli/commands.py +++ b/hermes_cli/commands.py @@ -187,8 +187,7 @@ COMMAND_REGISTRY: list[CommandDef] = [ subcommands=("manual", "smart", "off")), CommandDef("reasoning", "Manage reasoning effort and display", "Configuration", args_hint="[level|show|hide|full|clamp] [--global]", - subcommands=("none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", "show", "hide", "on", "off", "full", "clamp", "--global"), - desktop="advanced"), + subcommands=("none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", "show", "hide", "on", "off", "full", "clamp", "--global")), CommandDef("fast", "Fast mode — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)", "Configuration", args_hint="[normal|fast|auto|cold|status] [--global]", subcommands=("normal", "fast", "auto", "cold", "status", "on", "off", "--global"), diff --git a/website/docs/user-guide/desktop.md b/website/docs/user-guide/desktop.md index fc77e3ce87..8beec726cf 100644 --- a/website/docs/user-guide/desktop.md +++ b/website/docs/user-guide/desktop.md @@ -196,7 +196,7 @@ Manage providers, models, tools, and credentials from a real UI instead of editi - **xAI Grok OAuth** — Grok is a first-class OAuth provider in the launcher; sign in through the browser flow like the other OAuth providers. - **Tool-backend installs from the GUI** — run a tool backend's post-setup install steps directly from the app instead of dropping to a terminal. In the terminal backend picker, selecting a backend marked **Needs setup** asks for confirmation first; declining leaves the current backend selected. - **Terminal font picker** — choose an installed font in **Settings → Appearance**. Nerd Fonts such as `MesloLGS NF` render Powerlevel10k separators and icons in both interactive and agent terminals; the setting is saved per profile. -- **Reasoning Blocks** — **Settings → Chat → Reasoning Blocks** (`display.show_reasoning` in `config.yaml`) shows or hides the model's thinking in the transcript (off shows answers only). Open chats update as soon as the setting saves. +- **Reasoning Blocks** — **Settings → Chat → Reasoning Blocks** (`display.show_reasoning` in `config.yaml`) shows or hides the model's thinking in the transcript (off shows answers only). Open chats update as soon as the setting saves. Typing `/reasoning hide` or `/reasoning show` in the composer flips the same setting and the open transcript follows immediately. - **Reopen Last Chat on Launch** — by default the app picks up where you left off on cold start. Turn it off in **Settings → Appearance** (or set `display.resume_last_session: false` in `config.yaml`) to always begin with a fresh chat. Deep links and explicit destinations are never overridden either way. - **Auxiliary-model warning** — if you switch the main model to a new provider while auxiliary tasks (titling, summarization, and similar helpers) are still pinned to another provider, the app warns you so you don't unknowingly split work across two providers. - **Per-task reasoning effort** — each row under **Settings → Model → Auxiliary models** has a reasoning selector next to its provider/model pick: a level, **Off**, or **inherit · main model effort** (the default, which removes the task's override). It is saved as `auxiliary..reasoning_effort` in `config.yaml`, the same key `hermes model` writes, and shows in the row's summary when set. Use it to run frequent helpers such as compression or titling at low or no reasoning while the main agent stays at high. From acccd54c67c50ac6082e7a7bcaddc5b19166f754 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 19:54:34 -0700 Subject: [PATCH 57/65] fix(desktop): restore the unsent composer draft when a session is gone MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When a Desktop tab references a session id that exists in no profile DB (deleted elsewhere, or a stale id after a profile rename / wiped backend), the resume path already drops the window to a fresh draft without toasting or hot-looping (62af32efe7c, bounded by goneSessionVerdict). But the text the user had typed into that session stayed stashed under the dead key in `hermes:composer-drafts:v3` — not lost, yet invisible, because nothing ever opens that key again. From the user's side the message just vanished. The gone verdict now announces the dead key (`announceGoneSessionDraft`), and the composer's swap onto the fresh-draft scope consumes it once (`adoptGoneSessionDraft`): the stash moves into the new-chat bucket, the composer is seeded with it, and an inline "Restored your unsent message" strip with Undo appears above the input. Undo returns the text to the dead key while it is unchanged (still recoverable by the same path); once edited it only dismisses. Keyed on the one-shot announcement rather than on the composer observing an id→fresh transition, because the composer remounts across the drop (a loading route mounts no composer). A fresh draft that already has text is never clobbered; an empty stash shows nothing. Offer, don't hijack (apps/desktop/AGENTS.md): no navigation beyond the drop that already happened, no focus move, no toast. Live (headless Electron over CDP, worktree backend): type into a live session → delete its row + restart the backend → reload at the dead id. Before: route drops to #/, composer empty, text only in localStorage under the dead key. After: same drop, composer holds the text, strip + Undo shown; Undo empties the composer and puts the text back under the key; a second open of the dead id shows no strip; a dead id with no stash shows nothing. Fixes #111868 --- apps/desktop/AGENTS.md | 9 ++ .../hooks/use-composer-draft.test.tsx | 39 +++++++ .../chat/composer/hooks/use-composer-draft.ts | 9 ++ apps/desktop/src/app/chat/composer/index.tsx | 6 + .../chat/composer/restored-draft-notice.tsx | 63 +++++++++++ .../hooks/use-session-actions/index.ts | 7 +- apps/desktop/src/i18n/ar.ts | 2 + apps/desktop/src/i18n/en.ts | 2 + apps/desktop/src/i18n/ja.ts | 2 + apps/desktop/src/i18n/ru.ts | 2 + apps/desktop/src/i18n/types.ts | 2 + apps/desktop/src/i18n/zh-hant.ts | 2 + apps/desktop/src/i18n/zh.ts | 2 + apps/desktop/src/store/composer.test.ts | 36 ++++++ apps/desktop/src/store/composer.ts | 103 ++++++++++++++++++ website/docs/user-guide/desktop.md | 1 + 16 files changed, 286 insertions(+), 1 deletion(-) create mode 100644 apps/desktop/src/app/chat/composer/restored-draft-notice.tsx diff --git a/apps/desktop/AGENTS.md b/apps/desktop/AGENTS.md index ae540cca9d..34f6bab173 100644 --- a/apps/desktop/AGENTS.md +++ b/apps/desktop/AGENTS.md @@ -69,6 +69,15 @@ rename sibling of `dropTilesForProfile`. Add any new profile-keyed localStorage family to BOTH, or a rename leaves it pointing at a backend that no longer exists ("Couldn't open this session" on every restore). +When an id is verifiably gone anyway (`goneSessionVerdict` → `'draft'`), the +window drops to a fresh draft without toasting or looping — and the unsent +text stashed under the dead key follows it: the verdict calls +`announceGoneSessionDraft(id)` and the composer's swap onto the fresh scope +consumes it once (`adoptGoneSessionDraft`, `store/composer.ts`), seeding the +composer and publishing the inline, undoable `$restoredDraftNotice`. Offer, +don't hijack: no navigation beyond the drop itself, no focus steal, no toast, +and an already non-empty fresh draft is never clobbered. + ## Server truth is cached, not owned The renderer paints from a cache of backend truth, so it must reconcile, not diff --git a/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.test.tsx b/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.test.tsx index c413d81be7..50c61d19da 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.test.tsx +++ b/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.test.tsx @@ -4,9 +4,12 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import { PaneVisibleContext } from '@/components/pane-shell/pane-visibility' import { + $restoredDraftNotice, + announceGoneSessionDraft, announceNewSessionDraftKey, clearSessionDraft, type ComposerAttachment, + dismissRestoredDraftNotice, mainComposerScope, stashSessionDraft, takeSessionDraft @@ -122,6 +125,42 @@ describe('useComposerDraft — attachment scope stays coherent with the committe clearSessionDraft('session-created') }) + it('carries the unsent draft of a GONE session into the fresh chat once, with an undoable notice (#111868)', () => { + stashSessionDraft('session-gone', 'typed into a session that no longer exists', []) + + const { rerender } = render( + undefined} sessionId="session-gone" /> + ) + + // The resume's gone verdict announces the dead key, then drops the window + // to a fresh draft (route → /new, scope → the pre-session bucket). + announceGoneSessionDraft('session-gone') + act(() => { + rerender( undefined} sessionId="" />) + }) + + expect(takeSessionDraft(null).text).toBe('typed into a session that no longer exists') + expect(takeSessionDraft('session-gone').text).toBe('') + expect($restoredDraftNotice.get()).toEqual({ + fromKey: 'session-gone', + text: 'typed into a session that no longer exists' + }) + + // Fires once: a later trip through the fresh draft finds nothing to move + // and does not re-publish the notice the user already dismissed. + dismissRestoredDraftNotice() + act(() => { + rerender( undefined} sessionId="session-A" />) + }) + act(() => { + rerender( undefined} sessionId="" />) + }) + + expect($restoredDraftNotice.get()).toBeNull() + expect(takeSessionDraft(null).text).toBe('typed into a session that no longer exists') + clearSessionDraft(null) + }) + it('leaves the pre-session draft in its bucket when the user opens another session from a fresh chat', () => { stashSessionDraft(null, 'still composing a new chat', []) diff --git a/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.ts b/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.ts index 423572b1b6..3d6516c473 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.ts +++ b/apps/desktop/src/app/chat/composer/hooks/use-composer-draft.ts @@ -14,6 +14,7 @@ import { isElementInHiddenPane } from '@/components/pane-shell/pane-visibility' import { sanitizeComposerInput } from '@/lib/composer-input-sanitize' import { useStoreSelector } from '@/lib/use-session-slice' import { + adoptGoneSessionDraft, adoptNewSessionDraft, type ComposerAttachment, type ComposerDraftSyncMode, @@ -464,6 +465,14 @@ export function useComposerDraft({ // the route flips the scope, so it is not a usable signal here. if (!draftScopeRef.current && activeQueueSessionKey) { adoptNewSessionDraft(activeQueueSessionKey) + } else if (!activeQueueSessionKey) { + // The reverse handoff: a session the user was typing into turned out + // to be gone and the window dropped to a fresh draft (#111868). The + // outgoing composer's cleanup has already stashed the live text under + // the dead key — whether that was this instance's previous scope or an + // unmounted one's — so move it into the fresh draft when the gone + // verdict announced it. No announcement, no-op. + adoptGoneSessionDraft() } draftScopeRef.current = activeQueueSessionKey diff --git a/apps/desktop/src/app/chat/composer/index.tsx b/apps/desktop/src/app/chat/composer/index.tsx index 957aba3f69..dfb440c6f4 100644 --- a/apps/desktop/src/app/chat/composer/index.tsx +++ b/apps/desktop/src/app/chat/composer/index.tsx @@ -71,6 +71,7 @@ import { shouldConvertPasteToAttachment } from './large-paste' import { ActionBadges } from './micro-actions' import { chipTypedPathOnSpace, pathifyRefs } from './path-refs' import { QueuePanel } from './queue-panel' +import { RestoredDraftNotice } from './restored-draft-notice' import { beginComposerComposition, composerPlainText, @@ -1439,6 +1440,11 @@ export function ChatBar({ additions beside the "+" menu and before the controls. All four render nothing until something contributes. */} + {queueEdit && editingQueuedPrompt && ( diff --git a/apps/desktop/src/app/chat/composer/restored-draft-notice.tsx b/apps/desktop/src/app/chat/composer/restored-draft-notice.tsx new file mode 100644 index 0000000000..609a4d3cab --- /dev/null +++ b/apps/desktop/src/app/chat/composer/restored-draft-notice.tsx @@ -0,0 +1,63 @@ +import { useStore } from '@nanostores/react' + +import { Button } from '@/components/ui/button' +import { useI18n } from '@/i18n' +import { $restoredDraftNotice, dismissRestoredDraftNotice, undoRestoredDraft } from '@/store/composer' + +interface RestoredDraftNoticeProps { + /** The composer is showing the fresh draft (no session scope). */ + freshDraft: boolean + /** Clear the editor after Undo emptied the fresh draft. */ + onUndone: () => void + /** Latest live editor text — Undo only applies while it is still what was restored. */ + readLiveText: () => string +} + +/** + * "Restored your unsent message" strip above the fresh draft's input + * (#111868). Offers, never hijacks: no focus move, no navigation, no toast — + * the text is simply in the composer with a way to put it back. Renders + * nothing outside the fresh draft, so opening another session hides it + * without consuming the Undo. + */ +export function RestoredDraftNotice({ freshDraft, onUndone, readLiveText }: RestoredDraftNoticeProps) { + const notice = useStore($restoredDraftNotice) + const { t } = useI18n() + + if (!notice || !freshDraft) { + return null + } + + return ( +
+
{t.composer.restoredDraftNotice}
+
+ + +
+
+ ) +} diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts index d211801b9b..a59532638a 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts @@ -27,7 +27,7 @@ import { isMissingRpcMethod } from '@/lib/gateway-rpc' import { recoverInFlightTurnJournal } from '@/lib/inflight-turn-journal' import { setSessionYolo } from '@/lib/yolo-session' import { $clarifyRequests } from '@/store/clarify' -import { announceNewSessionDraftKey, migrateSessionDraft } from '@/store/composer' +import { announceGoneSessionDraft, announceNewSessionDraftKey, migrateSessionDraft } from '@/store/composer' import { clearQueuedPrompts, migrateQueuedPrompts } from '@/store/composer-queue' import { $connectionRequests } from '@/store/connection-request' import { @@ -2171,6 +2171,11 @@ export function useSessionActions({ return } + // The id is verifiably dead, but the text the user typed into it is + // still stashed under that key (#111868). Announce it so the + // composer's swap onto the fresh draft carries it over with an + // inline, undoable notice instead of leaving it stranded. + announceGoneSessionDraft(storedSessionId) startFreshSessionDraft(true) return diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index 84273e59bb..4ce519901c 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -2160,6 +2160,8 @@ export const ar = defineLocale({ attachments: count => `${count} مرفق`, editingInComposer: 'جار التحرير في صندوق الكتابة', editingQueuedInComposer: 'جار تحرير رسالة في الطابور', + restoredDraftNotice: 'تمت استعادة رسالتك غير المُرسلة', + restoredDraftUndo: 'تراجع', queueEdit: 'تحرير الرسالة المجدولة', queueSendNext: 'إرسالها تاليا', queueSteer: 'توجيه — تصحيح الدور الجاري فورا', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index f54b272c54..23445ac2fb 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -3038,6 +3038,8 @@ export const en: Translations = { attachments: count => `${count} attachment${count === 1 ? '' : 's'}`, editingInComposer: 'Editing in composer', editingQueuedInComposer: 'Editing queued turn in composer', + restoredDraftNotice: 'Restored your unsent message', + restoredDraftUndo: 'Undo', queueEdit: 'Edit', queueSendNext: 'Next', queueSteer: 'Steer — redirect the live turn now', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 2d6f25a159..26bb6ccf4b 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -2519,6 +2519,8 @@ export const ja = defineLocale({ attachments: count => `${count} 件の添付`, editingInComposer: 'コンポーザーで編集中', editingQueuedInComposer: 'コンポーザーでキュー済みターンを編集中', + restoredDraftNotice: '未送信のメッセージを復元しました', + restoredDraftUndo: '元に戻す', queueEdit: '編集', queueSendNext: '次に送信', queueSteer: 'ステア — 現在のターンを今すぐ修正', diff --git a/apps/desktop/src/i18n/ru.ts b/apps/desktop/src/i18n/ru.ts index 66104c48cb..2115298a84 100644 --- a/apps/desktop/src/i18n/ru.ts +++ b/apps/desktop/src/i18n/ru.ts @@ -2794,6 +2794,8 @@ export const ru = defineLocale({ attachments: count => `${count} ${RU_NOUN(count, 'вложение', 'вложения', 'вложений')}`, editingInComposer: 'Редактирование в композере', editingQueuedInComposer: 'Редактирование хода в очереди в композере', + restoredDraftNotice: 'Восстановлено ваше неотправленное сообщение', + restoredDraftUndo: 'Отменить', queueEdit: 'Изменить', queueSendNext: 'Дальше', queueSteer: 'Направить — изменить текущий ход сейчас', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index 30c2474084..a4e1ee1506 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -2617,6 +2617,8 @@ export interface Translations { attachments: (count: number) => string editingInComposer: string editingQueuedInComposer: string + restoredDraftNotice: string + restoredDraftUndo: string queueEdit: string queueSendNext: string queueSend: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 421362188a..f324224bd1 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -2502,6 +2502,8 @@ export const zhHant = defineLocale({ attachments: count => `${count} 個附件`, editingInComposer: '在輸入框中編輯', editingQueuedInComposer: '在輸入框中編輯排隊回合', + restoredDraftNotice: '已還原你未送出的訊息', + restoredDraftUndo: '復原', queueEdit: '編輯', queueSendNext: '下一個', queueSteer: '引導 — 立即修正目前回合', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index dd0818fa4d..b750eb2e82 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -3172,6 +3172,8 @@ export const zh = defineLocale({ attachments: count => `${count} 个附件`, editingInComposer: '正在输入框中编辑', editingQueuedInComposer: '正在输入框中编辑排队回合', + restoredDraftNotice: '已恢复你未发送的消息', + restoredDraftUndo: '撤销', queueEdit: '编辑', queueSendNext: '下一个', queueSteer: '引导 — 立即修正当前回合', diff --git a/apps/desktop/src/store/composer.test.ts b/apps/desktop/src/store/composer.test.ts index 0055e5ae33..a0f13584d6 100644 --- a/apps/desktop/src/store/composer.test.ts +++ b/apps/desktop/src/store/composer.test.ts @@ -2,8 +2,11 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import { $composerAttachments, + $restoredDraftNotice, $voiceConversationStartRequest, addComposerAttachment, + adoptGoneSessionDraft, + announceGoneSessionDraft, clearSessionDraft, type ComposerAttachment, createComposerAttachmentOccurrenceId, @@ -15,6 +18,7 @@ import { stashSessionDraft, takeSessionDraft, takeVoiceConversationStart, + undoRestoredDraft, updateComposerAttachment } from './composer' @@ -267,6 +271,38 @@ describe('session drafts', () => { expect(takeSessionDraft('session-a').attachments[0]?.label).toBe('doc.pdf') }) + it('restores a gone session draft only into an EMPTY fresh chat, and Undo puts it back where it was (#111868)', () => { + // Never clobber what the user is already typing in the new chat. + stashSessionDraft('session-a', 'from the dead session', []) + stashSessionDraft(null, 'already composing here', []) + announceGoneSessionDraft('session-a') + + expect(adoptGoneSessionDraft()).toBe(false) + expect($restoredDraftNotice.get()).toBeNull() + expect(takeSessionDraft(null).text).toBe('already composing here') + expect(takeSessionDraft('session-a').text).toBe('from the dead session') + + // Empty fresh chat → restored; Undo (text untouched) returns it to the + // dead key, so the same recovery path can find it again later. + clearSessionDraft(null) + announceGoneSessionDraft('session-a') + + expect(adoptGoneSessionDraft()).toBe(true) + expect(takeSessionDraft(null).text).toBe('from the dead session') + expect(undoRestoredDraft('from the dead session')).toBe(true) + expect(takeSessionDraft(null).text).toBe('') + expect(takeSessionDraft('session-a').text).toBe('from the dead session') + expect($restoredDraftNotice.get()).toBeNull() + + // Once the user has edited the restored text, Undo would destroy their + // work: it only dismisses. + announceGoneSessionDraft('session-a') + adoptGoneSessionDraft() + + expect(undoRestoredDraft('from the dead session, edited')).toBe(false) + expect(takeSessionDraft(null).text).toBe('from the dead session') + }) + it('migrates a tip-keyed draft onto the post-compression tip', () => { const tipBefore = '20260720_062637_ad96b3' const tipAfter = '20260720_071049_a28905' diff --git a/apps/desktop/src/store/composer.ts b/apps/desktop/src/store/composer.ts index bdaa5d5083..7a581b2272 100644 --- a/apps/desktop/src/store/composer.ts +++ b/apps/desktop/src/store/composer.ts @@ -187,6 +187,17 @@ export interface SessionDraft { const draftKey = (scope: string | null | undefined) => scope?.trim() || NEW_SESSION_DRAFT_KEY +/** Inline "Restored your unsent message" notice for the fresh draft (see + * `adoptGoneSessionDraft`). `null` = nothing to show. */ +export interface RestoredDraftNotice { + /** The dead stored-session key the text came from. */ + fromKey: string + /** The text as restored — Undo only applies while the draft still equals it. */ + text: string +} + +export const $restoredDraftNotice = atom(null) + const cloneDraft = (draft: SessionDraft): SessionDraft => ({ attachments: draft.attachments.map(attachment => ({ ...attachment })), text: draft.text @@ -389,6 +400,10 @@ export function stashSessionDraft(scope: string | null | undefined, text: string if (text.trim() || attachments.length > 0) { draftsBySession.set(key, cloneDraft({ attachments, text })) + } else if (key === NEW_SESSION_DRAFT_KEY) { + // The fresh draft was sent or emptied — a restore notice has nothing left + // to undo. + $restoredDraftNotice.set(null) } persistDraftTexts() @@ -463,6 +478,94 @@ export function adoptNewSessionDraft(toKey: string | null | undefined): boolean return !!announced && announced === toKey?.trim() && migrateSessionDraft(null, toKey) } +/** + * Recovery for the unsent text of a session that turned out to be GONE + * (#111868): deleted, or a stale id from a wiped / renamed backend. + * + * The resume path already drops such a window to a fresh draft without + * toasting or looping (62af32efe7c, bounded by `goneSessionVerdict`). The + * composer's draft stash is keyed per stored session, so the text the user + * typed into the dead id is not lost — but nothing will ever open that key + * again, so it is invisible. The gone verdict announces the dead key here; + * the composer's scope swap (concrete id → the `__new__` bucket) consumes + * it AFTER the outgoing cleanup stashed the live editor text, so even + * keystrokes still inside the persist debounce ride along. + * + * Offer, don't hijack: the fresh draft is seeded and an inline notice with + * Undo is published — no navigation, no focus steal, no toast. Fires once: + * the source key is cleared by the move, so re-opening the dead id later + * finds nothing to restore. + */ +let announcedGoneSessionDraftKey: string | null = null + +export function announceGoneSessionDraft(fromKey: string | null | undefined): void { + announcedGoneSessionDraftKey = fromKey?.trim() || null +} + +/** + * Consume the announcement when a composer enters the fresh-draft scope. + * Moves the dead key's draft into the `__new__` bucket and publishes the + * notice. Declines (no notice) when nothing was announced, the key holds no + * text, or the user is already composing a new chat — never clobber what + * they are typing. Keyed on the announcement, not on the composer observing + * an id → fresh transition: the composer can remount across the drop (a + * loading route mounts no composer), so the dead scope may never have been + * this instance's previous scope. + */ +export function adoptGoneSessionDraft(): boolean { + const announced = announcedGoneSessionDraftKey + announcedGoneSessionDraftKey = null + + if (!announced) { + return false + } + + const source = draftsBySession.get(draftKey(announced)) + + if (!source?.text.trim()) { + return false + } + + const dest = draftsBySession.get(NEW_SESSION_DRAFT_KEY) + + if (dest && (dest.text.trim() || dest.attachments.length > 0)) { + return false + } + + const { attachments, text } = source + stashSessionDraft(null, text, attachments) + clearSessionDraft(announced) + $restoredDraftNotice.set({ fromKey: announced, text }) + + return true +} + +export function dismissRestoredDraftNotice(): void { + $restoredDraftNotice.set(null) +} + +/** + * Undo the restore: put the text back under the dead key (where it was, + * still recoverable by the same path) and empty the fresh draft. Only while + * the live text is still exactly what was restored — once the user has + * edited it, Undo would destroy their work, so it only dismisses the notice. + * Returns whether the fresh draft was emptied (the caller repaints). + */ +export function undoRestoredDraft(liveText: string): boolean { + const notice = $restoredDraftNotice.get() + $restoredDraftNotice.set(null) + + if (!notice || liveText !== notice.text) { + return false + } + + const current = draftsBySession.get(NEW_SESSION_DRAFT_KEY) + stashSessionDraft(notice.fromKey, notice.text, current?.attachments ?? []) + clearSessionDraft(null) + + return true +} + export function setComposerDraft(value: string) { $composerDraft.set(value) } diff --git a/website/docs/user-guide/desktop.md b/website/docs/user-guide/desktop.md index 8beec726cf..65bfbaf62c 100644 --- a/website/docs/user-guide/desktop.md +++ b/website/docs/user-guide/desktop.md @@ -47,6 +47,7 @@ The center of the app. You get: - **The same conversation history** as every other Hermes surface — sessions started here resume in the CLI/TUI and vice versa. - **Drag-and-drop files** anywhere in the chat area to attach them to your next message. - **Independent background drafts** — hidden chat tabs can update their drafts without moving the caret or selection in the visible composer. +- **Unsent drafts survive a lost session** — a draft you typed into a chat whose session no longer exists (deleted elsewhere, or a stale tab after a profile rename or wiped backend) is carried into the fresh chat the app falls back to, with an inline **Restored your unsent message** strip above the input. **Undo** puts the text back where it was; nothing is sent, navigated, or focused on your behalf, and the strip appears once per lost draft. - **Directive chip actions** — hover an actionable reference (such as a URL) to reveal its action pill. A short grace period lets you move from the chip to the pill before it dismisses. The pill stays available while you move within it; after leaving, unrelated pointer movement does not delay dismissal. Clicking its action preserves the draft selection. - **A right-hand preview rail** — render web pages, files, and tool outputs side by side while you keep chatting. - **Hide vs. Close for stateful panes** — a zone holding a Browser (or HTML preview) or the Terminal pane offers **Hide** instead of Minimize. Hiding collapses the zone to its restore rail but keeps the pane's body mounted: a live page keeps its unsaved form input, timers, scroll position, and the agent's `drive_preview` automation target; a terminal keeps its running shell and scrollback; and **Restore** shows the same content without a reload. The hidden body is inert — it never takes keyboard shortcuts or steals composer focus. **Close** (the tab's ×) is what actually releases the page or shell. From 40778c74e27d7615e03503beec580444cf36ad52 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 19:56:41 -0700 Subject: [PATCH 58/65] fix: MoA aggregator merges adjacent same-role messages only for destinations that reject them The aggregator request deliberately ends `user(task), user(guidance)` on iteration 1 of every turn (#113175) so the whole prefix stays byte-stable for the provider prompt cache. Strict-alternation chat templates (llama.cpp / vLLM Jinja templates, Mistral, some OpenRouter routes) 400 on that adjacency ("Conversation roles must alternate ..."), and the turn failed. Merging proactively for everyone was declined because it brings back the byte divergence #113175/#113784 removed and only moves the 400. Reactive, destination-scoped recovery instead: - error_classifier: new `FailoverReason.role_alternation` for the vendor alternation wordings (checked before the request-validation table since the body also carries `invalid_request_error`); same abort+fallback hints as format_error so non-MoA consumers behave exactly as before. - moa_alternation (new sibling): `merge_same_role_messages` (reuses the loop's `_merge_user_content`), `destination_key` (base_url|provider, model), `is_role_alternation_rejection`. - moa_loop._call_prepared_aggregator: on that 400, retry ONCE with the adjacent user turns merged, remember the destination on the facade for the session so later iterations pre-merge, never touch destinations that accepted the split shape. The trace records the messages actually sent. - docs: caching section explains the reactive merge. Live loopback (real call_llm -> SDK -> HTTP stub that 400s on same-role adjacency): before, iteration 1 fails with BadRequestError after 1 request; after, 2 requests (split -> 400 -> merged -> 200), next turn pre-merged in 1 request; the accepting-stub control sends byte-identical requests. Fixes #112358 --- agent/chat_completion_helpers.py | 1 + agent/error_classifier.py | 22 ++++- agent/moa_alternation.py | 64 +++++++++++++ agent/moa_loop.py | 40 +++++++-- tests/agent/test_error_classifier.py | 2 +- tests/agent/test_moa_alternation_recovery.py | 90 +++++++++++++++++++ .../user-guide/features/mixture-of-agents.md | 2 + 7 files changed, 213 insertions(+), 8 deletions(-) create mode 100644 agent/moa_alternation.py create mode 100644 tests/agent/test_moa_alternation_recovery.py diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 89d3357c11..f1a9d4541a 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -1701,6 +1701,7 @@ _FALLBACK_REASON_LABELS = { FailoverReason.provider_policy_blocked: "provider policy blocked the request", FailoverReason.content_policy_blocked: "content policy blocked the request", FailoverReason.format_error: "request format rejected", + FailoverReason.role_alternation: "adjacent same-role messages rejected", FailoverReason.invalid_encrypted_content: "encrypted reasoning state rejected", FailoverReason.multimodal_tool_content_unsupported: "multimodal tool content unsupported", FailoverReason.thinking_signature: "thinking signature rejected", diff --git a/agent/error_classifier.py b/agent/error_classifier.py index 29bb23bee0..d6b0088d32 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -50,6 +50,7 @@ class FailoverReason(enum.Enum): provider_policy_blocked = "provider_policy_blocked" # Aggregator account data/privacy policy excluded the only endpoint content_policy_blocked = "content_policy_blocked" # Provider safety filter rejected this prompt — don't retry unchanged format_error = "format_error" # 400 bad request — abort or strip + retry + role_alternation = "role_alternation" # Strict chat template rejected adjacent same-role messages — merge them for this destination and retry invalid_encrypted_content = "invalid_encrypted_content" # Responses replay blob rejected — strip replay state and retry multimodal_tool_content_unsupported = "multimodal_tool_content_unsupported" # Provider rejected list-type content in tool messages (e.g. Xiaomi MiMo) — downgrade to text and retry reasoning_mandatory = "reasoning_mandatory" # Route rejects reasoning: {enabled: false} — send the disable no more this session and retry @@ -277,6 +278,19 @@ _INVALID_MESSAGE_BODY_PATTERNS = ( "messages: at least one message is required", _NO_USER_QUERY_SIGNAL, ) +# Strict-alternation chat templates (llama.cpp / vLLM Jinja templates, Mistral, some +# OpenRouter routes) 400 when two adjacent messages share a role. Deterministic for the +# request shape, and the only bad thing is the adjacency, so the caller that produced it +# (the MoA aggregator appends ``user(guidance)`` after ``user(task)`` on iteration 1 — +# #112358) merges the pair for THAT destination and retries once. Checked before the +# request-validation table: the body usually also carries ``invalid_request_error``. +_ROLE_ALTERNATION_PATTERNS = ( + "roles must alternate", "role must alternate", "must alternate between", + "consecutive user messages", "consecutive messages with the same role", + "consecutive messages of the same role", "same role in a row", "multiple user messages in a row", + "adjacent messages with the same role", +) + # Proxy-side rejection of the model's own tool-call JSON (Ollama "invalid tool call arguments", # OpenRouter-wrapped "function_call arguments"). Checked before the generic 400 validation and # overflow heuristics: on a large session the bare message would otherwise read as overflow. @@ -428,6 +442,9 @@ _V_OVERLOADED, _V_SERVER_ERROR, _V_TIMEOUT, _V_UNKNOWN = map(_v, (_R.overloaded, _V_IMAGE_TOO_LARGE, _V_IMAGE_CORRUPT = _v(_R.image_too_large), _v(_R.image_corrupt) _V_MULTIMODAL, _V_INVALID_ENCRYPTED = _v(_R.multimodal_tool_content_unsupported), _v(_R.invalid_encrypted_content) _V_REASONING_MANDATORY = _v(_R.reasoning_mandatory, should_compress=False, should_fallback=False) +# Same recovery hints as format_error: consumers without a merge-and-retry step (the main loop +# already merges adjacent users before the call) keep aborting to the fallback chain. +_V_ROLE_ALTERNATION = _v(_R.role_alternation, **_ABORT_FALLBACK) # The MODEL emitted unparseable tool-call JSON and the proxy (Ollama, OpenRouter) rejected it: no # other provider can fix that output, so falling back only replays the same broken turn 4-5 times # (20-60s per occurrence, #12770). Abort this call; the loop's argument repair handles the retry. @@ -524,7 +541,8 @@ _400_TAIL_RULES = _OVERFLOW_AS_5XX_RULES + ( # Status-less message path, head (before usage-limit disambiguation). _MESSAGE_HEAD_RULES = ((_MEMORY_CEILING_PATTERNS, _V_OVERLOADED), - (_PAYLOAD_TOO_LARGE_PATTERNS, _V_PAYLOAD_TOO_LARGE)) + _IMAGE_TOOL_RULES + (_PAYLOAD_TOO_LARGE_PATTERNS, _V_PAYLOAD_TOO_LARGE), + (_ROLE_ALTERNATION_PATTERNS, _V_ROLE_ALTERNATION)) + _IMAGE_TOOL_RULES # Status-less tail. Overload before rate_limit/billing so "overloaded" backs off # instead of rotating; policy block before model_not_found; timeout/connection @@ -966,6 +984,8 @@ def _classify_400(c: _Ctx) -> Verdict: return _V_SERVER_ERROR if any(p in msg for p in _MALFORMED_TOOL_ARGS_PATTERNS): return _V_MALFORMED_TOOL_ARGS + if any(p in msg for p in _ROLE_ALTERNATION_PATTERNS): + return _V_ROLE_ALTERNATION # Before overflow: GPT-5's "Unsupported parameter: 'max_tokens'" contains it. if any(p in msg for p in _400_VALIDATION_PATTERNS) or code in _400_VALIDATION_CODES: return _V_FORMAT_ERROR diff --git a/agent/moa_alternation.py b/agent/moa_alternation.py new file mode 100644 index 0000000000..4e94608b8c --- /dev/null +++ b/agent/moa_alternation.py @@ -0,0 +1,64 @@ +"""Reactive same-role merge for the MoA aggregator request (#112358, last atom). + +The aggregator request deliberately ends ``user(task), user(guidance)`` on iteration 1 of a +turn: the guidance is its own trailing message so every earlier message stays byte-stable and +the provider prefix cache keeps growing (``moa_loop._attach_reference_guidance``). Strict- +alternation chat templates (llama.cpp / vLLM Jinja templates, Mistral, some OpenRouter routes) +400 on that adjacency. Merging proactively for everyone would re-introduce the divergence the +split shape removed and only move the 400, so the merge is reactive and destination-scoped: + +* a 400 classified ``FailoverReason.role_alternation`` → retry ONCE with adjacent same-role + messages merged; +* the destination (``base_url``/provider + model) is remembered on the facade for the rest of + the session, so later iterations pre-merge for it and never pay the 400 again; +* destinations that accepted the split shape are never touched — their prefix stays byte-stable. +""" + +from __future__ import annotations + +import logging +from typing import Any + +logger = logging.getLogger(__name__) + + +def destination_key(runtime: dict[str, Any]) -> tuple[str, str]: + """``(route, model)`` identity of an aggregator destination: the base_url when the slot + resolved one (two providers can share a model id), else the provider slug.""" + route = str(runtime.get("base_url") or runtime.get("provider") or "").strip().rstrip("/") + return route, str(runtime.get("model") or "").strip() + + +def merge_same_role_messages(messages: list[dict[str, Any]]) -> list[dict[str, Any]]: + """Return ``messages`` with adjacent user turns folded into one (the only same-role adjacency + the aggregator request produces); other rows are shared, never mutated. Returns the input + object itself when nothing merged so callers can detect a no-op.""" + from agent.agent_runtime_helpers import _UNMERGEABLE, _merge_user_content + + merged: list[dict[str, Any]] = [] + changed = False + for message in messages: + prev = merged[-1] if merged else None + content: Any = _UNMERGEABLE + if prev is not None and prev.get("role") == "user" and message.get("role") == "user": + content = _merge_user_content(prev.get("content", ""), message.get("content", "")) + if content is _UNMERGEABLE: + merged.append(message) + continue + merged[-1] = {**prev, "content": content} + changed = True + return merged if changed else messages + + +def is_role_alternation_rejection(exc: Exception, runtime: dict[str, Any]) -> bool: + """True when the aggregator destination rejected the request for adjacent same-role messages.""" + from agent.error_classifier import FailoverReason, classify_api_error + + try: + classified = classify_api_error( + exc, provider=str(runtime.get("provider") or ""), model=str(runtime.get("model") or ""), + base_url=str(runtime.get("base_url") or ""), + ) + except Exception: # pragma: no cover - classification must never mask the original error + return False + return classified.reason is FailoverReason.role_alternation diff --git a/agent/moa_loop.py b/agent/moa_loop.py index 21b747fac8..5ce755d512 100644 --- a/agent/moa_loop.py +++ b/agent/moa_loop.py @@ -22,6 +22,7 @@ from typing import Any from agent.auxiliary_client import call_llm from agent.message_content import flatten_message_text +from agent.moa_alternation import destination_key, is_role_alternation_rejection, merge_same_role_messages from agent.transports import get_transport from agent.usage_pricing import CanonicalUsage @@ -957,6 +958,10 @@ class MoAChatCompletions: self._fanout_turn_sig: str | None = None self._fanout_last_state_sig: str | None = None self._privacy_mode: str = "" # normalized moa.privacy_filter, refreshed per create() + # Destinations (route, model) that 400'd on adjacent same-role messages this session: + # their aggregator requests are pre-merged; every other destination keeps the split, + # cache-stable shape (agent/moa_alternation.py). + self._merge_same_role_destinations: set[tuple[str, str]] = set() def consume_reference_usage(self) -> tuple[Any, Any]: """Pop pending fan-out ``(CanonicalUsage, cost_usd_or_None)`` and reset both @@ -1081,10 +1086,6 @@ class MoAChatCompletions: ) trace = self._pending_trace if trace is not None: - # Trace the exact aggregator INPUT (persisted copy redacted; live input raw). - trace["aggregator_input_messages"] = ( - _redact_trace_messages([dict(m) for m in agg_messages]) if getattr(self, "_privacy_mode", "") else agg_messages - ) trace["aggregator_label"] = _slot_label(aggregator) # stream=True returns the RAW token stream (consumer reassembles + retries); # the non-streaming path forwards no stream/stream_options/timeout. The @@ -1097,13 +1098,40 @@ class MoAChatCompletions: stream_kwargs["timeout"] = api_kwargs["timeout"] # Pop the runtime's extra_body so the explicit kwarg never collides with **agg_runtime. agg_extra_body = _merge_slot_extra_body(agg_runtime.pop("extra_body", None), api_kwargs.get("extra_body")) - agg_response = call_llm( - task="moa_aggregator", messages=agg_messages, temperature=prepared["aggregator_temperature"], + destination = destination_key(agg_runtime) + # Facades built via __new__ (tests, swaps) have no __init__ state. + remembered = getattr(self, "_merge_same_role_destinations", None) + if remembered is None: + remembered = self._merge_same_role_destinations = set() + merged = destination in remembered + if merged: + agg_messages = merge_same_role_messages(agg_messages) + send = functools.partial( + call_llm, task="moa_aggregator", temperature=prepared["aggregator_temperature"], max_tokens=api_kwargs.get("max_tokens"), tools=tools, extra_body=agg_extra_body, reasoning_config=_aggregator_reasoning_config(aggregator), # same policy as direct create() **stream_kwargs, **agg_runtime, ) + try: + agg_response = send(messages=agg_messages) + except Exception as exc: + # Strict-alternation template rejected ``user(task), user(guidance)``: merge the pair for + # THIS destination only and retry once; remember it so later iterations pre-merge. + retry_messages = None if merged else merge_same_role_messages(agg_messages) + if retry_messages is None or retry_messages is agg_messages or not is_role_alternation_rejection(exc, agg_runtime): + raise + remembered.add(destination) + logger.warning( + "MoA aggregator %s rejected adjacent same-role messages — merging them for this " + "destination for the rest of the session and retrying once: %.200s", _slot_label(aggregator), exc, + ) + agg_messages = retry_messages + agg_response = send(messages=agg_messages) if trace is not None: + # Trace the exact aggregator INPUT as sent (persisted copy redacted; live input raw). + trace["aggregator_input_messages"] = ( + _redact_trace_messages([dict(m) for m in agg_messages]) if getattr(self, "_privacy_mode", "") else agg_messages + ) # Streaming output lands as the turn's assistant message; the trace marks it. trace["aggregator_streamed"] = stream output = None diff --git a/tests/agent/test_error_classifier.py b/tests/agent/test_error_classifier.py index c71cc07f3b..8e8dab5aa1 100644 --- a/tests/agent/test_error_classifier.py +++ b/tests/agent/test_error_classifier.py @@ -64,7 +64,7 @@ class TestFailoverReason: "ssl_cert_verification", "context_overflow", "payload_too_large", "image_too_large", "image_corrupt", - "model_not_found", "format_error", + "model_not_found", "format_error", "role_alternation", "invalid_encrypted_content", "multimodal_tool_content_unsupported", "reasoning_mandatory", diff --git a/tests/agent/test_moa_alternation_recovery.py b/tests/agent/test_moa_alternation_recovery.py new file mode 100644 index 0000000000..e8d0c05a81 --- /dev/null +++ b/tests/agent/test_moa_alternation_recovery.py @@ -0,0 +1,90 @@ +"""MoA aggregator: strict-alternation destinations get their adjacent same-role messages merged +reactively and per destination (#112358, last atom). + +The guidance is deliberately a separate trailing ``user`` message so the prefix stays cache-stable; +a chat template that 400s on ``user(task), user(guidance)`` must get ONE merged retry, be remembered +for the session, and never change the bytes sent to destinations that accepted the split shape. +""" + +from types import SimpleNamespace + +import pytest + +from agent import moa_loop +from agent.error_classifier import FailoverReason, classify_api_error + +ALTERNATION_MSG = "Conversation roles must alternate user/assistant/user/assistant/..." + + +class _AlternationRejected(Exception): + status_code = 400 + + def __init__(self): + super().__init__(f"Error code: 400 - {{'error': {{'message': '{ALTERNATION_MSG}', 'type': 'invalid_request_error'}}}}") + self.response = SimpleNamespace(status_code=400, headers={}) + self.body = {"message": ALTERNATION_MSG, "type": "invalid_request_error"} + + +def _strict_destination(calls, strict_models): + """``call_llm`` double: 400s like a strict chat template when adjacent non-system messages share a role.""" + def call_llm(**kw): + calls.append(kw) + msgs = [m for m in kw["messages"] if m["role"] != "system"] + if kw["model"] in strict_models and any(a["role"] == b["role"] for a, b in zip(msgs, msgs[1:])): + raise _AlternationRejected() + return SimpleNamespace(choices=[]) + return call_llm + + +@pytest.fixture +def facade(monkeypatch): + monkeypatch.setattr( + moa_loop, "_slot_runtime", + lambda slot: {"provider": "custom", "model": slot["model"], "base_url": "http://strict.local/v1", + "api_mode": "chat_completions"}, + ) + f = moa_loop.MoAChatCompletions("default", agent=None) + f._pending_trace = None + return f + + +def _send(facade, model, messages, guidance="[Mixture of Agents reference context]\nadvice"): + prepared = facade.rebase_prepared_request( + {"guidance": guidance, "aggregator": {"provider": "custom", "model": model}, "aggregator_temperature": None}, + messages, + ) + return facade._call_prepared_aggregator(prepared, {"tools": None}) + + +def test_alternation_400_is_classified_and_retried_once_merged_then_remembered(monkeypatch, facade): + classified = classify_api_error(_AlternationRejected(), provider="custom", model="strict-model") + assert classified.reason is FailoverReason.role_alternation + + calls = [] + monkeypatch.setattr(moa_loop, "call_llm", _strict_destination(calls, {"strict-model"})) + task = [{"role": "system", "content": "sys"}, {"role": "user", "content": "task"}] + + _send(facade, "strict-model", task) # iteration 1 of turn 1: split shape rejected → one merged retry + assert [[m["role"] for m in c["messages"]] for c in calls] == [["system", "user", "user"], ["system", "user"]] + assert calls[-1]["messages"][-1]["content"] == "task\n\n[Mixture of Agents reference context]\nadvice" + + del calls[:] + _send(facade, "strict-model", [*task, {"role": "assistant", "content": "answer"}, {"role": "user", "content": "task2"}]) + # Remembered destination: iteration 1 of the next turn is pre-merged, no 400 paid again. + assert [[m["role"] for m in c["messages"]] for c in calls] == [["system", "user", "assistant", "user"]] + assert calls[0]["messages"][-1]["content"].startswith("task2\n\n[Mixture of Agents reference context]") + + +def test_destinations_that_accept_the_split_shape_keep_byte_identical_requests(monkeypatch, facade): + calls = [] + monkeypatch.setattr(moa_loop, "call_llm", _strict_destination(calls, {"strict-model"})) + task = [{"role": "system", "content": "sys"}, {"role": "user", "content": "task"}] + guidance = "[Mixture of Agents reference context]\nadvice" + + _send(facade, "strict-model", task, guidance) # teaches the facade about the strict destination + del calls[:] + _send(facade, "lenient-model", task, guidance) + + # The lenient destination on the SAME facade still gets the split, cache-stable shape in one request. + assert len(calls) == 1 + assert calls[0]["messages"] == [*task, {"role": "user", "content": guidance}] diff --git a/website/docs/user-guide/features/mixture-of-agents.md b/website/docs/user-guide/features/mixture-of-agents.md index 72ae9e465b..18abd47e9e 100644 --- a/website/docs/user-guide/features/mixture-of-agents.md +++ b/website/docs/user-guide/features/mixture-of-agents.md @@ -245,6 +245,8 @@ Both internal call types cache normally: - **Reference models** receive a trimmed, deterministic view of the conversation (system prompt and tool transcript stripped — see the loop above). Because that view is a stable function of the stable history, a reference model's prompt prefix repeats across iterations and caches normally. References are short advisory calls with no tools. - **The aggregator** is the acting model. The reference outputs are appended as their *own* trailing user message of private guidance — never merged into your message. Because that block sits at the tail — below the entire stable prefix (system prompt + your message + prior tool history) — it does not invalidate any cached prefix: every request in a tool loop is a byte-identical extension of the previous one minus its guidance block, so the aggregator gets a cache hit on everything above the injection and only the freshly appended tail is new. That is exactly how every normal turn behaves, where each new user message is also uncached tail tokens. Aggregators on the Anthropic Messages, Bedrock Converse or native Gemini wire merge the two adjacent user turns into one message, but as separate content blocks: your message's block is byte-identical to the one later iterations replay, and the guidance block follows it, so the cached prefix still runs through your message. + On the OpenAI-compatible wire the request ends `user(your message), user(guidance)` on the first iteration of a turn. A few chat templates that enforce strict user/assistant alternation (llama.cpp and vLLM Jinja templates, some OpenRouter routes) reject that with a 400 such as `Conversation roles must alternate`. Hermes recovers on its own: it retries that one request with the two adjacent user messages merged, remembers that aggregator destination (endpoint + model) for the rest of the session so later turns are merged up front, and leaves every other destination on the split, cache-stable shape. The merge is applied only where a destination demanded it, because merging everywhere would reintroduce the prefix divergence described above. + So MoA does not sacrifice prompt caching on either call type. Its only real cost is the extra reference calls (once per user turn with the default `fanout`) — you pay for multiple model perspectives, not for broken caches. The long-lived conversation prefix shared with the rest of Hermes is fully intact. ## Notes From 76724f3dc2557fcc288c8640cc7ae710093e3e66 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 20:34:01 -0700 Subject: [PATCH 59/65] fix(agent): terminal copy for the new role_alternation failure reason Reclassifying alternation 400s away from format_error left non-MoA callers on the generic default text. --- agent/turn_failure_copy.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/agent/turn_failure_copy.py b/agent/turn_failure_copy.py index e8337db5b6..fec973dd59 100644 --- a/agent/turn_failure_copy.py +++ b/agent/turn_failure_copy.py @@ -154,6 +154,10 @@ _NONRETRYABLE_COPY: Dict[str, str] = { "{label} rejected this request as malformed, so the model didn't answer. Start a clean " "session with /new or switch models with /model; if it keeps happening, run `hermes doctor`." ), + FailoverReason.role_alternation.value: ( + "{label} requires user and assistant turns to strictly alternate and rejected this " + "conversation's shape. Start a clean session with /new or switch models with /model." + ), FailoverReason.ssl_cert_verification.value: ( "Hermes couldn't verify {label}'s security certificate, so the connection was refused. " "This is usually a corporate proxy or an outdated certificate store on this computer — " From 46503f16727dba3be9f5bc2d62090c33b2104507 Mon Sep 17 00:00:00 2001 From: Marc Caelier Date: Fri, 4 Sep 2026 14:08:24 +0800 Subject: [PATCH 60/65] fix(auth): isolate Codex singleton sync by principal --- agent/credential_pool.py | 36 ++++++++++++++++++++++++++++++++++++ 1 file changed, 36 insertions(+) diff --git a/agent/credential_pool.py b/agent/credential_pool.py index 306eb1323e..ab4706e800 100644 --- a/agent/credential_pool.py +++ b/agent/credential_pool.py @@ -306,6 +306,40 @@ def label_from_token(token: str, fallback: str) -> str: return fallback +def _codex_principal_identity(access_token: Any) -> Optional[Tuple[str, str]]: + """``(chatgpt_account_id, sub)`` of a Codex access token, or None when either claim is missing. + + Decoded without signature verification: this only decides whether two credentials Hermes + already holds belong to the same principal, never whether a token is valid. Both claims are + required because members of one ChatGPT workspace share ``chatgpt_account_id`` yet have their + own subjects and quotas. + """ + claims = _decode_jwt_claims(access_token) + auth_claims = claims.get("https://api.openai.com/auth") if isinstance(claims, dict) else None + account_id = auth_claims.get("chatgpt_account_id") if isinstance(auth_claims, dict) else None + subject = claims.get("sub") if isinstance(claims, dict) else None + if not (isinstance(account_id, str) and account_id.strip() and isinstance(subject, str) and subject.strip()): + return None + return account_id.strip(), subject.strip() + + +def _codex_entry_tracks_singleton(entry: PooledCredential, singleton_tokens: Dict[str, Any]) -> bool: + """Whether a Codex pool entry may adopt the auth.json singleton's token pair. + + ``device_code`` IS the singleton. ``manual:device_code`` is ambiguous: a legacy alias of the + singleton (same account, must follow its rotations) or an independent account added with + ``hermes auth add openai-codex`` (must never be overwritten — adopting turned two logins into + one account, both hitting the same usage limit). Same principal proves the alias; unknown + identity fails closed. + """ + if entry.source == "device_code": + return True + if entry.source != SOURCE_MANUAL_DEVICE_CODE: + return False + entry_identity = _codex_principal_identity(entry.access_token) + return entry_identity is not None and entry_identity == _codex_principal_identity(singleton_tokens.get("access_token")) + + def _next_priority(entries: List[PooledCredential]) -> int: return max((entry.priority for entry in entries), default=-1) + 1 @@ -1043,6 +1077,8 @@ class CredentialPool(CredentialPoolAdminMixin, CredentialPoolModelCooldownMixin) tokens = state.get("tokens") if isinstance(state, dict) else None if not isinstance(tokens, dict): return entry + if is_codex and not _codex_entry_tracks_singleton(entry, tokens): + return entry store_access = tokens.get("access_token", "") store_refresh = tokens.get("refresh_token", "") entry_refresh = entry.refresh_token or "" From 20512a47a6d7bf9be7d3bc1e8c3c03436a7693dc Mon Sep 17 00:00:00 2001 From: Tranquil-Flow <66773372+Tranquil-Flow@users.noreply.github.com> Date: Wed, 9 Sep 2026 20:21:56 +0200 Subject: [PATCH 61/65] fix(agent): refuse stale singleton adoption over rotated manual:device_code entries (#106705) A manual:device_code Codex pool entry never writes its rotation back to the auth.json singleton (independent-credential contract, #39236). The singleton sync adopted differing singleton tokens with no staleness proof, so after a pool-side rotation the stale singleton was re-adopted over the pool's fresh chain and the already-consumed refresh token was POSTed again (refresh_token_reused). Gate adoption on the singleton's last_refresh not predating the entry's own rotation; missing stamps on either side keep the historical adopt-on-difference behavior (#70111). --- agent/credential_pool.py | 34 ++ .../test_credential_pool_singleton_replay.py | 313 ++++++++++++++++++ 2 files changed, 347 insertions(+) create mode 100644 tests/agent/test_credential_pool_singleton_replay.py diff --git a/agent/credential_pool.py b/agent/credential_pool.py index ab4706e800..ede5f7d0e1 100644 --- a/agent/credential_pool.py +++ b/agent/credential_pool.py @@ -404,6 +404,23 @@ def _parse_absolute_timestamp(value: Any) -> Optional[float]: return None +def _singleton_predates_entry(state: Any, entry: "PooledCredential") -> bool: + """True only when the auth.json singleton is PROVABLY older than *entry*. + + Both sides stamp ``last_refresh`` on every successful rotation. When + either side lacks a parseable stamp this returns False (cannot prove), + which keeps the historical adopt-on-difference behavior (#70111) intact + for legacy writers. + """ + entry_ts = _parse_absolute_timestamp(entry.last_refresh) + if entry_ts is None: + return False + state_ts = _parse_absolute_timestamp(state.get("last_refresh") if isinstance(state, dict) else None) + if state_ts is None: + return False + return state_ts < entry_ts + + def _normalize_error_context(error_context: Optional[Dict[str, Any]]) -> Dict[str, Any]: if not isinstance(error_context, dict): return {} @@ -1099,6 +1116,23 @@ class CredentialPool(CredentialPoolAdminMixin, CredentialPoolModelCooldownMixin) entry.id, ) should_adopt = True + if should_adopt and _singleton_predates_entry(state, entry): + # #106705: manual:* entries never write back to the singleton + # (#39236), so after a pool-side rotation the singleton sits + # one chain behind. Adopting it would replay the consumed + # refresh token. ``last_refresh`` is stamped on every + # successful rotation on both sides; when either side lacks a + # parseable stamp this falls through to the historical + # adopt-on-difference above (#70111). + logger.info( + "Pool entry %s: auth.json singleton predates this entry's " + "rotation (last_refresh %s < %s); keeping pool chain to " + "avoid replaying the consumed refresh token", + entry.id, + state.get("last_refresh") if isinstance(state, dict) else None, + entry.last_refresh, + ) + should_adopt = False if should_adopt: logger.debug( "Pool entry %s: syncing %s tokens from auth.json (refreshed by another process)", diff --git a/tests/agent/test_credential_pool_singleton_replay.py b/tests/agent/test_credential_pool_singleton_replay.py new file mode 100644 index 0000000000..e69999cbc0 --- /dev/null +++ b/tests/agent/test_credential_pool_singleton_replay.py @@ -0,0 +1,313 @@ +"""Regression tests for #106705: Codex manual:device_code pool entries must +not replay an already-consumed refresh token by re-adopting a stale singleton. + +Causal chain (agent/credential_pool.py): + +1. ``_sync_entry_from_auth_store`` adopts differing singleton tokens for BOTH + ``device_code`` and ``manual:device_code`` Codex entries with no staleness + proof. +2. A successful pool-side rotation persists the fresh chain into the + credential-pool store, but ``_sync_device_code_entry_to_auth_store`` + deliberately skips singleton write-back for ``manual:*`` sources + (independent-credential contract, #39236), so the singleton stays one + rotation behind. +3. The next refresh syncs that STALE singleton over the pool's fresh entry and + POSTs the consumed refresh token again -> ``refresh_token_reused``. + +The fix gates adoption on the singleton being provably newer (``last_refresh`` +comparison); these tests run the real production path (``load_pool`` -> +``_refresh_entry``) with only the HTTP transport boundary mocked. +""" + +from __future__ import annotations + +import base64 +import json +from datetime import datetime, timezone + +import pytest + +from agent.credential_pool import load_pool + +# Synthetic JWT-ish access tokens carrying a far-future expiry claim so +# token-expiry probes see a valid token and the refresh path is driven by +# ``force`` alone. +_FAR_FUTURE_EXP = 4102444800 # 2100-01-01 + + +def _jwt(exp: int = _FAR_FUTURE_EXP) -> str: + def _part(payload: dict) -> str: + raw = json.dumps(payload, separators=(",", ":")).encode("utf-8") + return base64.urlsafe_b64encode(raw).decode("ascii").rstrip("=") + + return f"{_part({'alg': 'none', 'typ': 'JWT'})}.{_part({'exp': exp})}.sig" + + +def _iso(seconds: float) -> str: + return datetime.fromtimestamp(seconds, tz=timezone.utc).isoformat().replace("+00:00", "Z") + + +# Rotation chains: old -> new1 -> new2. Timestamps strictly increase so the +# ``last_refresh`` ordering is provable. +_T_OLD = 1_800_000_000.0 +_T_NEW1 = _T_OLD + 600.0 + +_AT_OLD, _RT_OLD = _jwt(), "rt-old" +_AT_NEW1, _RT_NEW1 = _jwt(), "rt-new1" +_AT_NEW2, _RT_NEW2 = _jwt(), "rt-new2" + + +def _manual_entry_payload(id: str = "manual-1", access_token: str = _AT_OLD, + refresh_token: str = _RT_OLD, last_refresh=None) -> dict: + return { + "id": id, + "label": "manual codex grant", + "auth_type": "oauth", + "priority": 0, + "source": "manual:device_code", + "access_token": access_token, + "refresh_token": refresh_token, + "last_refresh": last_refresh, + } + + +def _store(provider_state: dict, pool_entries: list) -> dict: + return { + "version": 1, + "providers": {"openai-codex": provider_state}, + "credential_pool": {"openai-codex": pool_entries}, + } + + +def _tokens_state(access_token: str, refresh_token: str, last_refresh) -> dict: + return { + "tokens": {"access_token": access_token, "refresh_token": refresh_token}, + "last_refresh": last_refresh, + } + + +def _write_store(tmp_path, monkeypatch, store: dict) -> None: + home = tmp_path / "hermes" + home.mkdir(parents=True, exist_ok=True) + (home / "auth.json").write_text(json.dumps(store, indent=2)) + monkeypatch.setenv("HERMES_HOME", str(home)) + + +def _install_fake_refresh(monkeypatch, chains: dict, stamp=_T_NEW1 + 600.0): + """Mock only the HTTP transport; record every refresh_token POSTed.""" + posted = [] + + def fake_refresh(access_token, refresh_token, **kwargs): + posted.append(refresh_token) + if refresh_token not in chains: + raise AssertionError(f"unexpected refresh POST for {refresh_token!r}") + at, rt = chains[refresh_token] + return {"access_token": at, "refresh_token": rt, "last_refresh": _iso(stamp)} + + monkeypatch.setattr( + "agent.credential_pool.auth_mod.refresh_codex_oauth_pure", fake_refresh + ) + return posted + + +class TestPoolRotationDoesNotReplayConsumedRefreshToken: + """L1 + L6: after a pool-side rotation, the stale singleton must not win.""" + + def test_second_forced_refresh_uses_rotated_chain_not_stale_singleton(self, tmp_path, monkeypatch): + """L1 — the reporter's exact scenario. + + Entry holds the chain rotated by the pool (new1, stamped new1-time); + the singleton still holds the consumed old chain (stamped old-time — + write-back deliberately skipped for manual sources). A second forced + refresh must POST the entry's own new1 refresh token — NOT re-adopt + the stale singleton and replay old. + """ + _write_store(tmp_path, monkeypatch, _store( + _tokens_state(_AT_OLD, _RT_OLD, _iso(_T_OLD)), + [_manual_entry_payload( + access_token=_AT_NEW1, refresh_token=_RT_NEW1, last_refresh=_iso(_T_NEW1), + )], + )) + posted = _install_fake_refresh(monkeypatch, {_RT_NEW1: (_AT_NEW2, _RT_NEW2)}) + + pool = load_pool("openai-codex") + updated = pool._refresh_entry(pool._entries[0], force=True) + + assert updated is not None + assert updated.refresh_token == _RT_NEW2 + # The consumed old refresh token must never be re-POSTed. + assert posted == [_RT_NEW1] + + def test_recover_path_does_not_adopt_stale_singleton_after_failed_post(self, tmp_path, monkeypatch): + """L6 — failed-POST recovery must not regress the fresh chain. + + After the pool rotated to new1 (write-back skipped), a refresh POST + that fails must NOT make ``_recover_failed_refresh`` adopt the stale + old singleton as "newer tokens" — that regression is exactly the + replay loop. + """ + _write_store(tmp_path, monkeypatch, _store( + _tokens_state(_AT_OLD, _RT_OLD, _iso(_T_OLD)), + [_manual_entry_payload( + access_token=_AT_NEW1, refresh_token=_RT_NEW1, last_refresh=_iso(_T_NEW1), + )], + )) + + def failing_refresh(access_token, refresh_token, **kwargs): + raise RuntimeError("synthetic network failure") + + monkeypatch.setattr( + "agent.credential_pool.auth_mod.refresh_codex_oauth_pure", failing_refresh + ) + + pool = load_pool("openai-codex") + pool._refresh_entry(pool._entries[0], force=True) + + # The entry must not be regressed onto the stale singleton chain. + entry_now = next(e for e in pool._entries if e.id == "manual-1") + assert entry_now.refresh_token == _RT_NEW1 + assert entry_now.access_token == _AT_NEW1 + + +class TestFreshSingletonAdoptionStillWorks: + """L2 — #70111 regression guard: a provably NEWER singleton must win.""" + + def test_newer_singleton_is_adopted_over_stale_pool_entry(self, tmp_path, monkeypatch): + """The fleet-outage fix (commit 7380b48589) must keep working. + + The singleton was rotated by another process (e.g. ``hermes model`` + re-auth); the pool entry still holds the old consumed chain. The sync + must adopt the newer singleton and refresh with IT. + """ + _write_store(tmp_path, monkeypatch, _store( + _tokens_state(_AT_NEW1, _RT_NEW1, _iso(_T_NEW1)), + [_manual_entry_payload( + access_token=_AT_OLD, refresh_token=_RT_OLD, last_refresh=_iso(_T_OLD), + )], + )) + posted = _install_fake_refresh(monkeypatch, {_RT_NEW1: (_AT_NEW2, _RT_NEW2)}) + + pool = load_pool("openai-codex") + updated = pool._refresh_entry(pool._entries[0], force=True) + + assert updated is not None + assert updated.refresh_token == _RT_NEW2 + assert posted == [_RT_NEW1] + + def test_refresh_token_only_singleton_is_adopted(self, tmp_path, monkeypatch): + """L2b — refresh_token-only singleton (the consumed-access branch). + + Another process rotated and the access_token was consumed; only the + new refresh_token remains on disk. Adoption must still fire (newer + ``last_refresh``) so the consumed token is not replayed. + """ + _write_store(tmp_path, monkeypatch, _store( + _tokens_state("", _RT_NEW1, _iso(_T_NEW1)), + [_manual_entry_payload( + access_token=_AT_OLD, refresh_token=_RT_OLD, last_refresh=_iso(_T_OLD), + )], + )) + posted = _install_fake_refresh(monkeypatch, {_RT_NEW1: (_AT_NEW2, _RT_NEW2)}) + + pool = load_pool("openai-codex") + updated = pool._refresh_entry(pool._entries[0], force=True) + + assert updated is not None + assert updated.refresh_token == _RT_NEW2 + assert posted == [_RT_NEW1] + + +class TestTimestampEdgeCases: + """L3 — missing timestamps fail open to the historical adopt-on-difference.""" + + def test_missing_entry_timestamp_still_adopts_differing_singleton(self, tmp_path, monkeypatch): + """Entry carries no ``last_refresh`` (pre-stamping pool writer). + + Old behavior (adopt on difference) must be preserved so a fresh + re-auth is never stranded when the entry side has no timestamp. + """ + _write_store(tmp_path, monkeypatch, _store( + _tokens_state(_AT_NEW1, _RT_NEW1, _iso(_T_NEW1)), + [_manual_entry_payload( + access_token=_AT_OLD, refresh_token=_RT_OLD, last_refresh=None, + )], + )) + posted = _install_fake_refresh(monkeypatch, {_RT_NEW1: (_AT_NEW2, _RT_NEW2)}) + + pool = load_pool("openai-codex") + updated = pool._refresh_entry(pool._entries[0], force=True) + + assert updated is not None + assert updated.refresh_token == _RT_NEW2 + assert posted == [_RT_NEW1] + + def test_missing_singleton_timestamp_still_adopts_differing_singleton(self, tmp_path, monkeypatch): + """Singleton carries no ``last_refresh`` (legacy auth.json writer). + + Cannot prove the singleton older -> fall back to adopt-on-difference + so a fresh re-auth from a legacy writer is not stranded either. + """ + _write_store(tmp_path, monkeypatch, _store( + _tokens_state(_AT_NEW1, _RT_NEW1, None), + [_manual_entry_payload( + access_token=_AT_OLD, refresh_token=_RT_OLD, last_refresh=_iso(_T_OLD), + )], + )) + posted = _install_fake_refresh(monkeypatch, {_RT_NEW1: (_AT_NEW2, _RT_NEW2)}) + + pool = load_pool("openai-codex") + updated = pool._refresh_entry(pool._entries[0], force=True) + + assert updated is not None + assert updated.refresh_token == _RT_NEW2 + assert posted == [_RT_NEW1] + + +class TestDeviceCodeSeededEntries: + """L4 — the singleton-seeded source is unaffected by the guard.""" + + def test_device_code_entry_rotates_and_writeback_converges(self, tmp_path, monkeypatch): + _write_store(tmp_path, monkeypatch, _store( + _tokens_state(_AT_OLD, _RT_OLD, _iso(_T_OLD)), + [], + )) + posted = _install_fake_refresh(monkeypatch, {_RT_OLD: (_AT_NEW1, _RT_NEW1)}) + + pool = load_pool("openai-codex") + seeded = [e for e in pool._entries if e.source == "device_code"] + assert seeded, "device_code entry must be seeded from the singleton" + + updated = pool._refresh_entry(seeded[0], force=True) + assert updated is not None + assert updated.refresh_token == _RT_NEW1 + assert posted == [_RT_OLD] + + # Write-back must converge the singleton onto the fresh chain. + on_disk = json.loads((tmp_path / "hermes" / "auth.json").read_text()) + synced = on_disk["providers"]["openai-codex"]["tokens"] + assert synced["refresh_token"] == _RT_NEW1 + + +class TestIndependentAccountNotClobbered: + """L5 — #39236 guard: independent manual grants keep their own chain.""" + + def test_independent_manual_entry_with_newer_own_timestamp_is_not_overwritten(self, tmp_path, monkeypatch): + """An independent account's entry whose own chain is NEWER (rotated + by this pool) must never be overwritten by the older singleton — + same mechanism as L1, asserted as the #39236 no-clobber contract. + """ + _write_store(tmp_path, monkeypatch, _store( + _tokens_state(_AT_OLD, _RT_OLD, _iso(_T_OLD)), + [_manual_entry_payload( + id="indep-1", + access_token=_AT_NEW1, refresh_token=_RT_NEW1, last_refresh=_iso(_T_NEW1), + )], + )) + posted = _install_fake_refresh(monkeypatch, {_RT_NEW1: (_AT_NEW2, _RT_NEW2)}) + + pool = load_pool("openai-codex") + updated = pool._refresh_entry(pool._entries[0], force=True) + + assert updated is not None + assert updated.refresh_token == _RT_NEW2 + assert _RT_OLD not in posted From c1ad9daa8f6cc402feef2eb6b53bbcbdaad2915f Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 20:38:40 -0700 Subject: [PATCH 62/65] test: two invariant tests for Codex singleton adoption, replacing the salvaged suites The independent-account test drives the real load_pool() -> _refresh_entry() path and asserts the second account POSTs its own refresh token and keeps its principal (red on base: the row silently became account A). The alias test pins both halves of the same-account rule: never fall back onto an older singleton, always follow a newer re-auth. --- ...edential_pool_codex_singleton_isolation.py | 106 ++++++ .../test_credential_pool_singleton_replay.py | 313 ------------------ 2 files changed, 106 insertions(+), 313 deletions(-) create mode 100644 tests/agent/test_credential_pool_codex_singleton_isolation.py delete mode 100644 tests/agent/test_credential_pool_singleton_replay.py diff --git a/tests/agent/test_credential_pool_codex_singleton_isolation.py b/tests/agent/test_credential_pool_codex_singleton_isolation.py new file mode 100644 index 0000000000..f57ab0392c --- /dev/null +++ b/tests/agent/test_credential_pool_codex_singleton_isolation.py @@ -0,0 +1,106 @@ +"""Codex pool entries and the auth.json singleton: who may adopt whose tokens. + +``manual:device_code`` is ambiguous — a legacy alias of the singleton or an independent account +added with ``hermes auth add openai-codex``. Adopting the singleton into an independent account +silently turned two logins into one (both then hit the same usage limit; salvaged from #100423 and +#106788, cluster #92198 / #95297, issue #106705). +""" +from __future__ import annotations + +import base64 +import json +import time + +import pytest + +import hermes_cli.auth as auth_mod +from agent.credential_pool import load_pool + + +def _jwt(account: str, sub: str, exp: float) -> str: + def seg(obj: dict) -> str: + return base64.urlsafe_b64encode(json.dumps(obj).encode()).rstrip(b"=").decode() + + payload = {"exp": int(exp), "sub": sub, "https://api.openai.com/auth": {"chatgpt_account_id": account}} + return f"{seg({'alg': 'none'})}.{seg(payload)}.sig" + + +def _iso(ts: float) -> str: + return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime(ts)) + + +def _write_store(home, singleton_tokens: dict, singleton_last_refresh: str, manual: dict) -> None: + home.mkdir(parents=True, exist_ok=True) + (home / "auth.json").write_text(json.dumps({ + "version": 1, + "active_provider": "openai-codex", + "providers": {"openai-codex": { + "tokens": singleton_tokens, "last_refresh": singleton_last_refresh, "auth_mode": "chatgpt"}}, + "credential_pool": {"openai-codex": [ + {"id": "seeded", "label": "device_code", "auth_type": "oauth", "priority": 0, + "source": "device_code", **singleton_tokens, "last_refresh": singleton_last_refresh}, + {"id": "manual", "label": "second", "auth_type": "oauth", "priority": 1, + "source": "manual:device_code", **manual}, + ]}, + }), encoding="utf-8") + + +@pytest.fixture +def home(tmp_path, monkeypatch): + home = tmp_path / "hermes" + monkeypatch.setenv("HERMES_HOME", str(home)) + return home + + +def _stub_refresh(monkeypatch, minted: str, minted_rt: str, posted: list) -> None: + def fake(access_token, refresh_token): + posted.append(refresh_token) + return {"access_token": minted, "refresh_token": minted_rt, "last_refresh": _iso(time.time())} + + monkeypatch.setattr(auth_mod, "refresh_codex_oauth_pure", fake) + + +def test_independent_manual_account_refreshes_with_its_own_pair(home, monkeypatch): + """An independent second account never adopts the singleton: its refresh POSTs its OWN refresh + token, and the persisted row still identifies the second principal afterwards.""" + now = time.time() + a_at = _jwt("acct-A", "user-A", now + 8 * 3600) + b_at = _jwt("acct-B", "user-B", now + 60) # expiring → the pool defers it to _refresh_entry() + _write_store(home, {"access_token": a_at, "refresh_token": "rt-A"}, _iso(now - 3600), + {"access_token": b_at, "refresh_token": "rt-B", "last_refresh": _iso(now - 7200)}) + posted: list = [] + b_new = _jwt("acct-B", "user-B", now + 8 * 3600) + _stub_refresh(monkeypatch, b_new, "rt-B2", posted) + + pool = load_pool("openai-codex") + refreshed = pool._refresh_entry(next(e for e in pool.entries() if e.id == "manual"), force=False) + + assert posted == ["rt-B"] + assert refreshed is not None and refreshed.refresh_token == "rt-B2" + on_disk = json.loads((home / "auth.json").read_text(encoding="utf-8")) + manual = next(e for e in on_disk["credential_pool"]["openai-codex"] if e["id"] == "manual") + assert (manual["access_token"], manual["refresh_token"]) == (b_new, "rt-B2") + # The singleton (account A) is untouched by account B's rotation. + assert on_disk["providers"]["openai-codex"]["tokens"] == {"access_token": a_at, "refresh_token": "rt-A"} + + +def test_same_account_alias_adopts_only_a_newer_singleton(home, monkeypatch): + """A legacy alias (same principal) must follow a singleton that was re-authed AFTER it, but must + not fall back onto a singleton older than its own rotation — that replays a consumed token.""" + now = time.time() + alias_at = _jwt("acct-A", "user-A", now + 8 * 3600) + stale_singleton_at = _jwt("acct-A", "user-A", now + 3600) + _write_store(home, {"access_token": stale_singleton_at, "refresh_token": "rt-consumed"}, _iso(now - 7200), + {"access_token": alias_at, "refresh_token": "rt-alias", "last_refresh": _iso(now - 60)}) + pool = load_pool("openai-codex") + alias = next(e for e in pool.entries() if e.id == "manual") + + synced = pool._sync_entry_from_auth_store(alias) + assert (synced.access_token, synced.refresh_token) == (alias_at, "rt-alias") + + # The user re-authenticates the singleton (newer stamp, same principal): the alias follows. + fresh_at = _jwt("acct-A", "user-A", now + 9 * 3600) + _write_store(home, {"access_token": fresh_at, "refresh_token": "rt-fresh"}, _iso(now + 5), + {"access_token": alias_at, "refresh_token": "rt-alias", "last_refresh": _iso(now - 60)}) + synced = load_pool("openai-codex")._sync_entry_from_auth_store(alias) + assert (synced.access_token, synced.refresh_token) == (fresh_at, "rt-fresh") diff --git a/tests/agent/test_credential_pool_singleton_replay.py b/tests/agent/test_credential_pool_singleton_replay.py deleted file mode 100644 index e69999cbc0..0000000000 --- a/tests/agent/test_credential_pool_singleton_replay.py +++ /dev/null @@ -1,313 +0,0 @@ -"""Regression tests for #106705: Codex manual:device_code pool entries must -not replay an already-consumed refresh token by re-adopting a stale singleton. - -Causal chain (agent/credential_pool.py): - -1. ``_sync_entry_from_auth_store`` adopts differing singleton tokens for BOTH - ``device_code`` and ``manual:device_code`` Codex entries with no staleness - proof. -2. A successful pool-side rotation persists the fresh chain into the - credential-pool store, but ``_sync_device_code_entry_to_auth_store`` - deliberately skips singleton write-back for ``manual:*`` sources - (independent-credential contract, #39236), so the singleton stays one - rotation behind. -3. The next refresh syncs that STALE singleton over the pool's fresh entry and - POSTs the consumed refresh token again -> ``refresh_token_reused``. - -The fix gates adoption on the singleton being provably newer (``last_refresh`` -comparison); these tests run the real production path (``load_pool`` -> -``_refresh_entry``) with only the HTTP transport boundary mocked. -""" - -from __future__ import annotations - -import base64 -import json -from datetime import datetime, timezone - -import pytest - -from agent.credential_pool import load_pool - -# Synthetic JWT-ish access tokens carrying a far-future expiry claim so -# token-expiry probes see a valid token and the refresh path is driven by -# ``force`` alone. -_FAR_FUTURE_EXP = 4102444800 # 2100-01-01 - - -def _jwt(exp: int = _FAR_FUTURE_EXP) -> str: - def _part(payload: dict) -> str: - raw = json.dumps(payload, separators=(",", ":")).encode("utf-8") - return base64.urlsafe_b64encode(raw).decode("ascii").rstrip("=") - - return f"{_part({'alg': 'none', 'typ': 'JWT'})}.{_part({'exp': exp})}.sig" - - -def _iso(seconds: float) -> str: - return datetime.fromtimestamp(seconds, tz=timezone.utc).isoformat().replace("+00:00", "Z") - - -# Rotation chains: old -> new1 -> new2. Timestamps strictly increase so the -# ``last_refresh`` ordering is provable. -_T_OLD = 1_800_000_000.0 -_T_NEW1 = _T_OLD + 600.0 - -_AT_OLD, _RT_OLD = _jwt(), "rt-old" -_AT_NEW1, _RT_NEW1 = _jwt(), "rt-new1" -_AT_NEW2, _RT_NEW2 = _jwt(), "rt-new2" - - -def _manual_entry_payload(id: str = "manual-1", access_token: str = _AT_OLD, - refresh_token: str = _RT_OLD, last_refresh=None) -> dict: - return { - "id": id, - "label": "manual codex grant", - "auth_type": "oauth", - "priority": 0, - "source": "manual:device_code", - "access_token": access_token, - "refresh_token": refresh_token, - "last_refresh": last_refresh, - } - - -def _store(provider_state: dict, pool_entries: list) -> dict: - return { - "version": 1, - "providers": {"openai-codex": provider_state}, - "credential_pool": {"openai-codex": pool_entries}, - } - - -def _tokens_state(access_token: str, refresh_token: str, last_refresh) -> dict: - return { - "tokens": {"access_token": access_token, "refresh_token": refresh_token}, - "last_refresh": last_refresh, - } - - -def _write_store(tmp_path, monkeypatch, store: dict) -> None: - home = tmp_path / "hermes" - home.mkdir(parents=True, exist_ok=True) - (home / "auth.json").write_text(json.dumps(store, indent=2)) - monkeypatch.setenv("HERMES_HOME", str(home)) - - -def _install_fake_refresh(monkeypatch, chains: dict, stamp=_T_NEW1 + 600.0): - """Mock only the HTTP transport; record every refresh_token POSTed.""" - posted = [] - - def fake_refresh(access_token, refresh_token, **kwargs): - posted.append(refresh_token) - if refresh_token not in chains: - raise AssertionError(f"unexpected refresh POST for {refresh_token!r}") - at, rt = chains[refresh_token] - return {"access_token": at, "refresh_token": rt, "last_refresh": _iso(stamp)} - - monkeypatch.setattr( - "agent.credential_pool.auth_mod.refresh_codex_oauth_pure", fake_refresh - ) - return posted - - -class TestPoolRotationDoesNotReplayConsumedRefreshToken: - """L1 + L6: after a pool-side rotation, the stale singleton must not win.""" - - def test_second_forced_refresh_uses_rotated_chain_not_stale_singleton(self, tmp_path, monkeypatch): - """L1 — the reporter's exact scenario. - - Entry holds the chain rotated by the pool (new1, stamped new1-time); - the singleton still holds the consumed old chain (stamped old-time — - write-back deliberately skipped for manual sources). A second forced - refresh must POST the entry's own new1 refresh token — NOT re-adopt - the stale singleton and replay old. - """ - _write_store(tmp_path, monkeypatch, _store( - _tokens_state(_AT_OLD, _RT_OLD, _iso(_T_OLD)), - [_manual_entry_payload( - access_token=_AT_NEW1, refresh_token=_RT_NEW1, last_refresh=_iso(_T_NEW1), - )], - )) - posted = _install_fake_refresh(monkeypatch, {_RT_NEW1: (_AT_NEW2, _RT_NEW2)}) - - pool = load_pool("openai-codex") - updated = pool._refresh_entry(pool._entries[0], force=True) - - assert updated is not None - assert updated.refresh_token == _RT_NEW2 - # The consumed old refresh token must never be re-POSTed. - assert posted == [_RT_NEW1] - - def test_recover_path_does_not_adopt_stale_singleton_after_failed_post(self, tmp_path, monkeypatch): - """L6 — failed-POST recovery must not regress the fresh chain. - - After the pool rotated to new1 (write-back skipped), a refresh POST - that fails must NOT make ``_recover_failed_refresh`` adopt the stale - old singleton as "newer tokens" — that regression is exactly the - replay loop. - """ - _write_store(tmp_path, monkeypatch, _store( - _tokens_state(_AT_OLD, _RT_OLD, _iso(_T_OLD)), - [_manual_entry_payload( - access_token=_AT_NEW1, refresh_token=_RT_NEW1, last_refresh=_iso(_T_NEW1), - )], - )) - - def failing_refresh(access_token, refresh_token, **kwargs): - raise RuntimeError("synthetic network failure") - - monkeypatch.setattr( - "agent.credential_pool.auth_mod.refresh_codex_oauth_pure", failing_refresh - ) - - pool = load_pool("openai-codex") - pool._refresh_entry(pool._entries[0], force=True) - - # The entry must not be regressed onto the stale singleton chain. - entry_now = next(e for e in pool._entries if e.id == "manual-1") - assert entry_now.refresh_token == _RT_NEW1 - assert entry_now.access_token == _AT_NEW1 - - -class TestFreshSingletonAdoptionStillWorks: - """L2 — #70111 regression guard: a provably NEWER singleton must win.""" - - def test_newer_singleton_is_adopted_over_stale_pool_entry(self, tmp_path, monkeypatch): - """The fleet-outage fix (commit 7380b48589) must keep working. - - The singleton was rotated by another process (e.g. ``hermes model`` - re-auth); the pool entry still holds the old consumed chain. The sync - must adopt the newer singleton and refresh with IT. - """ - _write_store(tmp_path, monkeypatch, _store( - _tokens_state(_AT_NEW1, _RT_NEW1, _iso(_T_NEW1)), - [_manual_entry_payload( - access_token=_AT_OLD, refresh_token=_RT_OLD, last_refresh=_iso(_T_OLD), - )], - )) - posted = _install_fake_refresh(monkeypatch, {_RT_NEW1: (_AT_NEW2, _RT_NEW2)}) - - pool = load_pool("openai-codex") - updated = pool._refresh_entry(pool._entries[0], force=True) - - assert updated is not None - assert updated.refresh_token == _RT_NEW2 - assert posted == [_RT_NEW1] - - def test_refresh_token_only_singleton_is_adopted(self, tmp_path, monkeypatch): - """L2b — refresh_token-only singleton (the consumed-access branch). - - Another process rotated and the access_token was consumed; only the - new refresh_token remains on disk. Adoption must still fire (newer - ``last_refresh``) so the consumed token is not replayed. - """ - _write_store(tmp_path, monkeypatch, _store( - _tokens_state("", _RT_NEW1, _iso(_T_NEW1)), - [_manual_entry_payload( - access_token=_AT_OLD, refresh_token=_RT_OLD, last_refresh=_iso(_T_OLD), - )], - )) - posted = _install_fake_refresh(monkeypatch, {_RT_NEW1: (_AT_NEW2, _RT_NEW2)}) - - pool = load_pool("openai-codex") - updated = pool._refresh_entry(pool._entries[0], force=True) - - assert updated is not None - assert updated.refresh_token == _RT_NEW2 - assert posted == [_RT_NEW1] - - -class TestTimestampEdgeCases: - """L3 — missing timestamps fail open to the historical adopt-on-difference.""" - - def test_missing_entry_timestamp_still_adopts_differing_singleton(self, tmp_path, monkeypatch): - """Entry carries no ``last_refresh`` (pre-stamping pool writer). - - Old behavior (adopt on difference) must be preserved so a fresh - re-auth is never stranded when the entry side has no timestamp. - """ - _write_store(tmp_path, monkeypatch, _store( - _tokens_state(_AT_NEW1, _RT_NEW1, _iso(_T_NEW1)), - [_manual_entry_payload( - access_token=_AT_OLD, refresh_token=_RT_OLD, last_refresh=None, - )], - )) - posted = _install_fake_refresh(monkeypatch, {_RT_NEW1: (_AT_NEW2, _RT_NEW2)}) - - pool = load_pool("openai-codex") - updated = pool._refresh_entry(pool._entries[0], force=True) - - assert updated is not None - assert updated.refresh_token == _RT_NEW2 - assert posted == [_RT_NEW1] - - def test_missing_singleton_timestamp_still_adopts_differing_singleton(self, tmp_path, monkeypatch): - """Singleton carries no ``last_refresh`` (legacy auth.json writer). - - Cannot prove the singleton older -> fall back to adopt-on-difference - so a fresh re-auth from a legacy writer is not stranded either. - """ - _write_store(tmp_path, monkeypatch, _store( - _tokens_state(_AT_NEW1, _RT_NEW1, None), - [_manual_entry_payload( - access_token=_AT_OLD, refresh_token=_RT_OLD, last_refresh=_iso(_T_OLD), - )], - )) - posted = _install_fake_refresh(monkeypatch, {_RT_NEW1: (_AT_NEW2, _RT_NEW2)}) - - pool = load_pool("openai-codex") - updated = pool._refresh_entry(pool._entries[0], force=True) - - assert updated is not None - assert updated.refresh_token == _RT_NEW2 - assert posted == [_RT_NEW1] - - -class TestDeviceCodeSeededEntries: - """L4 — the singleton-seeded source is unaffected by the guard.""" - - def test_device_code_entry_rotates_and_writeback_converges(self, tmp_path, monkeypatch): - _write_store(tmp_path, monkeypatch, _store( - _tokens_state(_AT_OLD, _RT_OLD, _iso(_T_OLD)), - [], - )) - posted = _install_fake_refresh(monkeypatch, {_RT_OLD: (_AT_NEW1, _RT_NEW1)}) - - pool = load_pool("openai-codex") - seeded = [e for e in pool._entries if e.source == "device_code"] - assert seeded, "device_code entry must be seeded from the singleton" - - updated = pool._refresh_entry(seeded[0], force=True) - assert updated is not None - assert updated.refresh_token == _RT_NEW1 - assert posted == [_RT_OLD] - - # Write-back must converge the singleton onto the fresh chain. - on_disk = json.loads((tmp_path / "hermes" / "auth.json").read_text()) - synced = on_disk["providers"]["openai-codex"]["tokens"] - assert synced["refresh_token"] == _RT_NEW1 - - -class TestIndependentAccountNotClobbered: - """L5 — #39236 guard: independent manual grants keep their own chain.""" - - def test_independent_manual_entry_with_newer_own_timestamp_is_not_overwritten(self, tmp_path, monkeypatch): - """An independent account's entry whose own chain is NEWER (rotated - by this pool) must never be overwritten by the older singleton — - same mechanism as L1, asserted as the #39236 no-clobber contract. - """ - _write_store(tmp_path, monkeypatch, _store( - _tokens_state(_AT_OLD, _RT_OLD, _iso(_T_OLD)), - [_manual_entry_payload( - id="indep-1", - access_token=_AT_NEW1, refresh_token=_RT_NEW1, last_refresh=_iso(_T_NEW1), - )], - )) - posted = _install_fake_refresh(monkeypatch, {_RT_NEW1: (_AT_NEW2, _RT_NEW2)}) - - pool = load_pool("openai-codex") - updated = pool._refresh_entry(pool._entries[0], force=True) - - assert updated is not None - assert updated.refresh_token == _RT_NEW2 - assert _RT_OLD not in posted From 03fee43ca344ead7245a3b0ae20d38de0ae75642 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 20:50:41 -0700 Subject: [PATCH 63/65] chore: map contributor email for @Caelier (#100423 salvage) --- contributors/emails/yapache@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/yapache@gmail.com diff --git a/contributors/emails/yapache@gmail.com b/contributors/emails/yapache@gmail.com new file mode 100644 index 0000000000..7eba2f413e --- /dev/null +++ b/contributors/emails/yapache@gmail.com @@ -0,0 +1 @@ +Caelier From ad651b825076d31f66cf77c76304c60ab5511a66 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 21:44:23 -0700 Subject: [PATCH 64/65] =?UTF-8?q?feat(gateway):=20RoutingIdentity=20?= =?UTF-8?q?=E2=80=94=20one=20frozen=20identity=20per=20inbound=20event?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A multiplexed gateway answered "which bot received this / who may admit it / where does it run" in three places (`_transport_owner`, `_authorization_home_for_source`, `_resolve_profile_home_for_source` + `_session_key_profile`) that agreed only because they read the same fallback chain. `gateway/session_identity.py` answers them once: `resolve_identity()` folds `_admit_primary_source` + `_stamp_routed_profile` + the transport-owner lookup and pins a frozen `RoutingIdentity` (transport_profile, runtime_profile, authorization_home, runtime_home, weak transport ref) on the source as a wire-invisible attribute, like `_transport_adapter_ref`. Under multiplexing a route to an unserved profile raises `IdentityUnresolved` instead of a `None`-means-default return; `"default"` is spelled out inside the object. Additive: the existing helpers become thin readers of the identity when it is present and keep their fallback chain when it is not, `source.profile` stays the serialized runtime profile (None on the wire ⇔ default) and every historical `agent:main` key is byte-identical. `replace_source()` copies a source without losing its provenance (run_topics used to hand-copy the transport ref). Phase 1 of #88715; the gateway rows of #90142 / #93943. --- gateway/AGENTS.md | 9 + gateway/authz_mixin.py | 11 +- gateway/platforms/base.py | 10 +- gateway/run.py | 9 +- gateway/run_adapters.py | 39 ++-- gateway/run_topics.py | 9 +- gateway/session_identity.py | 183 ++++++++++++++++++ gateway/session_recovery.py | 7 +- tests/gateway/test_session_identity.py | 147 ++++++++++++++ .../developer-guide/multiplexing-gateway.md | 24 ++- 10 files changed, 405 insertions(+), 43 deletions(-) create mode 100644 gateway/session_identity.py create mode 100644 tests/gateway/test_session_identity.py diff --git a/gateway/AGENTS.md b/gateway/AGENTS.md index 3f7fc691c9..d41911e2d2 100644 --- a/gateway/AGENTS.md +++ b/gateway/AGENTS.md @@ -133,6 +133,15 @@ gateway under the backend, and do NOT "fix" update locks by widening the tree-ki ## Profile scope (adapters, turns, and everything between turns) +- **One identity per inbound event.** `gateway/session_identity.py::resolve_identity` answers + "which bot received it / who may admit it / where does it run" ONCE per event and pins a frozen + `RoutingIdentity` on the source (wire-invisible, like `_transport_adapter_ref`); + `_transport_owner`, `_authorization_home_for_source`, `_resolve_profile_home_for_source`, + `_session_key_profile` and `_resolve_profile_for_key` read it when present and fall back to + their old chain only for sources nothing resolved (restored rows, hand-built sources). Never + derive a second answer next to the identity; extend the object. `transport_profile` ≠ + `runtime_profile` is normal (shared bot → routed satellite). A source copy goes through + `session_identity.replace_source` so the identity travels with it. - **Token locks.** An adapter that connects with a unique credential (bot token, API key) calls `acquire_scoped_lock()` from `gateway.status` in `connect()`/`start()` and `release_scoped_lock()` in `disconnect()`/`stop()`, so two profiles cannot share one credential. Canonical: diff --git a/gateway/authz_mixin.py b/gateway/authz_mixin.py index 6ea612b7bd..dc9a98af87 100644 --- a/gateway/authz_mixin.py +++ b/gateway/authz_mixin.py @@ -293,13 +293,18 @@ class GatewayAuthorizationMixin: return (adapter, profile) if registered else None def _authorization_home_for_source(self, source: SessionSource): - """HERMES_HOME whose allowlist admits *source*: the ingress-stamped transport home, else the home of - the profile owning the adapter that delivers it. ``None`` = authorize in the ambient scope - (multiplex off, or no live adapter — the check then fails closed on its own). + """HERMES_HOME whose allowlist admits *source*: the identity's transport home (or the + ingress-stamped one), else the home of the profile owning the adapter that delivers it. + ``None`` = authorize in the ambient scope (multiplex off, or no live adapter — the check then + fails closed on its own). Inside a routed satellite's turn the ambient scope is the satellite's, whose ``.env`` has no token/allowlist; every authorization decision made mid-turn (``/topic``, sibling ``/stop``, plugin injection, voice, auto-resume) must read the admitting bot's allowlist instead.""" + from gateway.session_identity import identity_of + identity = identity_of(source) + if identity is not None: + return identity.authorization_home if identity.multiplexed else None stamped = getattr(source, "_authorization_profile_home", None) if stamped is not None: return Path(stamped) diff --git a/gateway/platforms/base.py b/gateway/platforms/base.py index 83b8a03985..0aeadcf2fe 100644 --- a/gateway/platforms/base.py +++ b/gateway/platforms/base.py @@ -2307,9 +2307,13 @@ class BasePlatformAdapter(ABC): def _session_key_profile(self, source: Optional[Any] = None) -> Optional[str]: """Profile namespace for an adapter-derived session key. Ingress runs BEFORE the runner stamps ``source.profile``, so without this every bot in a multiplexed gateway shares one - ``agent:main:`` lane. Order: ``source.profile`` → ``_owner_profile`` → session-store - resolver; getattr-guarded (object.__new__ in tests), type-checked (no MagicMock in the - key).""" + ``agent:main:`` lane. Order: pinned ``RoutingIdentity`` → ``source.profile`` → + ``_owner_profile`` → session-store resolver; getattr-guarded (object.__new__ in tests), + type-checked (no MagicMock in the key).""" + from gateway.session_identity import identity_of + identity = identity_of(source) + if identity is not None: + return identity.session_key_profile for candidate in ( getattr(source, "profile", None) if source is not None else None, getattr(self, "_owner_profile", None)): diff --git a/gateway/run.py b/gateway/run.py index cbfff36c69..080420bb5d 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -4336,11 +4336,16 @@ class GatewayRunner( return None def _resolve_profile_home_for_source(self, source: SessionSource) -> "Path": - """Resolve which profile's HERMES_HOME serves this source: ``source.profile``, then - ``_profile_name_for_source`` (sources bypassing ``build_source``), then the active profile.""" + """Resolve which profile's HERMES_HOME serves this source: the pinned identity's runtime + home, else ``source.profile``, then ``_profile_name_for_source`` (sources bypassing + ``build_source``), then the active profile.""" from gateway.profile_routing import ProfileRouteRejected + from gateway.session_identity import identity_of from hermes_cli.profiles import get_active_profile_name, get_profile_dir, profile_exists from hermes_constants import get_hermes_home + identity = identity_of(source) + if identity is not None: + return identity.runtime_home explicit_profile = None # explicitly requested (source or routing) vs. default fallback try: name = (source.profile or "").strip() or self._profile_name_for_source(source) diff --git a/gateway/run_adapters.py b/gateway/run_adapters.py index bb9aa3b9e9..bedf66a852 100644 --- a/gateway/run_adapters.py +++ b/gateway/run_adapters.py @@ -1348,12 +1348,18 @@ class GatewayAdapterLifecycleMixin: """``scope_factory(profile_home)`` or a nullcontext when the profile home is unknown.""" return scope_factory(profile_home) if profile_home is not None else contextlib.nullcontext() - @staticmethod - def _stamp_event_profile(event, profile_name: str) -> None: - """Best-effort: stamp ``source.profile`` on an inbound event that has none yet.""" + def _stamp_event_profile(self, event, profile_name: str) -> None: + """Best-effort: pin the secondary's identity on an inbound event (stamps ``source.profile`` + when none yet). A source that cannot resolve keeps today's fallback readers.""" + source = getattr(event, "source", None) + if source is None: + return with suppress(Exception): - if getattr(event, "source", None) is not None and not event.source.profile: - event.source.profile = profile_name + from gateway.session_identity import resolve_identity + resolve_identity(source, runner=self, transport_profile=profile_name) + with suppress(Exception): + if not source.profile: + source.profile = profile_name def _make_profile_message_handler(self, profile_name: str): """Message handler that stamps source.profile, then delegates under the profile scope @@ -1415,23 +1421,14 @@ class GatewayAdapterLifecycleMixin: return _handler def _admit_primary_source(self, source, default_home: Path) -> Optional[Path]: - """Stamp the transport home (authorization) and routed profile on a primary-adapter source and - return the runtime home to scope the turn under; ``None`` when the route targets an unserved - profile. ``_authorization_profile_home`` is in-process only (serialization ignores dynamic attrs); - route ≠ admitting bot.""" - source._authorization_profile_home = default_home - if ( - not getattr(source, "profile", None) - and getattr(source, "profile_route_rejected", False) is not True - and not self._stamp_routed_profile(source) - ): - source.profile_route_rejected = True - if getattr(source, "profile_route_rejected", False) is True: + """Resolve the primary-adapter source's identity (transport home for authorization, routed + profile for the runtime) and return the runtime home to scope the turn under; ``None`` when + the route targets an unserved profile. Route ≠ admitting bot.""" + from gateway.session_identity import IdentityUnresolved, resolve_identity + try: + return resolve_identity(source, runner=self, primary_home=default_home).runtime_home + except IdentityUnresolved: return None - return ( - self._resolve_profile_home_for_source(source) - if getattr(source, "profile", None) else default_home - ) def _stamp_routed_profile(self, source) -> bool: """Stamp ``source.profile`` from ``profile_routes``; False when the route is rejected.""" diff --git a/gateway/run_topics.py b/gateway/run_topics.py index 05b6abf70a..cc467c1b43 100644 --- a/gateway/run_topics.py +++ b/gateway/run_topics.py @@ -5,7 +5,6 @@ from __future__ import annotations import asyncio -import dataclasses import logging import re import time @@ -434,12 +433,10 @@ class GatewayTopicThreadsMixin: return copied_source = source with suppress(Exception): - copied_source = dataclasses.replace(source) - # Keep the live transport owner; multiplex routes may run under a + # Keep the live transport owner and identity; multiplex routes may run under a # profile that does not own the Discord adapter/token. - transport_ref = getattr(source, "_transport_adapter_ref", None) - if transport_ref is not None: - setattr(copied_source, "_transport_adapter_ref", transport_ref) + from gateway.session_identity import replace_source + copied_source = replace_source(source) future = safe_schedule_threadsafe( make_coro(copied_source), loop, logger=logger, log_message=f"{label} failed to schedule", ) diff --git a/gateway/session_identity.py b/gateway/session_identity.py new file mode 100644 index 0000000000..d746cb6e4f --- /dev/null +++ b/gateway/session_identity.py @@ -0,0 +1,183 @@ +"""One frozen routing identity per inbound gateway event. + +A multiplexed gateway answers three questions about every event, and until now answered them in +three places that only agreed because they read the same fallback chain: WHICH bot received it +(``_transport_owner``), WHO may admit it (``_authorization_home_for_source``) and WHERE the turn +runs (``_resolve_profile_home_for_source`` / ``_session_key_profile``). :func:`resolve_identity` +answers all three once and pins the result on the source as a wire-invisible dynamic attribute +(like ``_transport_adapter_ref``); the existing helpers read it when present and keep their +fallback chain when absent, so a source built outside the runner still resolves as before. + +``SessionSource.profile`` stays the serialized runtime profile — ``None`` on the wire means the +receiving bot's own profile — so nothing here changes the wire format or any historical +``agent:main`` key. +""" +from __future__ import annotations + +import dataclasses +import weakref +from dataclasses import dataclass, field +from pathlib import Path +from typing import TYPE_CHECKING, Any, Optional + +if TYPE_CHECKING: + from gateway.session import SessionSource + +_IDENTITY_ATTR = "_identity" +# Wire-invisible provenance copied alongside the identity when a source is duplicated. +_PROVENANCE_ATTRS = ("_transport_adapter_ref", "_authorization_profile_home", _IDENTITY_ATTR) + + +class IdentityUnresolved(RuntimeError): + """Under multiplexing the event's runtime profile could not be established (an explicit + ``profile_routes`` entry targets a profile this gateway does not serve). Callers drop the + event; it must never fall through to the default profile.""" + + +@dataclass(frozen=True) +class RoutingIdentity: + """Everything a turn needs to know about who it is, resolved once at ingress. + + ``transport_profile`` owns the receiving adapter (its credential and allowlist); + ``runtime_profile`` is the profile that executes the turn — the same name unless a + ``profile_routes`` entry re-homed the event. Both are explicit (``"default"`` is spelled out); + ``None`` never means default here. ``multiplexed`` is False for a standalone gateway, whose + keys stay in the legacy ``agent:main`` namespace whatever profile it was launched with. + """ + + transport_profile: str + runtime_profile: str + authorization_home: Path + runtime_home: Path + multiplexed: bool = True + # Receiving adapter; None for restored/synthetic sources (no live provenance → fail closed). + # Provenance, not identity: two events from the same bot share one identity. + transport: Optional[weakref.ref] = field(default=None, compare=False, hash=False) + + @property + def namespace(self) -> str: + """``agent:`` prefix for this identity's session keys — byte-identical to + :func:`gateway.session._session_key_namespace` for every historical key.""" + from gateway.session import _session_key_namespace + return _session_key_namespace(self.session_key_profile) + + @property + def store_path(self) -> Path: + return self.runtime_home / "state.db" + + @property + def session_key_profile(self) -> Optional[str]: + """The ``profile=`` argument :func:`gateway.session.build_session_key` expects for this + identity: the runtime profile under multiplexing, else ``None`` (legacy namespace).""" + return self.runtime_profile if self.multiplexed else None + + def adapter(self) -> Any: + """The live receiving adapter, or None when it is gone or was never known.""" + return self.transport() if self.transport is not None else None + + +def identity_of(source: Any) -> Optional[RoutingIdentity]: + """The identity pinned on *source* by :func:`resolve_identity`, if any.""" + identity = getattr(source, _IDENTITY_ATTR, None) + return identity if isinstance(identity, RoutingIdentity) else None + + +def replace_source(source: "SessionSource", **changes: Any) -> "SessionSource": + """:func:`dataclasses.replace` that keeps the wire-invisible provenance (transport ref, + authorization home, identity). A plain ``replace`` silently produces a source the runner + can only route through heuristics.""" + copied = dataclasses.replace(source, **changes) + for name in _PROVENANCE_ATTRS: + value = getattr(source, name, None) + if value is not None: + setattr(copied, name, value) + return copied + + +def _name(value: Any) -> Optional[str]: + text = value.strip() if isinstance(value, str) else "" + return text or None + + +def resolve_identity( + source: "SessionSource", *, runner: Any, adapter: Any = None, + transport_profile: Optional[str] = None, primary_home: Optional[Path] = None, +) -> RoutingIdentity: + """Resolve and pin the :class:`RoutingIdentity` of an inbound *source*. + + *adapter* is the receiving adapter when the caller holds it; otherwise the source's own + transport provenance is consulted. *transport_profile* names the receiving bot's owning + profile when the caller knows it by construction (the runner's per-profile handlers); + ``None`` = derive it from the adapter registry, primary when unknown. *primary_home* is the + primary bot's home for authorization (default: the process home, never a per-turn override). + + Stamps ``source.profile`` the way the ingress handlers always did (routed name, else a + secondary's own name; ``None`` stays ``None`` for the primary so the wire is unchanged) and + ``_authorization_profile_home`` for the existing authorization readers. + + Raises :class:`IdentityUnresolved` under multiplexing when the route is rejected. + """ + from gateway.profile_routing import ProfileRouteRejected + from hermes_constants import get_hermes_home, get_process_hermes_home + + multiplexed = bool(getattr(getattr(runner, "config", None), "multiplex_profiles", False)) + primary_profile = _name(getattr(runner, "_primary_profile_name", None)) + if primary_profile is None: + active = getattr(runner, "_active_profile_name", None) + primary_profile = (_name(active()) if callable(active) else None) or "default" + platform = getattr(source, "platform", None) + + owner_profile: Optional[str] = None + if adapter is None: + owner = runner._transport_owner(source) + if owner is not None: + adapter, owner_profile = owner + else: + if getattr(source, "_transport_adapter_ref", None) is None: + source._transport_adapter_ref = weakref.ref(adapter) + _registered, owner_profile = runner._owning_profile(adapter, platform) + transport_name = _name(transport_profile) or _name(owner_profile) or primary_profile + transport_ref = weakref.ref(adapter) if adapter is not None else None + + if not multiplexed: + home = Path(get_hermes_home()) + identity = RoutingIdentity( + transport_profile=primary_profile, runtime_profile=primary_profile, + authorization_home=home, runtime_home=home, multiplexed=False, transport=transport_ref) + setattr(source, _IDENTITY_ATTR, identity) + return identity + + if transport_name == primary_profile: + authorization_home = Path(primary_home) if primary_home is not None else Path(get_process_hermes_home()) + else: + from hermes_cli.profiles import get_profile_dir + authorization_home = get_profile_dir(transport_name) + source._authorization_profile_home = authorization_home + + where = f"{getattr(platform, 'value', platform)}/{getattr(source, 'chat_id', '')}" + if getattr(source, "profile_route_rejected", False) is True: + raise IdentityUnresolved(f"{where}: profile route rejected") + if _name(getattr(source, "profile", None)) is None: + adapter_profile = None if transport_name == primary_profile else transport_name + try: + # The primary keeps the historical one-argument call (its adapter profile is None). + routed = ( + runner._profile_name_for_source(source) if adapter_profile is None + else runner._profile_name_for_source(source, adapter_profile=adapter_profile)) + except ProfileRouteRejected as exc: + source.profile_route_rejected = True + raise IdentityUnresolved(f"{where}: {exc}") from exc + source.profile = routed or adapter_profile + + runtime_name = _name(source.profile) or primary_profile + # A routed runtime goes through the runner's resolver (missing-profile fallback + warning); + # a bot serving its own profile runs where it authorizes. + runtime_home = ( + authorization_home if runtime_name == transport_name + else Path(runner._resolve_profile_home_for_source(source))) + identity = RoutingIdentity( + transport_profile=transport_name, runtime_profile=runtime_name, + authorization_home=authorization_home, runtime_home=runtime_home, + multiplexed=True, transport=transport_ref) + setattr(source, _IDENTITY_ATTR, identity) + return identity diff --git a/gateway/session_recovery.py b/gateway/session_recovery.py index cdc13acd43..770f69295a 100644 --- a/gateway/session_recovery.py +++ b/gateway/session_recovery.py @@ -35,9 +35,14 @@ class SessionRecoveryMixin: def _resolve_profile_for_key(self, source: Optional[SessionSource] = None) -> Optional[str]: """Profile namespace for session keys: None when multiplexing is off (legacy - ``agent:main``), else ``source.profile`` or the active profile.""" + ``agent:main``), else the pinned identity's runtime profile, ``source.profile`` or the + active profile.""" if not getattr(self.config, "multiplex_profiles", False): return None + from gateway.session_identity import identity_of + identity = identity_of(source) + if identity is not None: + return identity.session_key_profile if source is not None and source.profile: return source.profile try: diff --git a/tests/gateway/test_session_identity.py b/tests/gateway/test_session_identity.py new file mode 100644 index 0000000000..201d551b76 --- /dev/null +++ b/tests/gateway/test_session_identity.py @@ -0,0 +1,147 @@ +"""Invariants for ``gateway/session_identity.py`` (#88715 phase 1). + +One frozen ``RoutingIdentity`` per inbound event, resolved by the real ``GatewayRunner`` resolvers +(no patched predicates) against a temp ``HERMES_HOME`` with two served profiles. +""" + +import dataclasses +import weakref +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import patch + +import pytest + +from gateway.config import GatewayConfig, Platform, PlatformConfig +from gateway.pairing import PairingStore +from gateway.platforms.base import BasePlatformAdapter +from gateway.profile_routing import parse_profile_routes +from gateway.session import SessionSource, build_session_key +from gateway.session_identity import ( + IdentityUnresolved, RoutingIdentity, identity_of, replace_source, resolve_identity, +) + + +class _Stub(BasePlatformAdapter): + pass + + +_Stub.__abstractmethods__ = frozenset() + + +def _stub(platform, runner, label): + adapter = _Stub.__new__(_Stub) + adapter.platform, adapter.gateway_runner, adapter.label = platform, runner, label + adapter.config = PlatformConfig(enabled=True, extra={}) + adapter._pending_messages, adapter._active_sessions = {}, {} + return adapter + + +def _runner(home, *, multiplex, routes=()): + from gateway.run import GatewayRunner + + runner = object.__new__(GatewayRunner) + runner.config = GatewayConfig(multiplex_profiles=multiplex) + runner.config.platforms = {Platform.TELEGRAM: PlatformConfig(enabled=True, extra={})} + runner.config.profile_routes = parse_profile_routes(list(routes)) + runner.pairing_store = PairingStore(profile="default") + runner.pairing_stores = {} + runner._primary_profile_name = "default" + primary = _stub(Platform.TELEGRAM, runner, "PRIMARY") + team_b = _stub(Platform.TELEGRAM, runner, "TEAM_B") + team_b.set_owner_profile("team_b") + runner.adapters = {Platform.TELEGRAM: primary} + runner._profile_adapters = {"team_b": {Platform.TELEGRAM: team_b}, "ops": {}} + return SimpleNamespace(runner=runner, home=home, primary=primary, team_b=team_b) + + +@pytest.fixture +def mux(tmp_path, monkeypatch): + """Default home + satellite ``ops`` (routed through the shared bot for chat 72719239) + ``team_b`` + owning its own Telegram bot; a route to unserved ``ghost`` for chat 4040.""" + home = tmp_path / "hh" + for name in ("ops", "team_b"): + (home / "profiles" / name).mkdir(parents=True) + monkeypatch.setenv("HERMES_HOME", str(home)) + served = [("default", home), ("ops", home / "profiles" / "ops"), ("team_b", home / "profiles" / "team_b")] + rig = _runner(home, multiplex=True, routes=[ + {"name": "admin-dm", "platform": "telegram", "profile": "ops", "chat_id": "72719239"}, + {"name": "ghost", "platform": "telegram", "profile": "ghost", "chat_id": "4040"}, + ]) + with patch("hermes_cli.profiles.profiles_to_serve", return_value=served), \ + patch("hermes_cli.profiles.get_profile_dir", side_effect=lambda n: home if n == "default" else home / "profiles" / n), \ + patch("hermes_cli.profiles.profile_exists", return_value=True): + yield rig + + +def test_identity_is_one_frozen_value_that_every_reader_agrees_on(mux): + """Shared bot → routed satellite: transport stays default (authorization), runtime is ``ops``; + the pinned identity is frozen, equal by value, and the legacy readers all report the same + answer it does. A dedicated secondary resolves to itself on both axes.""" + routed = mux.primary.build_source(chat_id="72719239", chat_type="dm", user_id="72719239") + identity = resolve_identity(routed, runner=mux.runner) + + assert identity == RoutingIdentity( + transport_profile="default", runtime_profile="ops", authorization_home=mux.home, + runtime_home=mux.home / "profiles" / "ops") + assert identity.namespace == "agent:ops" and identity.store_path == mux.home / "profiles" / "ops" / "state.db" + assert identity.adapter() is mux.primary + with pytest.raises(dataclasses.FrozenInstanceError): + identity.runtime_profile = "default" + assert identity_of(routed) is identity + # Thin readers: same object, same answers — no second derivation. + assert mux.runner._authorization_home_for_source(routed) == mux.home + assert mux.runner._resolve_profile_home_for_source(routed) == mux.home / "profiles" / "ops" + assert mux.runner._transport_owner(routed) == (mux.primary, None) + assert mux.primary._source_session_key(routed) == "agent:ops:telegram:dm:72719239" + assert mux.runner._session_key_for_source(routed) == mux.primary._source_session_key(routed) + + own = mux.team_b.build_source(chat_id="72719239", chat_type="dm", user_id="72719239") + own_identity = resolve_identity(own, runner=mux.runner, transport_profile="team_b") + assert (own_identity.transport_profile, own_identity.runtime_profile) == ("team_b", "team_b") + assert own_identity.authorization_home == own_identity.runtime_home == mux.home / "profiles" / "team_b" + assert own_identity != identity + # Provenance is not identity: a second event from the same bot has the same identity. + again = mux.primary.build_source(chat_id="72719239", chat_type="dm", user_id="72719239") + assert resolve_identity(again, runner=mux.runner) == identity + assert hash(resolve_identity(again, runner=mux.runner)) == hash(identity) + + +def test_unresolved_under_multiplex_raises_and_never_means_default(mux, tmp_path, monkeypatch): + """A route to an unserved profile raises ``IdentityUnresolved`` (the runner drops the event); + outside multiplexing the same source resolves to an explicit default identity whose keys stay + byte-identical to the legacy ``agent:main`` namespace.""" + rejected = mux.primary.build_source(chat_id="4040", chat_type="dm", user_id="4040") + with pytest.raises(IdentityUnresolved): + resolve_identity(rejected, runner=mux.runner) + assert identity_of(rejected) is None + assert mux.runner._admit_primary_source(rejected, mux.home) is None + + solo_home = tmp_path / "solo" + solo_home.mkdir() + monkeypatch.setenv("HERMES_HOME", str(solo_home)) + solo = _runner(solo_home, multiplex=False) + source = solo.primary.build_source(chat_id="4040", chat_type="dm", user_id="4040") + identity = resolve_identity(source, runner=solo.runner) + assert (identity.transport_profile, identity.runtime_profile, identity.multiplexed) == ("default", "default", False) + assert identity.runtime_home == solo_home + assert identity.namespace == "agent:main" + assert solo.primary._source_session_key(source) == build_session_key(source) == "agent:main:telegram:dm:4040" + assert source.profile is None # wire format untouched + assert solo.runner._authorization_home_for_source(source) is None # ambient scope, as before + + +def test_replace_source_keeps_identity_and_transport_where_dataclasses_replace_drops_them(mux): + routed = mux.primary.build_source(chat_id="72719239", chat_type="dm", user_id="72719239") + identity = resolve_identity(routed, runner=mux.runner) + + bare = dataclasses.replace(routed) + assert identity_of(bare) is None and mux.runner._transport_owner(bare) is None + + copied = replace_source(routed, thread_id="7") + assert copied.thread_id == "7" and copied is not routed + assert identity_of(copied) is identity + assert mux.runner._transport_owner(copied) == (mux.primary, None) + assert isinstance(copied._transport_adapter_ref, weakref.ref) + assert Path(copied._authorization_profile_home) == mux.home + assert mux.primary._source_session_key(copied) == "agent:ops:telegram:dm:72719239:7" diff --git a/website/docs/developer-guide/multiplexing-gateway.md b/website/docs/developer-guide/multiplexing-gateway.md index 613f4891f9..2d6c91e74e 100644 --- a/website/docs/developer-guide/multiplexing-gateway.md +++ b/website/docs/developer-guide/multiplexing-gateway.md @@ -168,13 +168,23 @@ Pairing stores are constructed per served profile. ## Per-bot session lanes Session keys are namespaced by profile (`agent:main` for default, -`agent:` for named profiles). Adapters carry `_owner_profile` -(installed at adapter configuration time, before any inbound event) because -adapter ingress runs before `SessionSource.profile` is stamped; -`_session_key_profile` resolves source stamp → owner profile → store -resolver. Text/media batching, active-session tracking, and the busy-session -guard are all keyed per lane, so two bots sharing a chat do not share a -session lane. +`agent:` for named profiles). Every inbound event carries ONE frozen +`RoutingIdentity` (`gateway/session_identity.py`), resolved by +`resolve_identity()` at the runner's ingress handlers and pinned on the source +as a wire-invisible attribute: `transport_profile` (the bot that received it — +credential, allowlist, `authorization_home`), `runtime_profile` (the routed +profile that executes — `runtime_home`, key `namespace`, `store_path`) and a +weak `transport` ref to the receiving adapter. `"default"` is spelled out; +`None` never means default. Under multiplexing a route to an unserved profile +raises `IdentityUnresolved` and the event is dropped. + +Adapters also carry `_owner_profile` (installed at adapter configuration time, +before any inbound event) because adapter ingress runs before the runner pins +the identity; `_session_key_profile` resolves identity → source stamp → owner +profile → store resolver. Text/media batching, active-session tracking, and the +busy-session guard are all keyed per lane, so two bots sharing a chat do not +share a session lane. Copy a source with `session_identity.replace_source`, not +`dataclasses.replace`, or the copy loses its transport and identity. ## Control plane From c07708671d2952f1790f696fc134b1cb8b54ea96 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Fri, 18 Sep 2026 21:53:30 -0700 Subject: [PATCH 65/65] fix(gateway): every adapter session key goes through one seam (+ lint) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A secondary-owned Yuanbao bot keyed its per-group dispatch queue and RecallGuard entries with the free `build_session_key(source)` — no profile, so `agent:main:` — while `handle_message` popped under `agent::`. Two derivations of one identity: the group queue was shared across bots and the RecallGuard entries leaked. Weixin, Telegram's photo batch, Slack's thread key and Raft's wake key each carried their own copy of the call as well. Every adapter-side key now comes from `BasePlatformAdapter._source_session_key` / `_event_session_key` (owner namespace, runner-seeded isolation flags, and — after the RoutingIdentity PR — the pinned identity). Weixin's `_text_batch_key` override is deleted (the base does the same). Slack's thread key reads the isolation flags from the adapter config the runner seeds, not the store's. Lint: pattern P32 in `scripts/ci/profile_scope_patterns.json` flags `build_session_key(` / `SessionSource(` under `gateway/platforms/**` and `plugins/platforms/**` except `platforms/base.py`; the checker gains an optional `path_regex` per pattern. Advisory, like every other pattern. Phase 2 of #88715. --- gateway/platforms/ADDING_A_PLATFORM.md | 7 +- gateway/platforms/weixin.py | 6 -- gateway/platforms/yuanbao.py | 8 +-- plugins/platforms/raft/adapter.py | 6 +- plugins/platforms/slack/adapter.py | 16 ++--- plugins/platforms/telegram/adapter.py | 6 +- scripts/check_profile_scope_patterns.py | 7 +- scripts/ci/profile_scope_patterns.json | 12 +++- .../gateway/test_adapter_session_key_seam.py | 69 +++++++++++++++++++ tests/gateway/test_slack.py | 3 + .../test_check_profile_scope_patterns.py | 24 +++++++ 11 files changed, 127 insertions(+), 37 deletions(-) create mode 100644 tests/gateway/test_adapter_session_key_seam.py diff --git a/gateway/platforms/ADDING_A_PLATFORM.md b/gateway/platforms/ADDING_A_PLATFORM.md index 9887ad5dd9..3fb182a3c7 100644 --- a/gateway/platforms/ADDING_A_PLATFORM.md +++ b/gateway/platforms/ADDING_A_PLATFORM.md @@ -141,7 +141,12 @@ def check__requirements() -> bool: ### Key patterns to follow -- Use `self.build_source(...)` to construct `SessionSource` objects +- Use `self.build_source(...)` to construct `SessionSource` objects (never `SessionSource(...)` + directly — the transport provenance and profile route are stamped there) +- Derive every adapter-side session key (batching, per-chat queues, busy detection) through + `self._event_session_key(event)` / `self._source_session_key(source)`, never the free + `build_session_key()` — the seam keys in the owning profile's namespace under a multiplexed + gateway; the advisory lint (`scripts/check_profile_scope_patterns.py`, pattern P32) flags both - Call `self.handle_message(event)` to dispatch inbound messages to the gateway - Use `MessageEvent`, `MessageType` from `gateway.platforms.event` and `SendResult` from base - Use `cache_image_from_bytes`, `cache_audio_from_bytes`, `cache_document_from_bytes` for attachments diff --git a/gateway/platforms/weixin.py b/gateway/platforms/weixin.py index d8c3e835d3..c338b14d0d 100644 --- a/gateway/platforms/weixin.py +++ b/gateway/platforms/weixin.py @@ -913,12 +913,6 @@ class WeixinAdapter(OwnAccessPolicyMixin, BasePlatformAdapter): else: await self.handle_message(event) - def _text_batch_key(self, event: MessageEvent) -> str: - from gateway.session import build_session_key - return build_session_key( - event.source, group_sessions_per_user=self.config.extra.get("group_sessions_per_user", True), - thread_sessions_per_user=self.config.extra.get("thread_sessions_per_user", False), profile=event.source.profile) - async def _collect_media(self, item: Dict[str, Any], media_paths: List[str], media_types: List[str]) -> None: spec = _INBOUND_MEDIA.get(item.get("type")) path, mime = await self._download_media(item, spec) if spec else (None, "") diff --git a/gateway/platforms/yuanbao.py b/gateway/platforms/yuanbao.py index 7c4be70d82..ad1d60284a 100644 --- a/gateway/platforms/yuanbao.py +++ b/gateway/platforms/yuanbao.py @@ -64,7 +64,6 @@ from gateway.platforms.yuanbao_proto import ( encode_send_private_heartbeat, encode_send_group_heartbeat, encode_query_group_info, encode_get_group_member_list, next_seq_no, ) -from gateway.session import build_session_key from gateway.session_transcript import TranscriptReadError logger = logging.getLogger(__name__) @@ -1666,11 +1665,8 @@ class DispatchMiddleware(InboundMiddleware): async def handle(self, ctx: InboundContext, next_fn) -> None: adapter = ctx.adapter - _sk = build_session_key( - ctx.source, - group_sessions_per_user=adapter.config.extra.get("group_sessions_per_user", True), - thread_sessions_per_user=adapter.config.extra.get("thread_sessions_per_user", False), - ) + # The adapter seam: keyed in the owner profile's namespace, same as ``handle_message``. + _sk = adapter._source_session_key(ctx.source) async def _dispatch_inbound_event() -> None: if any(mt.startswith(("application/", "text/")) for mt in ctx.media_types): diff --git a/plugins/platforms/raft/adapter.py b/plugins/platforms/raft/adapter.py index d3d45a5c18..2f4d26ee83 100644 --- a/plugins/platforms/raft/adapter.py +++ b/plugins/platforms/raft/adapter.py @@ -39,7 +39,6 @@ sys.path.insert(0, str(_Path(__file__).resolve().parents[3])) from gateway.config import Platform, PlatformConfig from gateway.platforms.base import BasePlatformAdapter, SendResult, merge_pending_message_event from gateway.platforms.event import MessageEvent, MessageType -from gateway.session import build_session_key from gateway.platforms._shared import coerce_port, profile_scoped as _profile_scoped logger = logging.getLogger(__name__) @@ -491,10 +490,7 @@ class RaftAdapter(BasePlatformAdapter): return if not self._message_handler: return - session_key = build_session_key( - event.source, group_sessions_per_user=self.config.extra.get("group_sessions_per_user", True), - thread_sessions_per_user=self.config.extra.get("thread_sessions_per_user", False), - profile=self._session_key_profile(event.source)) + session_key = self._event_session_key(event) if session_key in self._active_sessions: logger.debug("[raft] Wake queued for busy session %s", session_key) merge_pending_message_event(self._pending_messages, session_key, event) diff --git a/plugins/platforms/slack/adapter.py b/plugins/platforms/slack/adapter.py index ea8374041f..7d39d80b0a 100644 --- a/plugins/platforms/slack/adapter.py +++ b/plugins/platforms/slack/adapter.py @@ -5961,20 +5961,14 @@ class SlackAdapter(BasePlatformAdapter): def _build_thread_session_key( self, channel_id: str, thread_ts: str, user_id: str, team_id: str = "", *, chat_type: str = "group") -> Optional[str]: - """Thread session key via ``build_session_key()`` (honours per-user isolation). - ``chat_type`` must come from the event's ``channel_type``, not the ID prefix (MPIM ids - start with ``G``).""" - session_store = getattr(self, "_session_store", None) - if not session_store: + """Thread session key through the adapter seam (``_source_session_key``: per-user isolation + from the adapter config the runner seeded, owner-profile namespace). ``chat_type`` must come + from the event's ``channel_type``, not the ID prefix (MPIM ids start with ``G``).""" + if not getattr(self, "_session_store", None): return None try: - from gateway.session import build_session_key source = self._thread_session_source(channel_id, thread_ts, user_id, team_id, chat_type) - store_cfg = getattr(session_store, "config", None) - return build_session_key( - source, group_sessions_per_user=getattr(store_cfg, "group_sessions_per_user", True), - thread_sessions_per_user=getattr(store_cfg, "thread_sessions_per_user", False), - profile=self._session_key_profile(source)) + return self._source_session_key(source) except Exception: return None diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index 4d1c7f0d9e..1ba65f1925 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -6271,11 +6271,7 @@ class TelegramAdapter(BasePlatformAdapter): def _photo_batch_key(self, event: MessageEvent, msg: Message) -> str: """Return a batching key for Telegram photos/albums.""" - from gateway.session import build_session_key - session_key = build_session_key( - event.source, group_sessions_per_user=self.config.extra.get("group_sessions_per_user", True), - thread_sessions_per_user=self.config.extra.get("thread_sessions_per_user", False), - profile=self._session_key_profile(event.source)) + session_key = self._event_session_key(event) media_group_id = getattr(msg, "media_group_id", None) return f"{session_key}:album:{media_group_id}" if media_group_id else f"{session_key}:photo-burst" diff --git a/scripts/check_profile_scope_patterns.py b/scripts/check_profile_scope_patterns.py index dc7e7f3821..79dbccfcad 100644 --- a/scripts/check_profile_scope_patterns.py +++ b/scripts/check_profile_scope_patterns.py @@ -49,7 +49,9 @@ def load_patterns(path: Path = PATTERNS) -> list[dict]: data = json.loads(path.read_text(encoding="utf-8")) out = [] for p in data["patterns"]: - out.append({**p, "_rx": re.compile(p["pattern_regex"], re.M)}) + # ``path_regex`` (optional) restricts a pattern to files whose repo-relative path matches. + path_rx = re.compile(p["path_regex"]) if p.get("path_regex") else None + out.append({**p, "_rx": re.compile(p["pattern_regex"], re.M), "_path_rx": path_rx}) return out @@ -63,6 +65,9 @@ def scan_text(rel: str, text: str, patterns: list[dict], lines: set[int] | None findings: list[Finding] = [] src_lines = text.split("\n") for p in patterns: + path_rx = p.get("_path_rx") + if path_rx is not None and not path_rx.search(rel): + continue for m in p["_rx"].finditer(text): line_no = text.count("\n", 0, m.start()) + 1 if lines is not None and line_no not in lines: diff --git a/scripts/ci/profile_scope_patterns.json b/scripts/ci/profile_scope_patterns.json index 50baf25932..94e786cc51 100644 --- a/scripts/ci/profile_scope_patterns.json +++ b/scripts/ci/profile_scope_patterns.json @@ -32,7 +32,7 @@ "P24: 62 hits on main", "P26: 252 hits on main" ], - "usage": "scripts/check_profile_scope_patterns.py --base origin/main [--head HEAD] | --files " + "usage": "scripts/check_profile_scope_patterns.py --base origin/main [--head HEAD] | --files ; optional path_regex restricts a pattern to matching repo-relative paths" }, "patterns": [ { @@ -158,8 +158,16 @@ "id": "P31", "class": "C6", "pattern_regex": "f\"agent:\\{|[\"']agent:[\"']\\s*\\+|session_key\\s*=\\s*f\"[a-z]+:", - "scope_hint": "Every adapter-built session key carries the agent:: namespace (profile 'main' is 'agent:main~'); yuanbao still builds keys with no profile component per the MindDragon probe.", + "scope_hint": "Every adapter-built session key carries the agent:: namespace (profile 'main' is 'agent:main~'); adapters derive keys through _source_session_key / _event_session_key (P32), never a hand-built prefix.", "why": "Rows for a served profile land in the root store; browser/computer_use caches never saw the namespace because turns pass the bare session id." + }, + { + "id": "P32", + "class": "C4", + "path_regex": "^(gateway|plugins)/platforms/(?!base\\.py$).+\\.py$", + "pattern_regex": "\\bbuild_session_key\\(|\\bSessionSource\\(", + "scope_hint": "Inside an adapter derive every key through self._source_session_key(source) / self._event_session_key(event) (owner-profile namespace, runner-seeded isolation flags) and build sources with self.build_source(...) so the transport provenance is kept; only platforms/base.py owns the free calls.", + "why": "Yuanbao keyed its per-group queue and RecallGuard with the free build_session_key() (no profile) while handle_message keyed under agent:: - two derivations of one identity, one lane shared across bots (#88715)." } ] } diff --git a/tests/gateway/test_adapter_session_key_seam.py b/tests/gateway/test_adapter_session_key_seam.py new file mode 100644 index 0000000000..1e807b6fa4 --- /dev/null +++ b/tests/gateway/test_adapter_session_key_seam.py @@ -0,0 +1,69 @@ +"""Every adapter-side session key goes through one seam (#88715, invariant 4). + +A key an adapter derives for batching / queueing / busy detection must be the key the runner +derives for the same event, for a primary bot and for a secondary-owned bot alike. Yuanbao's +``DispatchMiddleware`` used the free ``build_session_key()`` (no profile) while ``handle_message`` +keyed under ``agent::``, so a secondary bot's per-group queue and RecallGuard entries lived +in a lane the runner never popped. +""" + +import asyncio +from types import SimpleNamespace + +import pytest + +from gateway.config import GatewayConfig, Platform, PlatformConfig +from gateway.platforms.event import MessageEvent +from gateway.platforms.yuanbao import DispatchMiddleware, InboundContext, YuanbaoAdapter +from gateway.profile_routing import parse_profile_routes + + +def _yuanbao(owner): + adapter = YuanbaoAdapter(PlatformConfig(extra={ + "app_id": "k", "app_secret": "s", "ws_url": "wss://x", "api_domain": "https://x", + "group_sessions_per_user": True, "thread_sessions_per_user": False})) + adapter.set_owner_profile(owner) + return adapter + + +def _runner(adapter, owner): + from gateway.run import GatewayRunner + + runner = object.__new__(GatewayRunner) + runner.config = GatewayConfig(multiplex_profiles=True) + runner.config.profile_routes = parse_profile_routes([]) + runner._primary_profile_name = "default" + runner.adapters = {} if owner else {Platform.YUANBAO: adapter} + runner._profile_adapters = {owner: {Platform.YUANBAO: adapter}} if owner else {} + adapter.gateway_runner = runner + return runner + + +@pytest.mark.parametrize("owner", [None, "acme"]) +def test_adapter_batch_key_equals_runner_session_key(owner): + adapter = _yuanbao(owner) + runner = _runner(adapter, owner) + source = adapter.build_source(chat_id="grp-1", chat_type="group", user_id="u1") + ctx = InboundContext(adapter=adapter, chat_type="group", chat_id="grp-1", raw_text="hi", msg_id="m1", source=source) + seen = [] + + async def go(): + async def _next(): + pass + + async def _capture(event): + seen.append(adapter._event_session_key(event)) + + adapter.handle_message = _capture + await DispatchMiddleware().handle(ctx, _next) + queued = list(adapter._group_queues) + await asyncio.sleep(0.05) # let the group consumer dispatch once + return queued + + queued = asyncio.run(go()) + runner_key = runner._session_key_for_source(source) + expected_ns = f"agent:{owner}" if owner else "agent:main" + assert runner_key.startswith(expected_ns + ":"), runner_key + assert queued == [runner_key], (queued, runner_key) + assert seen == [runner_key] + assert adapter._processing_msg_ids == {runner_key: "m1"} diff --git a/tests/gateway/test_slack.py b/tests/gateway/test_slack.py index 851acf5005..790014273f 100644 --- a/tests/gateway/test_slack.py +++ b/tests/gateway/test_slack.py @@ -3070,7 +3070,10 @@ class TestThreadReplyHandling: from gateway.session import SessionEntry # Deserialize a legacy routing entry so lifecycle flags have real defaults. + # The thread key with a per-user suffix comes from the adapter's isolation flags (the runner + # seeds them into PlatformConfig.extra); this store has no bearing on the key any more. session_key = "agent:main:slack:group:T_TEAM:C123:123.000:U_USER" + adapter_with_session_store.config.extra["thread_sessions_per_user"] = True mock_session_store._entries = {session_key: SessionEntry.from_dict({ "session_key": session_key, "session_id": "slack-thread-session", diff --git a/tests/scripts/test_check_profile_scope_patterns.py b/tests/scripts/test_check_profile_scope_patterns.py index 7e5bb66322..1e13b8bb03 100644 --- a/tests/scripts/test_check_profile_scope_patterns.py +++ b/tests/scripts/test_check_profile_scope_patterns.py @@ -59,6 +59,30 @@ def test_child_env_from_environ_is_flagged_and_the_scoped_builder_is_not(): assert [f.line for f in mod.scan_text("tools/x.py", hazard, patterns, lines={5})] == [5] +def test_adapter_key_outside_the_seam_is_flagged_only_under_platforms(): + """An adapter that derives a session key with the free ``build_session_key()`` bypasses the + owner-profile seam (``_source_session_key``); the same call in ``platforms/base.py`` (the seam + itself) or in the runner is legitimate and must stay silent.""" + mod = _load() + patterns = mod.load_patterns() + free_key = textwrap.dedent(''' + from gateway.session import build_session_key + + def _batch_key(self, event): + return build_session_key(event.source, profile=event.source.profile) + ''') + seam = textwrap.dedent(''' + def _batch_key(self, event): + return self._event_session_key(event) + ''') + flagged = mod.scan_text("plugins/platforms/acme/adapter.py", free_key, patterns) + assert [(f.line, f.pattern_id, f.pattern_class) for f in flagged] == [(5, "P32", "C4")] + assert [f.pattern_id for f in mod.scan_text("gateway/platforms/acme.py", free_key, patterns)] == ["P32"] + assert mod.scan_text("plugins/platforms/acme/adapter.py", seam, patterns) == [] + for owner in ("gateway/platforms/base.py", "gateway/run_startup.py", "gateway/session_recovery.py"): + assert not [f for f in mod.scan_text(owner, free_key, patterns) if f.pattern_id == "P32"], owner + + def test_lint_is_advisory_and_exits_zero_with_findings(tmp_path, capsys): mod = _load() bad = tmp_path / "bad.py"