From 2a1a883fedabbbf02d4838c8bffaeae78f69e897 Mon Sep 17 00:00:00 2001 From: Markus Werle Date: Wed, 23 Sep 2026 11:50:44 +0200 Subject: [PATCH 001/112] catalog: add jev-skill-router --- plugin-catalog/jev-skill-router.yaml | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 plugin-catalog/jev-skill-router.yaml diff --git a/plugin-catalog/jev-skill-router.yaml b/plugin-catalog/jev-skill-router.yaml new file mode 100644 index 0000000000..e85a490a8e --- /dev/null +++ b/plugin-catalog/jev-skill-router.yaml @@ -0,0 +1,14 @@ +name: jev-skill-router +repo: https://github.com/ydmw74/jev-skill-router +sha: 5a83ac27a87ab9863514817b96b4850fe12de006 +description: "Agent Plugins v1 package: one skill plus one stdio MCP server (skill_select) started with uv run from the checkout. Needs uv on PATH. Suggests at most one skill per request via the TypeSafe (Jev) cookbook pattern; roster is read live from the profile skills dir." +maintainer: ydmw74 +tier: community +category: tools +docs_url: https://github.com/ydmw74/jev-skill-router#readme +platforms: [] +capabilities: + provides_tools: [] + provides_hooks: [] + provides_middleware: [] + requires_env: [JEV_TYPESAFE_API_KEY] From 6b5dca617ed2fb173f7d1b5711bec59e25a5517c Mon Sep 17 00:00:00 2001 From: Markus Werle Date: Wed, 23 Sep 2026 20:09:28 +0200 Subject: [PATCH 002/112] catalog: rename to jev-skill-router-mcp, bump sha (review) --- .../{jev-skill-router.yaml => jev-skill-router-mcp.yaml} | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) rename plugin-catalog/{jev-skill-router.yaml => jev-skill-router-mcp.yaml} (66%) diff --git a/plugin-catalog/jev-skill-router.yaml b/plugin-catalog/jev-skill-router-mcp.yaml similarity index 66% rename from plugin-catalog/jev-skill-router.yaml rename to plugin-catalog/jev-skill-router-mcp.yaml index e85a490a8e..3cd5f41d38 100644 --- a/plugin-catalog/jev-skill-router.yaml +++ b/plugin-catalog/jev-skill-router-mcp.yaml @@ -1,7 +1,7 @@ -name: jev-skill-router +name: jev-skill-router-mcp repo: https://github.com/ydmw74/jev-skill-router -sha: 5a83ac27a87ab9863514817b96b4850fe12de006 -description: "Agent Plugins v1 package: one skill plus one stdio MCP server (skill_select) started with uv run from the checkout. Needs uv on PATH. Suggests at most one skill per request via the TypeSafe (Jev) cookbook pattern; roster is read live from the profile skills dir." +sha: 76ee09e049dbc10c76f98cfefc6ca16b42cfad38 +description: "Agent Plugins v1 package: one skill plus one stdio MCP server (skill_select) started with uv run from the checkout. Needs uv on PATH. Suggests at most one skill per request via the TypeSafe (Jev) cookbook pattern; roster is read live from the profile skills dir. Note: the portable loader does not interpolate env in mcp.json — JEV_TYPESAFE_API_KEY must be present in the server's inherited environment (native mcp_servers declaration works too)." maintainer: ydmw74 tier: community category: tools From 921ab7a16310a373b19df5075fdb191d1a51a999 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:37:08 +0000 Subject: [PATCH 003/112] fix(agent): keep the session-start workspace snapshot across prompt rebuilds The workspace git snapshot is pinned per session so a rebuild (compaction, /compress) replays it instead of re-probing a repo that moved. Two gaps let the rebuild re-probe anyway and rewrite the system prompt mid-session: - The pin was keyed by resolve_context_cwd(), which is None when no cwd is bound (CLI launch dir) and the path once the TUI /compress binds the session cwd: the same directory under two keys. Key by the directory the probe actually inspects (resolve_context_cwd() or resolve_agent_cwd()). - An agent that did not build the session's prompt (resumed, a fresh gateway/TUI agent whose first act is /compress, a kill -9 restart) had no pin, so its first rebuild probed git now instead of replaying the session-start bytes. Seed the pin from the prompt the session already sends (cached copy, else its persisted row), only when that prompt names this cwd and its snapshot's Root covers it. reset_session_state still drops the pin, so /new, /resume and /branch re-snapshot at their own session start. --- agent/coding_context.py | 5 +- agent/conversation_loop.py | 18 +---- agent/surface_switch.py | 14 ++++ agent/system_prompt.py | 75 ++++++++++++++++++- tests/agent/test_compaction_prompt_rebuild.py | 66 ++++++++++++++++ 5 files changed, 157 insertions(+), 21 deletions(-) diff --git a/agent/coding_context.py b/agent/coding_context.py index 1cc962c1f6..181a114185 100644 --- a/agent/coding_context.py +++ b/agent/coding_context.py @@ -514,6 +514,9 @@ def project_facts_for(cwd: Optional[str | Path] = None) -> Optional[dict[str, An } +WORKSPACE_BLOCK_HEADER = "Workspace (snapshot at session start — re-check with `git` before acting on it):" + + def build_coding_workspace_block(cwd: Optional[str | Path] = None) -> str: """Workspace snapshot for the system prompt (empty outside a workspace): git state when in a repo, plus project facts — so marker-only (non-git) projects still get one.""" @@ -521,7 +524,7 @@ def build_coding_workspace_block(cwd: Optional[str | Path] = None) -> str: if root is None: return "" lines = [ - "Workspace (snapshot at session start — re-check with `git` before acting on it):", + WORKSPACE_BLOCK_HEADER, f"- Root: {root}", ] if git_root is not None: diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 57af87b558..d830d52558 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -31,7 +31,7 @@ from agent.prompt_caching import ( from agent.repetition_guard import REPETITION_LOOP_INTERRUPTED, is_runaway_repetition from agent.runtime_cwd import resolve_agent_cwd from agent.surface_switch import ( - identity_line_value, note_inert_pinned_tools, split_runtime_boundary, stage_surface_switch_note, + identity_line_value, note_inert_pinned_tools, runtime_host_value, stage_surface_switch_note, ) from agent.turn_context import PreflightCompressionTimedOut, build_turn_context from agent.turn_retry_state import TurnRetryState @@ -814,20 +814,6 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history) def _stored_prompt_matches_runtime(agent, prompt: str) -> bool: """Return False when the persisted runtime-identity lines are stale.""" - - _identity, runtime_marker, runtime = split_runtime_boundary(prompt) - - def host_info_value(label: str) -> str: - """New prompts delimit runtime hints; legacy prompts put them before context.""" - prefix = f"{label}:" - host_lines = (runtime.split("\n\n", 1)[0] if runtime_marker else prompt).splitlines() - for idx, line in enumerate(host_lines): - if line.startswith("User home directory:"): - for candidate in host_lines[idx + 1: idx + 4]: - if candidate.startswith(prefix): - return candidate[len(prefix):].strip() - return "" - # Model/provider identity, then cwd drift. A cwd change is a real content change (context # files, the workspace snapshot and the coding posture are all resolved from it), so it # still rebuilds; the runtime surface does not (agent/surface_switch.py). @@ -838,7 +824,7 @@ def _stored_prompt_matches_runtime(agent, prompt: str) -> bool: return False # Compare against resolve_agent_cwd() — the SAME resolver used to build the # prompt — so TERMINAL_CWD sessions are not falsely rejected. - stored_cwd = host_info_value("Current working directory") + stored_cwd = runtime_host_value(prompt, "Current working directory") if stored_cwd and stored_cwd != str(resolve_agent_cwd()): return False # Platform is deliberately NOT an identity field: a surface switch does not invalidate the diff --git a/agent/surface_switch.py b/agent/surface_switch.py index 5b642ecd0b..39e1f8f6bf 100644 --- a/agent/surface_switch.py +++ b/agent/surface_switch.py @@ -36,6 +36,20 @@ def split_runtime_boundary(prompt: str) -> tuple: return (identity, runtime_marker, runtime) if prompt.endswith(RUNTIME_ENVIRONMENT_END) else (prompt, "", "") +def runtime_host_value(prompt: str, label: str) -> str: + """``Label: value`` from the host lines of a persisted prompt (``Current working directory``); + new prompts delimit runtime hints, legacy prompts put them before context. "" when absent.""" + _identity, runtime_marker, runtime = split_runtime_boundary(prompt) + prefix = f"{label}:" + host_lines = (runtime.split("\n\n", 1)[0] if runtime_marker else prompt).splitlines() + for idx, line in enumerate(host_lines): + if line.startswith("User home directory:"): + for candidate in host_lines[idx + 1: idx + 4]: + if candidate.startswith(prefix): + return candidate[len(prefix):].strip() + return "" + + def identity_line_value(prompt: str, label: str) -> str: """Last ``Label: value`` line in the identity portion (the final runtime block is embedder prose, never identity). Last match wins — safe only for the volatile-tier trailer fields.""" diff --git a/agent/system_prompt.py b/agent/system_prompt.py index 77390ce01d..dd9a1e3173 100644 --- a/agent/system_prompt.py +++ b/agent/system_prompt.py @@ -26,7 +26,7 @@ from agent.prompt_builder import ( TOOL_USE_ENFORCEMENT_GUIDANCE, TOOL_USE_ENFORCEMENT_MODELS, drain_truncation_warnings, ) from agent import prompt_builder as _pb -from agent.runtime_cwd import resolve_context_cwd +from agent.runtime_cwd import resolve_agent_cwd, resolve_context_cwd from hermes_constants import get_default_hermes_root, get_hermes_home from utils import is_truthy_value @@ -592,6 +592,70 @@ def _alibaba_identity_part(agent: Any) -> List[str]: ] +def _workspace_pin_key() -> str: + """The directory the workspace probe inspects, which is also the prompt's ``Current working + directory``: a build with no cwd bound (launch dir) and a later one binding that same dir + (TUI ``/compress``) are one workspace, not two.""" + try: + return str(resolve_context_cwd() or resolve_agent_cwd()) + except OSError: # deleted cwd + return "" + + +def _persisted_workspace_block(prompt: str, key: str) -> Optional[str]: + """The workspace snapshot inside ``prompt`` taken for ``key`` (its ``- Root:`` is ``key`` or an + ancestor); "" when the prompt has none; None when it has one for another root.""" + from agent.coding_context import WORKSPACE_BLOCK_HEADER + head = f"\n\n{WORKSPACE_BLOCK_HEADER}\n- Root: " + start = prompt.find(head) + if start < 0: + return "" + cwd = Path(key).resolve() + while start >= 0: + block = prompt[start + 2:].split("\n\n", 1)[0] + root = Path(block.split("\n", 2)[1][len("- Root: "):]).resolve() + if root == cwd or root in cwd.parents: + return block + start = prompt.find(head, start + 2) + return None + + +def _session_prompt(agent: Any) -> Optional[str]: + """Prompt bytes this session already sends: the cached copy, else its persisted row.""" + cached = getattr(agent, "_cached_system_prompt", None) + if isinstance(cached, str) and cached: + return cached + db, session_id = getattr(agent, "_session_db", None), getattr(agent, "session_id", None) + if db is None or not isinstance(session_id, str) or not session_id: + return None + try: + row = db.get_session(session_id) + except Exception: + logger.debug("workspace snapshot: session row read failed (session=%s)", session_id, exc_info=True) + return None + prompt = row.get("system_prompt") if isinstance(row, dict) else None + return prompt if isinstance(prompt, str) and prompt else None + + +def _seed_workspace_pin(agent: Any, key: str) -> None: + """Pin the snapshot the session's existing prompt already carries. An agent that did not + build those bytes (resumed, or a fresh gateway/TUI agent whose first act is ``/compress``) + would otherwise re-probe git at its first rebuild and rewrite the prompt for any repo that + moved since session start. Only a snapshot provably taken in this cwd is adopted.""" + from agent.surface_switch import runtime_host_value + prompt = _session_prompt(agent) + if not prompt: + return + stored_cwd = runtime_host_value(prompt, "Current working directory") + if stored_cwd and stored_cwd != key: + return + block = _persisted_workspace_block(prompt, key) + # "" is only trustworthy when the prompt names this cwd: a legacy prompt without the + # runtime line might have been built with tools off. + if block or (block == "" and stored_cwd): + agent._frozen_workspace_snapshot = (key, block) + + def _coding_parts(agent: Any) -> Tuple[List[str], List[str], List[str]]: """``(prefix, workspace, trailing)`` coding-posture blocks; all empty without tools or when probing fails (it must never block prompt build). @@ -599,15 +663,18 @@ def _coding_parts(agent: Any) -> Tuple[List[str], List[str], List[str]]: The workspace block is a live git probe after project context, ahead of the whole volatile band; re-probing at the compaction rebuild re-emits different bytes for any repo that moved and defeats the keep-prompt fast path. So the bytes are pinned per - session on the agent, keyed by the resolved cwd (a gateway serves many cwds), and - replayed on rebuilds; ``reset_session_state`` drops the pin at a session boundary. + session on the agent, keyed by the probed cwd (a gateway serves many cwds), seeded from + the session's existing prompt when this agent did not build it, and replayed on + rebuilds; ``reset_session_state`` drops the pin at a session boundary. """ try: from agent.coding_context import coding_system_prompt_parts if not agent.valid_tool_names: return [], [], [] cwd = resolve_context_cwd() - cwd_key = str(cwd) if cwd is not None else "" + cwd_key = _workspace_pin_key() + if getattr(agent, "_frozen_workspace_snapshot", None) is None: + _seed_workspace_pin(agent, cwd_key) pinned = getattr(agent, "_frozen_workspace_snapshot", None) # "" is a real pinned value (no workspace here) — only a cwd mismatch re-probes. replay = pinned[1] if pinned is not None and pinned[0] == cwd_key else None diff --git a/tests/agent/test_compaction_prompt_rebuild.py b/tests/agent/test_compaction_prompt_rebuild.py index a3325c69e0..a9f40ab309 100644 --- a/tests/agent/test_compaction_prompt_rebuild.py +++ b/tests/agent/test_compaction_prompt_rebuild.py @@ -217,6 +217,72 @@ class TestWorkspaceSnapshotPinnedAcrossCompaction(unittest.TestCase): finally: shutil.rmtree(tmp, ignore_errors=True) + def _pin_agent(self, **over): + return _agent( + load_soul_identity=False, skip_context_files=True, valid_tool_names={"terminal"}, + platform="cli", model="gpt-4o", _task_completion_guidance=False, + _parallel_tool_call_guidance=False, _tool_use_enforcement=False, _execution_guidance=False, + _environment_probe=False, _bot_mode_protocol=False, _kanban_worker_guidance="", + pass_session_id=False, session_id="s1", _emit_status=lambda *a, **k: None, **over, + ) + + def test_binding_the_launch_dir_explicitly_replays_the_pin(self): + """CLI-shaped first build (no cwd bound -> launch dir), then TUI /compress binds that same dir: + one workspace, so the rebuild replays the session-start snapshot.""" + import os, tempfile, shutil + from pathlib import Path + from agent.system_prompt import build_system_prompt, invalidate_system_prompt + + tmp = Path(tempfile.mkdtemp(prefix="test-pinned-bind-")) + old_cwd = os.getcwd() + try: + repo = _init_repo(tmp / "proj", "init commit") + os.chdir(repo) + agent = self._pin_agent() + with patch("agent.prompt_builder.load_soul_md", return_value=""), \ + patch("agent.prompt_builder.build_environment_hints", return_value="ENV HINTS"): + with patch("agent.system_prompt.resolve_context_cwd", return_value=None): + p1 = build_system_prompt(agent) + self.assertIn("Status: clean", p1) + (repo / "untracked.txt").write_text("wip\n") + invalidate_system_prompt(agent) + with patch("agent.system_prompt.resolve_context_cwd", return_value=repo): + self.assertEqual(build_system_prompt(agent), p1) + finally: + os.chdir(old_cwd) + shutil.rmtree(tmp, ignore_errors=True) + + def test_agent_that_did_not_build_the_prompt_replays_the_persisted_snapshot(self): + """Resume / gateway / TUI shape: a fresh agent rebuilds (compaction, a first /compress) after + the repo moved and replays the snapshot its session row already holds — unless that prompt + was taken in another cwd.""" + import tempfile, shutil + from pathlib import Path + from agent.system_prompt import build_system_prompt + + tmp = Path(tempfile.mkdtemp(prefix="test-pinned-resume-")) + try: + repo, other = _init_repo(tmp / "proj", "init commit"), _init_repo(tmp / "other", "init other") + + def env(cwd): + return patch("agent.prompt_builder.build_environment_hints", + return_value=f"Host: x\nUser home directory: /h\nCurrent working directory: {cwd}") + + with patch("agent.prompt_builder.load_soul_md", return_value=""), env(repo), \ + patch("agent.system_prompt.resolve_context_cwd", return_value=repo): + stored = build_system_prompt(self._pin_agent()) + self.assertIn("Status: clean", stored) + (repo / "untracked.txt").write_text("wip\n") + db = SimpleNamespace(get_session=lambda sid: {"system_prompt": stored}) + resumed = self._pin_agent(_cached_system_prompt=None, _session_db=db) + self.assertEqual(build_system_prompt(resumed), stored) + with patch("agent.prompt_builder.load_soul_md", return_value=""), env(other), \ + patch("agent.system_prompt.resolve_context_cwd", return_value=other): + moved = self._pin_agent(_cached_system_prompt=None, _session_db=db) + self.assertIn(f"- Root: {other}", build_system_prompt(moved)) + finally: + shutil.rmtree(tmp, ignore_errors=True) + if __name__ == "__main__": unittest.main() From 2ec129528ddc7c47d8f9e715b6bc771f0fcae48b Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 11:26:12 -0700 Subject: [PATCH 004/112] fix(sessions): a /branch child row carries the parent's system prompt (CLI + TUI/Desktop) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The branch copies the parent transcript byte-for-byte so its first turn can hit the warm prefix cache, but the child row was created without a system prompt. The child's first build then found nothing to restore and re-probed the workspace, rewriting the prompt at byte 0 for any repo that moved since the parent's session start — the same rewrite this PR removes for /compress and resume, on the one session transition it did not reach. It also logged the "stored system prompt is null; investigate update_system_prompt" WARNING for every branch. Both branch writers now pass the parent's prompt into create_session: the CLI prefers the running agent's cached bytes (what this process sends) and falls back to the parent row; _persist_branch (TUI/Desktop, seeded and lazy) reads the parent row. With the row in place the seeding in agent/system_prompt.py replays the snapshot on the branch's later rebuilds. Tests: one per surface, red on the PR head (child["system_prompt"] is None), green here. --- hermes_cli/cli_commands_mixin.py | 9 ++++- tests/hermes_cli/test_branch_command.py | 18 ++++++++++ .../tui_gateway/test_branch_system_prompt.py | 35 +++++++++++++++++++ tui_gateway/methods_session.py | 10 +++++- 4 files changed, 70 insertions(+), 2 deletions(-) create mode 100644 tests/tui_gateway/test_branch_system_prompt.py diff --git a/hermes_cli/cli_commands_mixin.py b/hermes_cli/cli_commands_mixin.py index e165a56a45..469ce0eabf 100644 --- a/hermes_cli/cli_commands_mixin.py +++ b/hermes_cli/cli_commands_mixin.py @@ -1410,10 +1410,17 @@ class CLICommandsMixin: # user is still on open, not ended with end_reason="branched" and no branch (#11030). # The stable ``_branched_from`` marker keeps the branch visible in /resume + /sessions # even after the parent is re-ended with a different end_reason. + # The child sends the parent's exact system prompt: a row without one makes the branch's first + # turn rebuild (re-probing the workspace), so the warm cache the copied transcript buys is + # lost at byte 0 whenever the repo moved since the parent's session start. + parent_prompt = getattr(self.agent, "_cached_system_prompt", None) + if not isinstance(parent_prompt, str) or not parent_prompt: + with suppress(Exception): + parent_prompt = (self._session_db.get_session(parent_session_id) or {}).get("system_prompt") try: self._session_db.create_session( session_id=new_session_id, source=os.environ.get("HERMES_SESSION_SOURCE", "cli"), - model=self.model, parent_session_id=parent_session_id, + model=self.model, parent_session_id=parent_session_id, system_prompt=parent_prompt or None, model_config={"max_iterations": self.max_turns, "reasoning_config": self.reasoning_config, "_branched_from": parent_session_id}) except Exception as e: diff --git a/tests/hermes_cli/test_branch_command.py b/tests/hermes_cli/test_branch_command.py index 2c1ac73a9f..9b05fe8754 100644 --- a/tests/hermes_cli/test_branch_command.py +++ b/tests/hermes_cli/test_branch_command.py @@ -165,6 +165,24 @@ class TestBranchFlushesBeforeEndSession: conversation_history=cli_instance.conversation_history, ) + def test_branch_child_row_carries_the_parent_system_prompt(self, cli_instance, session_db): + """The branch's first turn must send the bytes the parent already sends: with no stored prompt the + child rebuilds (a fresh workspace probe) and the copied transcript's warm cache is lost at byte 0.""" + from cli import HermesCLI + + parent_id = cli_instance.session_id + session_db.update_system_prompt(parent_id, "PARENT PROMPT\n\nWorkspace snapshot: session start") + # No live agent: the parent's persisted row is the source. + HermesCLI._handle_branch_command(cli_instance, "/branch") + child = session_db.get_session(cli_instance.session_id) + assert child["parent_session_id"] == parent_id + assert child["system_prompt"] == "PARENT PROMPT\n\nWorkspace snapshot: session start" + + # A live agent's cached prompt (what THIS process sends) wins over the row. + cli_instance.agent = MagicMock(_cached_system_prompt="LIVE PROMPT") + HermesCLI._handle_branch_command(cli_instance, "/branch") + assert session_db.get_session(cli_instance.session_id)["system_prompt"] == "LIVE PROMPT" + REASONING_DETAILS = [ {"type": "reasoning.text", "text": "sort in place instead", "format": "unknown"} diff --git a/tests/tui_gateway/test_branch_system_prompt.py b/tests/tui_gateway/test_branch_system_prompt.py new file mode 100644 index 0000000000..117ac35044 --- /dev/null +++ b/tests/tui_gateway/test_branch_system_prompt.py @@ -0,0 +1,35 @@ +"""A TUI/Desktop branch child sends the bytes its parent already sends. + +Without a stored prompt the child's first turn rebuilds (a fresh workspace probe), so the warm +cache the copied transcript buys is lost at byte 0 whenever the repo moved since session start. +""" + +from __future__ import annotations + +from pathlib import Path + + +def test_persist_branch_copies_the_parent_system_prompt(tmp_path, monkeypatch): + from hermes_state import SessionDB + from tui_gateway import server + + home = tmp_path / ".hermes" + home.mkdir() + monkeypatch.setattr(Path, "home", lambda: tmp_path) + monkeypatch.setenv("HERMES_HOME", str(home)) + + with SessionDB(home / "state.db") as db: + db.create_session("parent", source="desktop", model="test-model") + db.update_system_prompt("parent", "PARENT PROMPT\n\nWorkspace snapshot: session start") + server._persist_branch(db, "child", "parent", "Branch", [{"role": "user", "content": "hello"}], + source="desktop", cwd=str(tmp_path), profile_name="default", + model="test-model") + # A parent with no stored prompt yields a child with none — never a phantom empty string. + db.create_session("bare", source="desktop", model="test-model") + server._persist_branch(db, "bare-child", "bare", "Branch 2", [{"role": "user", "content": "hi"}], + source="desktop", cwd=str(tmp_path), profile_name="default", + model="test-model") + + with SessionDB(home / "state.db") as db: + assert db.get_session("child")["system_prompt"] == "PARENT PROMPT\n\nWorkspace snapshot: session start" + assert db.get_session("bare-child")["system_prompt"] is None diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index 7c77411209..05d03d305c 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -231,8 +231,16 @@ def _persist_branch(db, new_key: str, parent_key: str, title: str, history: list deletes a committed row whose transcript/title failed (a durable-but-empty row would defeat the INSERT OR IGNORE first-prompt seed) — except on disk-full, where the delete cannot land. ``user_id`` is the creating login: the child is a Desktop session too, and the row only records identity at insert.""" + # The child sends the parent's exact system prompt: a row without one makes the branch's first + # turn rebuild (re-probing the workspace) and forfeits the warm cache the copied transcript buys. + parent_prompt = None + try: + parent_prompt = (db.get_session(parent_key) or {}).get("system_prompt") + except Exception: + logger.debug("branch: parent system prompt read failed for %s", parent_key, exc_info=True) db.create_session(new_key, source=source, model=model, model_config={"_branched_from": parent_key}, - parent_session_id=parent_key, cwd=cwd, profile_name=profile_name, user_id=user_id) + parent_session_id=parent_key, cwd=cwd, profile_name=profile_name, user_id=user_id, + system_prompt=parent_prompt or None) try: # Compensation guard (#93959 review): if the transcript copy or title write fails AFTER the row # committed, the durable-but-empty row would defeat the lazy first-prompt fallback From 027809f88eaea614cab44113f0293072293c1887 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:48:18 -0700 Subject: [PATCH 005/112] fix(agent): don't restore a foreign Session ID or pin an empty workspace snapshot - _stored_prompt_matches_runtime: with the Session ID trailer on (--pass-session-id / HERMES_TUI_PASS_SESSION_ID), a stored prompt whose Session ID line is not this session's is a mismatch. A /branch child now copies its parent's prompt bytes, and without this it told the model the parent's id. Gated on the flag so a "Session ID:" line in project text cannot force a rebuild every turn when the trailer is off. - _seed_workspace_pin: adopt only a real workspace block. A stored prompt that names this cwd but carries no block (built on a messaging surface, resumed on the CLI in the same repo) pinned "no workspace" and dropped the git snapshot for the rest of the session; now the next build captures one. --- agent/conversation_loop.py | 6 ++++++ agent/system_prompt.py | 6 +++--- tests/agent/test_compaction_prompt_rebuild.py | 21 +++++++++++++++++++ tests/agent/test_system_prompt.py | 13 ++++++++++++ 4 files changed, 43 insertions(+), 3 deletions(-) diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index d830d52558..fb79d6242f 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -822,6 +822,12 @@ def _stored_prompt_matches_runtime(agent, prompt: str) -> bool: current = str(getattr(agent, attr, "") or "").strip() if stored and current and stored != current: return False + # A prompt stamped for another session (a /branch child copies its parent's bytes) must not + # tell the model a foreign Session ID. Checked only when the trailer is on: with it off, a + # "Session ID:" line in project text would read as a mismatch and rebuild every turn. + stored_sid = identity_line_value(prompt, "Session ID") + if stored_sid and getattr(agent, "pass_session_id", False) and stored_sid != agent.session_id: + return False # Compare against resolve_agent_cwd() — the SAME resolver used to build the # prompt — so TERMINAL_CWD sessions are not falsely rejected. stored_cwd = runtime_host_value(prompt, "Current working directory") diff --git a/agent/system_prompt.py b/agent/system_prompt.py index dd9a1e3173..853462571c 100644 --- a/agent/system_prompt.py +++ b/agent/system_prompt.py @@ -650,9 +650,9 @@ def _seed_workspace_pin(agent: Any, key: str) -> None: if stored_cwd and stored_cwd != key: return block = _persisted_workspace_block(prompt, key) - # "" is only trustworthy when the prompt names this cwd: a legacy prompt without the - # runtime line might have been built with tools off. - if block or (block == "" and stored_cwd): + # Only a real snapshot is adopted: a prompt without one (built on a surface without the + # coding posture, or with tools off) leaves the pin open so this build captures one. + if block: agent._frozen_workspace_snapshot = (key, block) diff --git a/tests/agent/test_compaction_prompt_rebuild.py b/tests/agent/test_compaction_prompt_rebuild.py index a9f40ab309..c6f4545d5e 100644 --- a/tests/agent/test_compaction_prompt_rebuild.py +++ b/tests/agent/test_compaction_prompt_rebuild.py @@ -283,6 +283,27 @@ class TestWorkspaceSnapshotPinnedAcrossCompaction(unittest.TestCase): finally: shutil.rmtree(tmp, ignore_errors=True) + def test_persisted_prompt_without_a_snapshot_does_not_pin_an_empty_one(self): + """A session row built where no workspace block was emitted (a messaging surface) and + resumed in the same repo must capture a real snapshot, not pin "no workspace" for good.""" + import tempfile, shutil + from pathlib import Path + from agent.system_prompt import build_system_prompt + + tmp = Path(tempfile.mkdtemp(prefix="test-pinned-empty-")) + try: + repo = _init_repo(tmp / "proj", "init commit") + stored = f"Host: x\nUser home directory: /h\nCurrent working directory: {repo}\n\nBODY" + db = SimpleNamespace(get_session=lambda sid: {"system_prompt": stored}) + with patch("agent.prompt_builder.load_soul_md", return_value=""), \ + patch("agent.prompt_builder.build_environment_hints", + return_value=f"Host: x\nUser home directory: /h\nCurrent working directory: {repo}"), \ + patch("agent.system_prompt.resolve_context_cwd", return_value=repo): + resumed = self._pin_agent(_cached_system_prompt=None, _session_db=db) + self.assertIn(f"- Root: {repo}", build_system_prompt(resumed)) + finally: + shutil.rmtree(tmp, ignore_errors=True) + if __name__ == "__main__": unittest.main() diff --git a/tests/agent/test_system_prompt.py b/tests/agent/test_system_prompt.py index b14f0384b5..f9d680985c 100644 --- a/tests/agent/test_system_prompt.py +++ b/tests/agent/test_system_prompt.py @@ -280,6 +280,19 @@ def test_stored_prompt_cwd_ignores_project_host_decoys(monkeypatch, tmp_path): assert _stored_prompt_matches_runtime(agent, legacy) +def test_stored_prompt_stamped_for_another_session_is_not_restored(monkeypatch, tmp_path): + """With the Session ID trailer on, a prompt persisted for another session (a /branch child + copies its parent's bytes) must rebuild instead of telling the model the parent's id.""" + from agent.conversation_loop import _stored_prompt_matches_runtime + + monkeypatch.setenv("TERMINAL_ENV", "local") + monkeypatch.setenv("TERMINAL_CWD", str(tmp_path)) + fields = dict(platform="cli", model="test-model", provider="test-provider", pass_session_id=True) + parent_prompt = build_system_prompt(_make_agent(session_id="parent-sid", **fields)) + assert _stored_prompt_matches_runtime(_make_agent(session_id="parent-sid", **fields), parent_prompt) + assert not _stored_prompt_matches_runtime(_make_agent(session_id="child-sid", **fields), parent_prompt) + + class TestExecutionGuidanceInjection: """Injection gate for OPENAI_MODEL_EXECUTION_GUIDANCE via ``agent.execution_guidance`` (auto/true/false/list). From 60f2bcf4abb8a1027d60637ec739240f9e23e579 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E5=8D=83=E4=B9=98=E5=A6=8D=20=28Xiaoyaner=29?= Date: Thu, 30 Jul 2026 07:14:04 +0800 Subject: [PATCH 006/112] fix(model-switch): scope direct-alias base_url to the target provider An explicit `/model --provider X` switch could report target_provider=X while base_url still pointed at the previous provider's endpoint: resolve_alias finds a direct alias by reverse model-ID lookup, and the base_url override applied that alias unconditionally even when the alias belonged to a different provider. Requests would then be sent to the old provider's URL under the new provider's identity. Two narrow changes: - resolve_alias's reverse lookup now prefers an alias whose provider matches current_provider, falling back to first-match only when none does. Insertion order is not a routing decision. - The direct-alias base_url override now ignores (and stops reporting) an alias whose provider differs from target_provider. Exact alias-name lookup stays provider-agnostic, and the first-match fallback is preserved, both pinned by tests. [salvaged onto the refactored switch pipeline: the ownership guard now sits in _route_explicit_provider, where resolved_alias is settled, so both the runtime resolve (explicit_base_url) and the direct-alias override see the filtered alias] --- hermes_cli/model_switch.py | 20 +++++++++++++++++--- 1 file changed, 17 insertions(+), 3 deletions(-) diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py index 0074eaf496..404d61c4e2 100644 --- a/hermes_cli/model_switch.py +++ b/hermes_cli/model_switch.py @@ -752,10 +752,19 @@ def resolve_alias(raw_input: str, current_provider: str) -> Optional[tuple[str, return (direct.provider, direct.model, key) # Reverse lookup so full names ("kimi-k2.5") route through direct aliases instead of - # falling through to the catalog/OpenRouter. + # falling through to the catalog/OpenRouter. Several aliases may expose one model id on + # different providers: prefer the one served by current_provider, since insertion order is + # not a routing decision and the wrong alias hands back another provider's base_url. + reverse_fallback: Optional[tuple[str, str, str]] = None for alias_name, da in DIRECT_ALIASES.items(): - if da.model.lower() == key: + if da.model.lower() != key: + continue + if da.provider == current_provider: return (da.provider, da.model, alias_name) + if reverse_fallback is None: + reverse_fallback = (da.provider, da.model, alias_name) + if reverse_fallback is not None: + return reverse_fallback process_catalog, process_aliases = _external_process_catalog(current_provider) if process_catalog: @@ -1232,7 +1241,12 @@ def _route_explicit_provider(st: _Switch) -> Optional[ModelSwitchResult]: except AmbiguousAliasError as err: return st.fail(_ambiguous_alias_message(err), target_provider=st.target_provider) if alias_result is not None: - _, st.new_model, st.resolved_alias = alias_result + alias_provider, st.new_model, alias_name = alias_result + # Adopt the alias (and with it its base_url and key) only when it belongs to the provider + # the user named: a reverse model-id match may land on another provider's alias, and + # honouring it would send the turn to that provider's endpoint under this one's identity. + if alias_provider == st.target_provider: + st.resolved_alias = alias_name return None From 1ee38c80901f7648f84f363c7a402fb77f15f654 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E5=8D=83=E4=B9=98=E5=A6=8D=20=28Xiaoyaner=29?= Date: Wed, 23 Sep 2026 06:58:09 -0700 Subject: [PATCH 007/112] fix(model): normalize direct-alias provider ownership Normalize both provider labels at direct-alias ownership comparisons so alias spellings and case variants match canonical target IDs without weakening cross-provider rejection. Maintainer finding: https://github.com/NousResearch/hermes-agent/pull/75262#discussion_r3688616856 --- hermes_cli/model_switch.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py index 404d61c4e2..f6086f03a9 100644 --- a/hermes_cli/model_switch.py +++ b/hermes_cli/model_switch.py @@ -15,7 +15,7 @@ from typing import Any, NamedTuple, Optional from hermes_cli.providers import ( LLAMACPP_ALIASES, ProviderDef, custom_provider_aliases, determine_api_mode, get_label, - host_mandated_api_mode, is_aggregator, resolve_provider_full) + host_mandated_api_mode, is_aggregator, normalize_provider, resolve_provider_full) from hermes_cli.model_normalize import normalize_model_for_provider from agent.models_dev import ( ModelCapabilities, ModelInfo, get_model_capabilities, get_model_info, list_provider_models) @@ -759,7 +759,7 @@ def resolve_alias(raw_input: str, current_provider: str) -> Optional[tuple[str, for alias_name, da in DIRECT_ALIASES.items(): if da.model.lower() != key: continue - if da.provider == current_provider: + if normalize_provider(da.provider or "") == normalize_provider(current_provider or ""): return (da.provider, da.model, alias_name) if reverse_fallback is None: reverse_fallback = (da.provider, da.model, alias_name) @@ -1245,7 +1245,7 @@ def _route_explicit_provider(st: _Switch) -> Optional[ModelSwitchResult]: # Adopt the alias (and with it its base_url and key) only when it belongs to the provider # the user named: a reverse model-id match may land on another provider's alias, and # honouring it would send the turn to that provider's endpoint under this one's identity. - if alias_provider == st.target_provider: + if normalize_provider(alias_provider or "") == normalize_provider(st.target_provider or ""): st.resolved_alias = alias_name return None From af80401f4e4555a6e6fb6ae209a2ba0cd904a9bf Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 07:00:02 -0700 Subject: [PATCH 008/112] test(model-switch): pin --provider ownership of direct aliases Two invariants against a real config.yaml: an alias bound to another provider's endpoint never outranks an explicit --provider (host and key stay on the named provider), and when several aliases share a model id the one owned by the named provider wins regardless of mapping order. Found by the C11 routing truth table E2E (lands separately with the E2E suites). --- tests/hermes_cli/test_ollama_cloud_auth.py | 56 ++++++++++++++++++++++ 1 file changed, 56 insertions(+) diff --git a/tests/hermes_cli/test_ollama_cloud_auth.py b/tests/hermes_cli/test_ollama_cloud_auth.py index 56733adc05..858714cda4 100644 --- a/tests/hermes_cli/test_ollama_cloud_auth.py +++ b/tests/hermes_cli/test_ollama_cloud_auth.py @@ -284,3 +284,59 @@ class TestSwitchModelDirectAliasOverride: assert result.success assert result.api_key == "no-key-required" assert result.base_url == "http://localhost:11434/v1" + + @staticmethod + def _explicit_switch_to_provider_b(monkeypatch, aliases): + """``/model shared-model --provider provider-b`` against a real config.yaml, with the + given direct aliases loaded. Only model validation is stubbed (no network).""" + import os + from pathlib import Path + + import hermes_cli.model_switch as ms + from hermes_cli.config import load_config + + monkeypatch.setenv("PROVIDER_B_KEY", "sk-provider-b") + (Path(os.environ["HERMES_HOME"]) / "config.yaml").write_text( + "model:\n provider: provider-a\n default: old-model\n" + "providers:\n" + " provider-a:\n base_url: https://api-a.example.com/v1\n" + " provider-b:\n base_url: https://api-b.example.com/v1\n key_env: PROVIDER_B_KEY\n") + monkeypatch.setattr(ms, "DIRECT_ALIASES", aliases) + monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model", + lambda *a, **kw: {"accepted": True, "persist": True, "recognized": True, "message": None}) + return ms.switch_model( + "shared-model", "provider-a", "old-model", + current_base_url="https://api-a.example.com/v1", current_api_key="sk-provider-a", + explicit_provider="provider-b", user_providers=load_config()["providers"]) + + def test_explicit_provider_never_adopts_alias_bound_to_another_provider(self, monkeypatch): + """An alias on another provider's endpoint that targets the same model id must not + outrank --provider: the turn and the credential stay on the provider the user named.""" + from hermes_cli.model_switch import DirectAlias + + result = self._explicit_switch_to_provider_b(monkeypatch, { + "a-alias": DirectAlias("shared-model", "custom", "https://alias-host.example.com/v1", + api_key="sk-alias-host"), + }) + + assert result.success, result.error_message + assert result.target_provider == "provider-b" + assert result.base_url == "https://api-b.example.com/v1" + assert result.api_key == "sk-provider-b" + assert result.resolved_via_alias == "" + + def test_explicit_provider_prefers_its_own_alias_for_a_shared_model(self, monkeypatch): + """Several aliases expose one model id: the one owned by the named provider wins, + whatever the mapping order (provider spelling is normalized).""" + from hermes_cli.model_switch import DirectAlias + + result = self._explicit_switch_to_provider_b(monkeypatch, { + "a-alias": DirectAlias("shared-model", "custom", "https://alias-host.example.com/v1", + api_key="sk-alias-host"), + "b-alias": DirectAlias("shared-model", "Provider-B", "https://api-b.example.com/v2"), + }) + + assert result.success, result.error_message + assert result.resolved_via_alias == "b-alias" + assert result.base_url == "https://api-b.example.com/v2" + assert result.api_key != "sk-alias-host" From ed3e7203ee858bf6e8195bbf447fc69f005ceeb3 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 07:07:31 -0700 Subject: [PATCH 009/112] chore: map xiaoyaner0201 contributor email --- contributors/emails/xiaoyaner0201@users.noreply.github.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/xiaoyaner0201@users.noreply.github.com diff --git a/contributors/emails/xiaoyaner0201@users.noreply.github.com b/contributors/emails/xiaoyaner0201@users.noreply.github.com new file mode 100644 index 0000000000..729e4f5007 --- /dev/null +++ b/contributors/emails/xiaoyaner0201@users.noreply.github.com @@ -0,0 +1 @@ +xiaoyaner0201 From 79d012bd25945a65f51022f8a1a9ee04a294bd65 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:26:51 +0000 Subject: [PATCH 010/112] fix(model): match alias ownership on the resolved provider id The --provider ownership check compared normalize_provider() spellings, but a legacy custom_providers entry resolves to id "custom:". An alias with "provider: corp-llm" under "--provider corp-llm" was therefore dropped and the switch fell back to the provider's own base_url/key instead of the alias's. Compare resolve_provider_full(...).id on both sides (normalize_provider when unresolvable), in the ownership check and in resolve_alias's reverse-lookup preference loop, which now receives the user/custom provider config. --- hermes_cli/model_switch.py | 19 +++++++++++--- tests/hermes_cli/test_ollama_cloud_auth.py | 29 +++++++++++++++++++--- 2 files changed, 41 insertions(+), 7 deletions(-) diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py index f6086f03a9..b4652f8fa6 100644 --- a/hermes_cli/model_switch.py +++ b/hermes_cli/model_switch.py @@ -737,7 +737,16 @@ def _ambiguous_alias_message(err: "AmbiguousAliasError") -> str: f"Pick one with /model .") -def resolve_alias(raw_input: str, current_provider: str) -> Optional[tuple[str, str, str]]: +def _provider_identity(name: str, user_providers: Optional[dict] = None, + custom_providers: Optional[list] = None) -> str: + """Id a provider name routes to, e.g. ``custom:`` for a legacy ``custom_providers`` + entry, so an alias naming it by its bare name compares equal to the resolved provider.""" + pdef = resolve_provider_full(name, user_providers, custom_providers) if name else None + return pdef.id if pdef is not None else normalize_provider(name or "") + + +def resolve_alias(raw_input: str, current_provider: str, user_providers: Optional[dict] = None, + custom_providers: Optional[list] = None) -> Optional[tuple[str, str, str]]: """Resolve a short alias against the current provider's catalog. Direct aliases (and reverse lookup by exact model id) win; then :data:`MODEL_ALIASES` is @@ -756,10 +765,11 @@ def resolve_alias(raw_input: str, current_provider: str) -> Optional[tuple[str, # different providers: prefer the one served by current_provider, since insertion order is # not a routing decision and the wrong alias hands back another provider's base_url. reverse_fallback: Optional[tuple[str, str, str]] = None + current_id = _provider_identity(current_provider, user_providers, custom_providers) for alias_name, da in DIRECT_ALIASES.items(): if da.model.lower() != key: continue - if normalize_provider(da.provider or "") == normalize_provider(current_provider or ""): + if _provider_identity(da.provider, user_providers, custom_providers) == current_id: return (da.provider, da.model, alias_name) if reverse_fallback is None: reverse_fallback = (da.provider, da.model, alias_name) @@ -1237,7 +1247,7 @@ def _route_explicit_provider(st: _Switch) -> Optional[ModelSwitchResult]: f"Specify the model explicitly: /model --provider {st.explicit_provider}") try: - alias_result = resolve_alias(st.new_model, st.target_provider) + alias_result = resolve_alias(st.new_model, st.target_provider, st.user_providers, st.custom_providers) except AmbiguousAliasError as err: return st.fail(_ambiguous_alias_message(err), target_provider=st.target_provider) if alias_result is not None: @@ -1245,7 +1255,8 @@ def _route_explicit_provider(st: _Switch) -> Optional[ModelSwitchResult]: # Adopt the alias (and with it its base_url and key) only when it belongs to the provider # the user named: a reverse model-id match may land on another provider's alias, and # honouring it would send the turn to that provider's endpoint under this one's identity. - if normalize_provider(alias_provider or "") == normalize_provider(st.target_provider or ""): + if (_provider_identity(alias_provider, st.user_providers, st.custom_providers) + == _provider_identity(st.target_provider, st.user_providers, st.custom_providers)): st.resolved_alias = alias_name return None diff --git a/tests/hermes_cli/test_ollama_cloud_auth.py b/tests/hermes_cli/test_ollama_cloud_auth.py index 858714cda4..fe1d3416cc 100644 --- a/tests/hermes_cli/test_ollama_cloud_auth.py +++ b/tests/hermes_cli/test_ollama_cloud_auth.py @@ -286,7 +286,7 @@ class TestSwitchModelDirectAliasOverride: assert result.base_url == "http://localhost:11434/v1" @staticmethod - def _explicit_switch_to_provider_b(monkeypatch, aliases): + def _explicit_switch_to_provider_b(monkeypatch, aliases, explicit="provider-b", extra_cfg=""): """``/model shared-model --provider provider-b`` against a real config.yaml, with the given direct aliases loaded. Only model validation is stubbed (no network).""" import os @@ -300,14 +300,17 @@ class TestSwitchModelDirectAliasOverride: "model:\n provider: provider-a\n default: old-model\n" "providers:\n" " provider-a:\n base_url: https://api-a.example.com/v1\n" - " provider-b:\n base_url: https://api-b.example.com/v1\n key_env: PROVIDER_B_KEY\n") + " provider-b:\n base_url: https://api-b.example.com/v1\n key_env: PROVIDER_B_KEY\n" + + extra_cfg) monkeypatch.setattr(ms, "DIRECT_ALIASES", aliases) monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model", lambda *a, **kw: {"accepted": True, "persist": True, "recognized": True, "message": None}) + cfg = load_config() return ms.switch_model( "shared-model", "provider-a", "old-model", current_base_url="https://api-a.example.com/v1", current_api_key="sk-provider-a", - explicit_provider="provider-b", user_providers=load_config()["providers"]) + explicit_provider=explicit, user_providers=cfg["providers"], + custom_providers=cfg.get("custom_providers")) def test_explicit_provider_never_adopts_alias_bound_to_another_provider(self, monkeypatch): """An alias on another provider's endpoint that targets the same model id must not @@ -340,3 +343,23 @@ class TestSwitchModelDirectAliasOverride: assert result.resolved_via_alias == "b-alias" assert result.base_url == "https://api-b.example.com/v2" assert result.api_key != "sk-alias-host" + + def test_explicit_provider_keeps_alias_owned_by_legacy_custom_provider(self, monkeypatch): + """A legacy ``custom_providers`` entry resolves to ``custom:``; an alias that names + it by its bare name is still that provider's alias and keeps its own endpoint and key.""" + from hermes_cli.model_switch import DirectAlias + + result = self._explicit_switch_to_provider_b(monkeypatch, { + "a-alias": DirectAlias("shared-model", "provider-a", "https://alias-host.example.com/v1", + api_key="sk-alias-host"), + "corp-alias": DirectAlias("shared-model", "corp-llm", "https://corp.example.com/v2", + api_key="sk-corp-alias"), + }, explicit="corp-llm", extra_cfg=( + "custom_providers:\n - name: corp-llm\n" + " base_url: https://corp.example.com/v1\n api_key: sk-corp\n")) + + assert result.success, result.error_message + assert result.target_provider == "custom:corp-llm" + assert result.resolved_via_alias == "corp-alias" + assert result.base_url == "https://corp.example.com/v2" + assert result.api_key == "sk-corp-alias" From ffc44bbaf3a5be8fa72038320fa322a7fbe2400d Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:31:34 -0700 Subject: [PATCH 011/112] fix(model): prefer the current provider's alias on implicit switches too Only the --provider path handed user_providers/custom_providers to resolve_alias, so the provider-identity comparison behind the current-provider preference could not see legacy custom_providers entries on the implicit path (/model ) or the authenticated-provider fallback. On custom:corp-llm, /model shared-model picked another provider's alias for the same model id and switched to its base_url. Pass both provider maps on every resolve_alias call site. The two existing resolve_alias stubs in test_ollama_cloud_auth.py accept the extra positional args; their assertions are unchanged. --- hermes_cli/model_switch.py | 10 +++--- tests/hermes_cli/test_ollama_cloud_auth.py | 39 ++++++++++++++++++++-- 2 files changed, 43 insertions(+), 6 deletions(-) diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py index b4652f8fa6..6f35723962 100644 --- a/hermes_cli/model_switch.py +++ b/hermes_cli/model_switch.py @@ -855,12 +855,14 @@ def get_authenticated_provider_slugs( def _resolve_alias_fallback( - raw_input: str, authenticated_providers: list[str] = ()) -> Optional[tuple[str, str, str]]: + raw_input: str, authenticated_providers: list[str] = (), user_providers: Optional[dict] = None, + custom_providers: Optional[list] = None) -> Optional[tuple[str, str, str]]: """Resolve an alias on the user's authenticated providers (``("openrouter", "nous")`` when none given). AmbiguousAliasError propagates: the alias exists on this provider, the user just has to choose — trying the next provider would silently switch them somewhere they didn't ask for.""" - results = (resolve_alias(raw_input, p) for p in authenticated_providers or ("openrouter", "nous")) + results = (resolve_alias(raw_input, p, user_providers, custom_providers) + for p in authenticated_providers or ("openrouter", "nous")) return next((r for r in results if r is not None), None) @@ -1267,7 +1269,7 @@ def _route_alias_fallback(st: _Switch, key: str) -> Optional[ModelSwitchResult]: current_provider=st.current_provider, user_providers=st.user_providers, custom_providers=st.custom_providers, ) try: - fallback_result = _resolve_alias_fallback(st.raw_input, authed) + fallback_result = _resolve_alias_fallback(st.raw_input, authed, st.user_providers, st.custom_providers) except AmbiguousAliasError as err: return st.fail(_ambiguous_alias_message(err)) if fallback_result is None: @@ -1359,7 +1361,7 @@ def _route_from_model_input(st: _Switch) -> Optional[ModelSwitchResult]: st.target_provider, st.new_model, st.resolved_alias = "moa", moa_match, "" else: try: - alias_result = resolve_alias(raw_input, current_provider) + alias_result = resolve_alias(raw_input, current_provider, st.user_providers, st.custom_providers) except AmbiguousAliasError as err: return st.fail(_ambiguous_alias_message(err)) if alias_result is not None: diff --git a/tests/hermes_cli/test_ollama_cloud_auth.py b/tests/hermes_cli/test_ollama_cloud_auth.py index fe1d3416cc..0af660f4db 100644 --- a/tests/hermes_cli/test_ollama_cloud_auth.py +++ b/tests/hermes_cli/test_ollama_cloud_auth.py @@ -243,7 +243,7 @@ class TestSwitchModelDirectAliasOverride: monkeypatch.setattr(ms, "DIRECT_ALIASES", test_aliases) monkeypatch.setattr(ms, "resolve_alias", - lambda raw, prov: ("custom", "qwen3.5:397b", "qwen")) + lambda raw, prov, *_: ("custom", "qwen3.5:397b", "qwen")) monkeypatch.setattr( "hermes_cli.runtime_provider.resolve_runtime_provider", @@ -270,7 +270,7 @@ class TestSwitchModelDirectAliasOverride: } monkeypatch.setattr(ms, "DIRECT_ALIASES", test_aliases) monkeypatch.setattr(ms, "resolve_alias", - lambda raw, prov: ("custom", "local-model", "local")) + lambda raw, prov, *_: ("custom", "local-model", "local")) monkeypatch.setattr( "hermes_cli.runtime_provider.resolve_runtime_provider", lambda **kwargs: {"api_key": "", "base_url": "", "api_mode": "openai_compat", "provider": "custom"}, @@ -363,3 +363,38 @@ class TestSwitchModelDirectAliasOverride: assert result.resolved_via_alias == "corp-alias" assert result.base_url == "https://corp.example.com/v2" assert result.api_key == "sk-corp-alias" + + def test_implicit_switch_prefers_alias_of_current_legacy_custom_provider(self, monkeypatch): + """Without --provider the current provider still owns a shared model id: on + ``custom:corp-llm`` the alias naming ``corp-llm`` wins over another provider's alias.""" + import os + from pathlib import Path + + import hermes_cli.model_switch as ms + from hermes_cli.config import load_config + from hermes_cli.model_switch import DirectAlias + + (Path(os.environ["HERMES_HOME"]) / "config.yaml").write_text( + "model:\n provider: custom:corp-llm\n default: old-model\n" + "providers:\n provider-a:\n base_url: https://api-a.example.com/v1\n" + "custom_providers:\n - name: corp-llm\n" + " base_url: https://corp.example.com/v1\n api_key: sk-corp\n") + monkeypatch.setattr(ms, "DIRECT_ALIASES", { + "a-alias": DirectAlias("shared-model", "provider-a", "https://alias-host.example.com/v1", + api_key="sk-alias-host"), + "corp-alias": DirectAlias("shared-model", "corp-llm", "https://corp.example.com/v2", + api_key="sk-corp-alias"), + }) + monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model", + lambda *a, **kw: {"accepted": True, "persist": True, "recognized": True, "message": None}) + cfg = load_config() + + result = ms.switch_model( + "shared-model", "custom:corp-llm", "old-model", + current_base_url="https://corp.example.com/v1", current_api_key="sk-corp", + user_providers=cfg["providers"], custom_providers=cfg.get("custom_providers")) + + assert result.success, result.error_message + assert result.resolved_via_alias == "corp-alias" + assert result.base_url == "https://corp.example.com/v2" + assert result.api_key == "sk-corp-alias" From 537ce77f5288c24f20e90a99c27712b9d4145220 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 06:45:25 -0700 Subject: [PATCH 012/112] fix(tui-gateway): exiting mid-tool no longer orphans the foreground command's process tree Foreground terminal commands run in their own session (start_new_session) so an interrupt can kill the whole tree, which also puts them outside the host's process group. When the tui_gateway left mid-command (client closed stdin, or SIGTERM) nothing killed them: _shutdown_sessions closed the agents, the SIGTERM path hard-exits after a 1s grace, and the `bash -c` + child tree survived, reparented to init. - tools/environments/base.py: execute() records every in-flight foreground command; kill_live_foreground_processes() kills their trees through the backend's own _kill_process (the same kill an interrupt uses). - cleanup_all_environments() (the exit funnel of the CLI, one-shot, messaging gateway and terminal_tool's atexit, so `hermes serve` too) kills them first. - tui_gateway _shutdown_sessions (EOF atexit + SIGTERM handler) and the serve SIGTERM/SIGINT exit-flush handler interrupt running turns, wait up to 0.5s for them to settle so the tool call ends with a result the final persist records (no dangling tool_call in state.db), then kill any foreground command still alive. - ComputeHost.close() kills them too: every caller os._exit()s right after. Covers the case of b9dac83d366c (Desktop quit: serve SIGTERM handler) on every host. --- tests/tools/test_local_interrupt_cleanup.py | 42 +++++++++++++++ tests/tui_gateway/test_serve_exit_flush.py | 59 +++++++++++++++++++++ tools/environments/base.py | 32 +++++++++-- tools/terminal_tool_lifecycle.py | 5 +- tui_gateway/compute_host.py | 5 ++ tui_gateway/server.py | 2 +- tui_gateway/session_reaper.py | 27 ++++++++++ 7 files changed, 166 insertions(+), 6 deletions(-) diff --git a/tests/tools/test_local_interrupt_cleanup.py b/tests/tools/test_local_interrupt_cleanup.py index 74b1d55fd3..c81aa3a6ee 100644 --- a/tests/tools/test_local_interrupt_cleanup.py +++ b/tests/tools/test_local_interrupt_cleanup.py @@ -11,6 +11,7 @@ to the python process, sleep 300 survived with PPID=1 for the full 300 s because _wait_for_process never got to call _kill_process before python died. See commit message for full context. """ +import contextlib import os import signal import subprocess @@ -200,3 +201,44 @@ def test_wait_for_process_kills_subprocess_on_keyboardinterrupt(): env.cleanup() except Exception: pass + + +def _descendant_running(marker: str): + import psutil + for p in psutil.Process(os.getpid()).children(recursive=True): + try: + if marker in " ".join(p.cmdline()): + return p + except (psutil.NoSuchProcess, psutil.AccessDenied): + continue + return None + + +def test_exit_cleanup_kills_foreground_command_still_running(monkeypatch): + """The host-exit funnel (CLI, one-shot, messaging gateway, serve atexit) must take an + in-flight foreground command's process group with it: it runs in its own session, so + the host exiting mid-command would otherwise orphan it to init.""" + from tools import terminal_tool_lifecycle + + monkeypatch.setattr(terminal_tool_lifecycle, "_scratch_paths", lambda: []) + env = LocalEnvironment(cwd="/tmp") + result: dict = {} + t = threading.Thread(target=lambda: result.update(env.execute("sleep 3517", timeout=600)), daemon=True) + try: + t.start() + deadline = time.monotonic() + 20.0 + while (proc := _descendant_running("sleep 3517")) is None and time.monotonic() < deadline: + time.sleep(0.05) + assert proc is not None, "test setup: foreground sleep never started" + pgid = os.getpgid(proc.pid) + + terminal_tool_lifecycle.cleanup_all_environments() + + assert _wait_for_pgid_exit(pgid, timeout=15.0), ( + f"foreground command survived exit cleanup:\n{_process_group_snapshot(pgid)}") + t.join(timeout=15.0) + assert not t.is_alive() and result.get("returncode") not in (None, 0), result + finally: + with contextlib.suppress(Exception): + os.killpg(os.getpgid(proc.pid), signal.SIGKILL) + env.cleanup() diff --git a/tests/tui_gateway/test_serve_exit_flush.py b/tests/tui_gateway/test_serve_exit_flush.py index de69e54597..2a5e520d1e 100644 --- a/tests/tui_gateway/test_serve_exit_flush.py +++ b/tests/tui_gateway/test_serve_exit_flush.py @@ -157,6 +157,65 @@ def test_shutdown_sessions_flushes_before_teardown(monkeypatch): assert "close:sess-order" in order +def test_shutdown_mid_tool_kills_the_command_and_keeps_its_result(monkeypatch): + """SIGTERM/EOF while a turn's foreground terminal command runs: the shutdown chokepoint must + end the command's process group (it would outlive the gateway, reparented to init) and the + turn's tool result must be in the transcript before per-session teardown persists it.""" + import threading + + import psutil + + from tools.environments.local import LocalEnvironment + from tools.interrupt import set_interrupt + + env = LocalEnvironment(cwd=os.getcwd()) + messages = [{"role": "assistant", "tool_calls": [{"id": "call-1"}]}] + + def turn(): + out = env.execute("sleep 3518", timeout=600) + time.sleep(0.2) # the agent's post-tool bookkeeping before the result lands in history + messages.append({"role": "tool", "tool_call_id": "call-1", "content": out["output"]}) + + run_thread = threading.Thread(target=turn, daemon=True) + + class _Agent: + _session_messages = messages + + def interrupt(self, message=None): # the real agent fans this out to its tool threads + set_interrupt(True, thread_id=run_thread.ident) + + run_thread.start() + deadline = time.monotonic() + 20.0 + sleeper = None + while sleeper is None and time.monotonic() < deadline: + sleeper = next((p for p in psutil.Process().children(recursive=True) + if p.name() == "sleep" and "3518" in " ".join(p.cmdline())), None) + time.sleep(0.05) + assert sleeper is not None, "test setup: foreground sleep never started" + + at_teardown: list = [] + monkeypatch.setattr(server, "_release_gateway_wake_owner", lambda: None, raising=False) + monkeypatch.setattr(server, "_flush_sessions_before_exit", lambda budget_s=None: 0) + monkeypatch.setattr(server, "_close_session_by_id", lambda sid, **kw: at_teardown.append(list(messages))) + session = {"agent": _Agent(), "session_key": "sess-mid-tool", "running": True, + "_run_thread": run_thread, "history_lock": threading.RLock()} + with server._sessions_lock: + server._sessions["sess-mid-tool"] = session + try: + server._shutdown_sessions() + _gone, alive = psutil.wait_procs([sleeper], timeout=15.0) + assert not alive, "foreground command survived gateway shutdown" + assert at_teardown and at_teardown[0][-1].get("tool_call_id") == "call-1", ( + f"teardown persisted a tool_call with no result: {at_teardown}") + finally: + with server._sessions_lock: + server._sessions.pop("sess-mid-tool", None) + if sleeper.is_running(): + sleeper.kill() + set_interrupt(False, thread_id=run_thread.ident) + env.cleanup() + + def test_periodic_flush_respects_interval_with_fake_clock( registered_session, monkeypatch ): diff --git a/tools/environments/base.py b/tools/environments/base.py index f3e3627c6a..c68c542b35 100644 --- a/tools/environments/base.py +++ b/tools/environments/base.py @@ -52,6 +52,24 @@ if _DEBUG_INTERRUPT: # long-running _wait_for_process loops can report liveness to the gateway. _activity_callback_local = threading.local() +# Foreground commands in flight in THIS process, across every environment. Each runs in its +# own session/process group, so a host that exits mid-command (TUI client gone, SIGTERM) would +# orphan the whole tree; the process-exit funnel ``cleanup_all_environments`` kills them. +_live_foreground: dict[int, tuple["BaseEnvironment", "ProcessHandle"]] = {} +_live_foreground_lock = threading.Lock() + + +def kill_live_foreground_processes() -> int: + """Kill every in-flight foreground command's process tree; returns how many were signalled.""" + with _live_foreground_lock: + live = list(_live_foreground.values()) + for env, proc in live: + try: + env._kill_process(proc) + except Exception: + logger.debug("exit-time kill of a foreground command failed", exc_info=True) + return len(live) + class FileFetchError(RuntimeError): """A file could not be extracted from the backend filesystem.""" @@ -542,10 +560,16 @@ class BaseEnvironment(ABC): set_activity_callback(parent_activity_cb) spawned = self._run_bash(wrapped, login=login, timeout=effective_timeout, stdin_data=effective_stdin) proc_holder.append(spawned) - return self._wait_for_process( - spawned, timeout=effective_timeout, bounded_capture=bounded_capture, - watch_interrupt_tid=parent_tid, - **({"yield_handler": yield_handler} if yield_handler is not None else {})) + with _live_foreground_lock: + _live_foreground[id(spawned)] = (self, spawned) + try: + return self._wait_for_process( + spawned, timeout=effective_timeout, bounded_capture=bounded_capture, + watch_interrupt_tid=parent_tid, + **({"yield_handler": yield_handler} if yield_handler is not None else {})) + finally: + with _live_foreground_lock: + _live_foreground.pop(id(spawned), None) def _on_timeout() -> None: if proc_holder: diff --git a/tools/terminal_tool_lifecycle.py b/tools/terminal_tool_lifecycle.py index ba2c2cc9e1..74f48b73e9 100644 --- a/tools/terminal_tool_lifecycle.py +++ b/tools/terminal_tool_lifecycle.py @@ -263,8 +263,11 @@ def is_persistent_env(task_id: str) -> bool: def cleanup_all_environments(): - """Clean up ALL active environments. Use with caution.""" + """Clean up ALL active environments (process exit). Use with caution.""" + from tools.environments.base import kill_live_foreground_processes from tools.terminal_tool import _active_environments + # A command still running when the host exits would outlive it in its own process group. + kill_live_foreground_processes() cleaned = 0 for task_id in list(_active_environments.keys()): try: diff --git a/tui_gateway/compute_host.py b/tui_gateway/compute_host.py index 8e7c233cbe..201fff5303 100644 --- a/tui_gateway/compute_host.py +++ b/tui_gateway/compute_host.py @@ -96,6 +96,11 @@ class ComputeHost: def close(self) -> None: self._closed.set() self._executor.shutdown(wait=False, cancel_futures=True) + # Every caller hard-exits next (os._exit skips atexit): a foreground command still + # running in its own process group would outlive the host. + with contextlib.suppress(Exception): + from tools.environments.base import kill_live_foreground_processes + kill_live_foreground_processes() def shutdown(self, *, reason: str = "shutdown", wait: float = 10.0) -> None: """Drain in-flight turns, then finalize every session. diff --git a/tui_gateway/server.py b/tui_gateway/server.py index d292c8a7d5..b232f7343f 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -346,7 +346,7 @@ def _load_interim_assistant_messages() -> bool: def _shutdown_sessions() -> None: # Durable-first: flush transcripts (bounded budget) BEFORE the slow teardown so a supervisor SIGKILL can't lose them. - for step in (_flush_sessions_before_exit, _release_gateway_wake_owner): + for step in (_flush_sessions_before_exit, _release_gateway_wake_owner, _stop_turns_before_exit): with contextlib.suppress(Exception): step() with _sessions_lock: diff --git a/tui_gateway/session_reaper.py b/tui_gateway/session_reaper.py index 6c24492960..43e7b679ef 100644 --- a/tui_gateway/session_reaper.py +++ b/tui_gateway/session_reaper.py @@ -113,6 +113,29 @@ def _flush_sessions_before_exit(budget_s: float | None = None) -> int: return result["flushed"] +# Bounded wait for interrupted turns on the way out; the SIGTERM path hard-exits after a ~1s grace. +_EXIT_TURN_SETTLE_S = 0.5 + + +def _stop_turns_before_exit(budget_s: float | None = None) -> None: + """Interrupt every in-flight turn and give it ``budget_s`` to settle, so a running tool call ends + with a result the teardown's final persist records, then kill any foreground command still alive: + it runs in its own process group and would otherwise outlive the gateway, reparented to init.""" + with _sessions_lock: + running = [(sid, s) for sid, s in _sessions.items() if s.get("running")] + threads = [] + for sid, session in running: + with contextlib.suppress(Exception): + _interrupt_session_turn(sid, session) + if (t := session.get("_run_thread")) is not None and t is not threading.current_thread(): + threads.append(t) + deadline = time.monotonic() + (_EXIT_TURN_SETTLE_S if budget_s is None else max(0.0, budget_s)) + for t in threads: + t.join(max(0.0, deadline - time.monotonic())) + from tools.environments.base import kill_live_foreground_processes + kill_live_foreground_processes() + + _exit_flush_prev_handlers: dict[int, Any] = {} _exit_flush_handlers_installed = False @@ -122,6 +145,10 @@ def _handle_exit_flush_signal(signum, frame) -> None: handler, or the default disposition) — this only *prepends* a bounded flush.""" with contextlib.suppress(Exception): _flush_sessions_before_exit() + # The group signal that stopped us never reaches a command in its own session: reap it now, + # before a supervisor's SIGKILL can cut the graceful shutdown (and its atexit) short. + with contextlib.suppress(Exception): + _stop_turns_before_exit() import signal as _signal prev = _exit_flush_prev_handlers.get(signum) if callable(prev): From 17a137d8a3d4dbaa5081d558bfd271eeead9c58a Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 09:47:05 -0700 Subject: [PATCH 013/112] fix(tui-gateway): a SIGTERM-ignoring command no longer survives the gateway's SIGTERM exit The SIGTERM handler arms a 1s os._exit timer, then runs _shutdown_sessions: a flush of up to 5s, then _stop_turns_before_exit, whose kill was the graceful TERM, wait 1s, KILL. A command that ignores SIGTERM was still alive when the timer fired, and os._exit left it reparented to init (live: `trap '' TERM; sleep 3600` survived a SIGTERM to `python -m tui_gateway.entry`). - kill_live_foreground_processes(now=True): SIGKILL each in-flight foreground tree at once, no TERM grace, no wait (BaseEnvironment._force_kill_process; LocalEnvironment kills the recorded process group, never our own). - The grace timer's exit (entry._hard_exit) runs it before os._exit. - _stop_turns_before_exit SIGKILLs whatever is still alive halfway through its settle budget (it ignored the interrupt's TERM), so the tool call still ends with a result the teardown persists instead of a dangling tool_call in state.db. - The other hard exits that skip cleanup do the same before os._exit: the serve parent-death watchdog, the CLI exit watchdog, the kanban worker's SIGTERM path, and the messaging gateway's shutdown and loop-liveness watchdogs. - Deflake test_shutdown_mid_tool_kills_the_command_and_keeps_its_result: the 0.5s settle budget was too tight under -n 40 (1 red in 9 runs); the join returns when the turn ends. --- gateway/shutdown_watchdog.py | 13 +++++-- hermes_cli/cli_shutdown.py | 4 +++ hermes_cli/cli_single_query.py | 4 +++ hermes_cli/web_server_lifecycle.py | 6 ++++ tests/tui_gateway/test_serve_exit_flush.py | 42 ++++++++++++++++++++++ tools/environments/base.py | 13 +++++-- tools/environments/local.py | 11 ++++++ tui_gateway/entry.py | 12 ++++++- tui_gateway/session_reaper.py | 20 +++++++---- 9 files changed, 113 insertions(+), 12 deletions(-) diff --git a/gateway/shutdown_watchdog.py b/gateway/shutdown_watchdog.py index 1db12ea15c..db15623742 100644 --- a/gateway/shutdown_watchdog.py +++ b/gateway/shutdown_watchdog.py @@ -135,7 +135,7 @@ def start_loop_liveness_watchdog( if stop_event.is_set(): return _mark_exited_quietly(exit_code, "loop_liveness_watchdog") - os._exit(exit_code) + _hard_exit(exit_code) thread = threading.Thread(target=_watchdog, daemon=True, name="gateway-loop-liveness-watchdog") try: thread.start() @@ -145,6 +145,15 @@ def start_loop_liveness_watchdog( return _LoopLivenessWatchdogHandle(stop_event, thread) +def _hard_exit(exit_code: int) -> None: + """``os._exit`` skips every cleanup: SIGKILL in-flight foreground commands first, they run in their + own process group and would outlive the gateway, reparented to init.""" + with contextlib.suppress(Exception): + from tools.environments.base import kill_live_foreground_processes + kill_live_foreground_processes(now=True) + os._exit(exit_code) + + def _mark_exited_quietly(exit_code: int, reason: str) -> None: """Best-effort terminal stamp on BOTH lifecycle records before ``os._exit`` skips teardown: the lifecycle ledger (so the next boot names the watchdog, not SIGKILL/OOM) and @@ -293,7 +302,7 @@ def arm_shutdown_watchdog( from hermes_logging import drain_log_queue drain_log_queue(timeout=1.0) _mark_exited_quietly(exit_code, "shutdown_watchdog") - os._exit(exit_code) + _hard_exit(exit_code) try: threading.Thread(target=_watchdog, daemon=True, name=name).start() except Exception: diff --git a/hermes_cli/cli_shutdown.py b/hermes_cli/cli_shutdown.py index 4fda381bb0..0a6fa48f10 100644 --- a/hermes_cli/cli_shutdown.py +++ b/hermes_cli/cli_shutdown.py @@ -89,6 +89,10 @@ def _arm_exit_watchdog(timeout_s: float | None = None, *, from_signal: bool = Fa except Exception: pass _flush_logging_and_stdio() + # os._exit skips cleanup: a foreground command in its own process group would outlive us. + with suppress(Exception): + from tools.environments.base import kill_live_foreground_processes + kill_live_foreground_processes(now=True) os._exit(0) with suppress(Exception): # never block shutdown on watchdog setup diff --git a/hermes_cli/cli_single_query.py b/hermes_cli/cli_single_query.py index ed35d53ee5..eb1b6546a0 100644 --- a/hermes_cli/cli_single_query.py +++ b/hermes_cli/cli_single_query.py @@ -399,6 +399,10 @@ def _install_single_query_signal_handlers(cli): # store here or the worker's turn (and its usage deltas) never become durable (#88583 / # #50881 class). Best-effort under the SIGALRM deadman above. _flush_one_shot_session_store(cli) + # The worker's command runs in its own process group: SIGKILL it or it outlives os._exit. + with suppress(Exception): + from tools.environments.base import kill_live_foreground_processes + kill_live_foreground_processes(now=True) _flush_logging_and_stdio() os._exit(0) raise KeyboardInterrupt() diff --git a/hermes_cli/web_server_lifecycle.py b/hermes_cli/web_server_lifecycle.py index 792f96fa5f..89f59725e8 100644 --- a/hermes_cli/web_server_lifecycle.py +++ b/hermes_cli/web_server_lifecycle.py @@ -340,6 +340,12 @@ def _start_parent_death_watchdog() -> None: ) except Exception: pass + # os._exit skips every cleanup: a foreground command in its own process group would outlive us. + try: + from tools.environments.base import kill_live_foreground_processes + kill_live_foreground_processes(now=True) + except Exception: + pass os._exit(0) threading.Thread(target=_loop, daemon=True, name="serve-parent-watchdog").start() diff --git a/tests/tui_gateway/test_serve_exit_flush.py b/tests/tui_gateway/test_serve_exit_flush.py index 2a5e520d1e..44e7278309 100644 --- a/tests/tui_gateway/test_serve_exit_flush.py +++ b/tests/tui_gateway/test_serve_exit_flush.py @@ -197,6 +197,9 @@ def test_shutdown_mid_tool_kills_the_command_and_keeps_its_result(monkeypatch): monkeypatch.setattr(server, "_release_gateway_wake_owner", lambda: None, raising=False) monkeypatch.setattr(server, "_flush_sessions_before_exit", lambda budget_s=None: 0) monkeypatch.setattr(server, "_close_session_by_id", lambda sid, **kw: at_teardown.append(list(messages))) + # The join returns as soon as the turn ends; 0.5s is too tight for the kill + bookkeeping under -n 40. + from tui_gateway import session_reaper + monkeypatch.setattr(session_reaper, "_EXIT_TURN_SETTLE_S", 10.0) session = {"agent": _Agent(), "session_key": "sess-mid-tool", "running": True, "_run_thread": run_thread, "history_lock": threading.RLock()} with server._sessions_lock: @@ -216,6 +219,45 @@ def test_shutdown_mid_tool_kills_the_command_and_keeps_its_result(monkeypatch): env.cleanup() +@pytest.mark.skipif(os.name == "nt", reason="POSIX process groups + trap") +def test_sigterm_grace_hard_exit_kills_a_sigterm_ignoring_command(monkeypatch): + """The SIGTERM path os._exit()s after a ~1s grace, while the graceful foreground kill runs after a + flush of up to 5s and then waits 1s between TERM and KILL. The grace timer's exit must SIGKILL the + tree itself, at once, or a command that ignores SIGTERM survives, reparented to init.""" + import threading + + import psutil + + from tools.environments.local import LocalEnvironment + from tui_gateway import entry + + env = LocalEnvironment(cwd=os.getcwd()) + run_thread = threading.Thread( + target=lambda: env.execute("trap '' TERM; sleep 3522", timeout=600), daemon=True) + run_thread.start() + deadline = time.monotonic() + 20.0 + sleeper = None + while sleeper is None and time.monotonic() < deadline: + sleeper = next((p for p in psutil.Process().children(recursive=True) + if p.name() == "sleep" and "3522" in " ".join(p.cmdline())), None) + time.sleep(0.05) + assert sleeper is not None, "test setup: foreground sleep never started" + exits: list = [] + monkeypatch.setattr(entry.os, "_exit", exits.append) + try: + t0 = time.monotonic() + entry._hard_exit() + elapsed = time.monotonic() - t0 + _gone, alive = psutil.wait_procs([sleeper], timeout=5.0) + assert not alive, "SIGTERM-ignoring foreground command survived the hard exit" + assert exits == [0] and elapsed < 0.9, f"hard exit waited {elapsed:.2f}s (a TERM grace) first" + finally: + if sleeper.is_running(): + sleeper.kill() + run_thread.join(5.0) + env.cleanup() + + def test_periodic_flush_respects_interval_with_fake_clock( registered_session, monkeypatch ): diff --git a/tools/environments/base.py b/tools/environments/base.py index c68c542b35..9795ab6db2 100644 --- a/tools/environments/base.py +++ b/tools/environments/base.py @@ -59,13 +59,16 @@ _live_foreground: dict[int, tuple["BaseEnvironment", "ProcessHandle"]] = {} _live_foreground_lock = threading.Lock() -def kill_live_foreground_processes() -> int: - """Kill every in-flight foreground command's process tree; returns how many were signalled.""" +def kill_live_foreground_processes(*, now: bool = False) -> int: + """Kill every in-flight foreground command's process tree; returns how many were signalled. + + ``now=True`` is for a caller about to ``os._exit``: the graceful kill TERMs, waits and only then + KILLs, so a SIGTERM-ignoring command outlives a hard exit that lands inside that window.""" with _live_foreground_lock: live = list(_live_foreground.values()) for env, proc in live: try: - env._kill_process(proc) + (env._force_kill_process if now else env._kill_process)(proc) except Exception: logger.debug("exit-time kill of a foreground command failed", exc_info=True) return len(live) @@ -468,6 +471,10 @@ class BaseEnvironment(ABC): except (ProcessLookupError, PermissionError, OSError): pass + def _force_kill_process(self, proc: ProcessHandle): + """Kill without waiting, for a host that hard-exits next. Subclasses kill the whole tree.""" + self._kill_process(proc) + # --- CWD extraction --- def _update_cwd(self, result: dict): """Extract CWD from command output. Override for local file-based read.""" diff --git a/tools/environments/local.py b/tools/environments/local.py index fb0ea98835..89a06eec33 100644 --- a/tools/environments/local.py +++ b/tools/environments/local.py @@ -963,6 +963,17 @@ class LocalEnvironment(BaseEnvironment): with contextlib.suppress(Exception): proc.kill() + def _force_kill_process(self, proc): + """SIGKILL the whole group with no TERM grace or wait: the caller os._exit()s next.""" + if _IS_WINDOWS: # already a forced tree kill + return self._kill_process(proc) + with contextlib.suppress(OSError): + pgid = getattr(proc, "_hermes_pgid", None) or os.getpgid(proc.pid) + if pgid != os.getpgrp(): # never our own group (see _kill_process_group_posix) + os.killpg(pgid, signal.SIGKILL) # windows-footgun: ok — POSIX only (_IS_WINDOWS returned above) + with contextlib.suppress(OSError): + proc.kill() + def _extract_cwd_from_output(self, result: dict): """Base semantics plus: Git Bash ``pwd -P`` emits MSYS form on Windows — normalize to native and require the dir to exist, else ``_run_bash`` would diff --git a/tui_gateway/entry.py b/tui_gateway/entry.py index 2d948c25d9..aebfbeb802 100644 --- a/tui_gateway/entry.py +++ b/tui_gateway/entry.py @@ -100,6 +100,16 @@ def _append_crash_log(header: str, dump=None) -> None: dump(f) +def _hard_exit() -> None: + """The grace timer's ``os._exit``. The flush runs first and the graceful foreground kill + (TERM, wait, KILL) after it, so a SIGTERM-ignoring command is usually still alive here: + SIGKILL its tree now or it outlives us, reparented to init.""" + with suppress(Exception): + from tools.environments.base import kill_live_foreground_processes + kill_live_foreground_processes(now=True) + os._exit(0) + + def _log_signal(signum: int, frame) -> None: """Capture WHICH thread and WHERE a termination signal hit us, then exit. ``sys.exit(0)`` alone raced the worker pool (a thread holding ``_stdout_lock`` mid-flush blocks interpreter @@ -121,7 +131,7 @@ def _log_signal(signum: int, frame) -> None: _append_crash_log(f"{name} received · {time.strftime('%Y-%m-%d %H:%M:%S')}", _dump) print(f"[gateway-signal] {name}", file=sys.stderr, flush=True) # ``os._exit`` skips atexit but breaks the mid-flush deadlock; the crash log is the trail. - timer = threading.Timer(_shutdown_grace_seconds(), lambda: os._exit(0)) + timer = threading.Timer(_shutdown_grace_seconds(), _hard_exit) timer.daemon = True timer.start() # atexit (_shutdown_sessions) can be blocked past the grace window by a worker holding diff --git a/tui_gateway/session_reaper.py b/tui_gateway/session_reaper.py index 43e7b679ef..21e92dad12 100644 --- a/tui_gateway/session_reaper.py +++ b/tui_gateway/session_reaper.py @@ -119,8 +119,10 @@ _EXIT_TURN_SETTLE_S = 0.5 def _stop_turns_before_exit(budget_s: float | None = None) -> None: """Interrupt every in-flight turn and give it ``budget_s`` to settle, so a running tool call ends - with a result the teardown's final persist records, then kill any foreground command still alive: - it runs in its own process group and would otherwise outlive the gateway, reparented to init.""" + with a result the teardown's final persist records. A foreground command runs in its own process + group and would outlive the gateway, reparented to init. One still alive halfway through the budget + ignored the interrupt's SIGTERM: SIGKILL it then, early enough for its result to land as well (the + interrupt's own TERM, 1s, KILL outlasts the SIGTERM path's ~1s grace).""" with _sessions_lock: running = [(sid, s) for sid, s in _sessions.items() if s.get("running")] threads = [] @@ -129,11 +131,17 @@ def _stop_turns_before_exit(budget_s: float | None = None) -> None: _interrupt_session_turn(sid, session) if (t := session.get("_run_thread")) is not None and t is not threading.current_thread(): threads.append(t) - deadline = time.monotonic() + (_EXIT_TURN_SETTLE_S if budget_s is None else max(0.0, budget_s)) - for t in threads: - t.join(max(0.0, deadline - time.monotonic())) + budget = _EXIT_TURN_SETTLE_S if budget_s is None else max(0.0, budget_s) + deadline = time.monotonic() + budget + + def _join(until: float) -> None: + for t in threads: + t.join(max(0.0, until - time.monotonic())) + + _join(deadline - budget / 2) from tools.environments.base import kill_live_foreground_processes - kill_live_foreground_processes() + kill_live_foreground_processes(now=True) + _join(deadline) _exit_flush_prev_handlers: dict[int, Any] = {} From 3a001a4d7fb925651fdb4509fbe0fe62d4e9c4ed Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:31:06 +0000 Subject: [PATCH 014/112] fix(tui-gateway): the hard-exit kill owns every foreground spawn and never blocks Re-review findings on the foreground-process registry: - Spawn vs exit races. A child spawned but not yet registered when the hard-exit kill ran, or a command launched after the kill took its snapshot, survived under init. The hard-exit kill now raises a one-way exit fence and waits (bounded) for spawns already past it to register; a spawn refuses once the fence is up, and one that registers after a timed-out wait kills itself. - The immediate kill fell back to proc.kill() for every handle. On Modal/Daytona/Vercel that is a blocking SDK cancel (an 8s cancel made _hard_exit take 8s). Popen handles are still killed inline (killpg, never blocks); every other handle's kill runs on a daemon thread under one shared 0.5s deadline. - The registry lock was a plain Lock: a signal landing on a thread that held it deadlocked the exit. It is now an RLock (via a Condition) and the hard-exit path only takes it with a timeout, falling back to a lock-free copy. - The kanban worker's SIGALRM deadman os._exit()ed without the kill; it now goes through it. --- hermes_cli/cli_single_query.py | 15 ++-- tests/conftest.py | 8 ++ tests/tools/test_local_interrupt_cleanup.py | 72 +++++++++++++++++ tools/environments/base.py | 85 ++++++++++++++++++--- 4 files changed, 162 insertions(+), 18 deletions(-) diff --git a/hermes_cli/cli_single_query.py b/hermes_cli/cli_single_query.py index eb1b6546a0..3c6257053d 100644 --- a/hermes_cli/cli_single_query.py +++ b/hermes_cli/cli_single_query.py @@ -371,6 +371,13 @@ def _install_single_query_signal_handlers(cli): from cli import _arm_exit_watchdog_on_shutdown_signal, _flush_logging_and_stdio, _flush_one_shot_session_store, _interrupt_agent_for_signal import signal as _signal + def _kill_foreground_and_exit(*_): + # The worker's command runs in its own process group: SIGKILL it or it outlives os._exit. + with suppress(Exception): + from tools.environments.base import kill_live_foreground_processes + kill_live_foreground_processes(now=True) + os._exit(0) + def _signal_handler_q(signum, frame): logger.debug("Received signal %s in single-query mode", signum) _arm_exit_watchdog_on_shutdown_signal() # covers wedges in the unwind below @@ -390,7 +397,7 @@ def _install_single_query_signal_handlers(cli): if os.environ.get("HERMES_KANBAN_TASK"): with suppress(Exception): if hasattr(_signal, "SIGALRM"): - _signal.signal(_signal.SIGALRM, lambda *_: os._exit(0)) + _signal.signal(_signal.SIGALRM, _kill_foreground_and_exit) _signal.alarm(5) with suppress(Exception): # Durable flush FIRST: memory-provider shutdown inside _run_cleanup can issue aux-LLM calls, @@ -399,12 +406,8 @@ def _install_single_query_signal_handlers(cli): # store here or the worker's turn (and its usage deltas) never become durable (#88583 / # #50881 class). Best-effort under the SIGALRM deadman above. _flush_one_shot_session_store(cli) - # The worker's command runs in its own process group: SIGKILL it or it outlives os._exit. - with suppress(Exception): - from tools.environments.base import kill_live_foreground_processes - kill_live_foreground_processes(now=True) _flush_logging_and_stdio() - os._exit(0) + _kill_foreground_and_exit() raise KeyboardInterrupt() with suppress(Exception): # restricted environments for _name in ("SIGINT", "SIGTERM", "SIGHUP"): diff --git a/tests/conftest.py b/tests/conftest.py index 21cbeb9679..308ec1c198 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -652,6 +652,14 @@ def _isolate_hermes_home(_hermetic_environment): return None +@pytest.fixture(autouse=True) +def _reset_foreground_exit_fence(): + """A test that drives a hard-exit path raises the one-way foreground-spawn fence; lower it after.""" + yield + if (base := sys.modules.get("tools.environments.base")) is not None: + base._exit_fenced = False + + @pytest.fixture(autouse=True) def _neutralize_kanban_memory_guard(request, monkeypatch): """Pin the kanban dispatcher's memory guard to "no data" for every test. diff --git a/tests/tools/test_local_interrupt_cleanup.py b/tests/tools/test_local_interrupt_cleanup.py index c81aa3a6ee..376e7ba4c5 100644 --- a/tests/tools/test_local_interrupt_cleanup.py +++ b/tests/tools/test_local_interrupt_cleanup.py @@ -242,3 +242,75 @@ def test_exit_cleanup_kills_foreground_command_still_running(monkeypatch): with contextlib.suppress(Exception): os.killpg(os.getpgid(proc.pid), signal.SIGKILL) env.cleanup() + + +# Child for the hard-exit race tests: it really os._exit()s right after the kill, like the watchdogs. +_HARD_EXIT_RACE_CHILD = r""" +import os, sys, threading +from tools.environments import base +from tools.environments.local import LocalEnvironment +scenario, cmd = sys.argv[1], sys.argv[2] +env = LocalEnvironment(cwd=os.getcwd()) +if scenario == "spawn_before_publish": + spawned, release = threading.Event(), threading.Event() + real_run_bash = LocalEnvironment._run_bash + def gated(self, command, **kw): + proc = real_run_bash(self, command, **kw) + if cmd in command: + spawned.set() + release.wait(10) # barrier: the child exists, it is not published yet + return proc + LocalEnvironment._run_bash = gated + threading.Thread(target=env.execute, args=(cmd,), kwargs={"timeout": 600}, daemon=True).start() + assert spawned.wait(20) + threading.Timer(0.2, release.set).start() # registration lands while the killer is in flight + base.kill_live_foreground_processes(now=True) +else: # launch_after_fence + base.kill_live_foreground_processes(now=True) + t = threading.Thread(target=env.execute, args=(cmd,), kwargs={"timeout": 600}, daemon=True) + t.start() + t.join(3) +os._exit(0) +""" + + +@pytest.mark.skipif(os.name == "nt", reason="POSIX process groups") +@pytest.mark.live_system_guard_bypass # a red run must reap survivors reparented to init +@pytest.mark.parametrize("scenario", ["spawn_before_publish", "launch_after_fence"]) +def test_hard_exit_leaves_no_foreground_survivor_around_the_spawn(scenario, tmp_path): + """A hard exit must own every foreground child: one spawned but not yet registered when the kill + runs, and one launched after the kill took its snapshot (the exit fence refuses it).""" + import sys + + import psutil + + cmd = f"sleep {35000 + os.getpid() % 1000}.{len(scenario)}" + repo = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + r = subprocess.run([sys.executable, "-c", _HARD_EXIT_RACE_CHILD, scenario, cmd], cwd=str(tmp_path), + env={**os.environ, "PYTHONPATH": repo}, capture_output=True, text=True, timeout=120) + assert r.returncode == 0, r.stderr[-2000:] + time.sleep(0.3) + survivors = [p for p in psutil.process_iter(["cmdline"]) if cmd in " ".join(p.info["cmdline"] or [])] + for p in survivors: + with contextlib.suppress(psutil.Error): + p.kill() + assert not survivors, f"{scenario}: {[' '.join(p.info['cmdline']) for p in survivors]} outlived the hard exit" + + +def test_hard_exit_kill_never_blocks_on_a_slow_remote_cancel(monkeypatch): + """Modal/Daytona/Vercel cancel through a blocking SDK call; the hard-exit kill runs just before + os._exit, so it must give those one short deadline instead of waiting them out.""" + from tools.environments import base + from tools.environments.base_output import _ThreadedProcessHandle + + class _SdkEnv: # the kill every SDK backend inherits: proc.kill() -> cancel_fn + _kill_process = base.BaseEnvironment._kill_process + _force_kill_process = base.BaseEnvironment._force_kill_process + + cancelled = threading.Event() + handle = _ThreadedProcessHandle(lambda: ("", 0), cancel_fn=lambda: (cancelled.set(), time.sleep(8))) + monkeypatch.setitem(base._live_foreground, id(handle), (_SdkEnv(), handle)) + t0 = time.monotonic() + base.kill_live_foreground_processes(now=True) + elapsed = time.monotonic() - t0 + assert cancelled.is_set() and elapsed < 1.0, f"blocked {elapsed:.2f}s" diff --git a/tools/environments/base.py b/tools/environments/base.py index 9795ab6db2..bf9efa5fe9 100644 --- a/tools/environments/base.py +++ b/tools/environments/base.py @@ -11,6 +11,7 @@ import json import logging import os import shlex +import subprocess import threading import time import uuid @@ -56,21 +57,75 @@ _activity_callback_local = threading.local() # own session/process group, so a host that exits mid-command (TUI client gone, SIGTERM) would # orphan the whole tree; the process-exit funnel ``cleanup_all_environments`` kills them. _live_foreground: dict[int, tuple["BaseEnvironment", "ProcessHandle"]] = {} -_live_foreground_lock = threading.Lock() +# Reentrant, and the hard-exit path only ever takes it with a timeout: a signal handler can run +# on a thread that already holds it. +_live_foreground_cond = threading.Condition(threading.RLock()) +_exit_fenced = False # one-way, set by the hard-exit kill: no foreground command spawns after it +_spawns_in_flight = 0 # past the fence check, child maybe alive, not yet in _live_foreground +_HARD_KILL_BUDGET_S = 0.5 + + +def _enter_foreground_spawn() -> bool: + global _spawns_in_flight + with _live_foreground_cond: + if _exit_fenced: + return False + _spawns_in_flight += 1 + return True + + +def _leave_foreground_spawn(env: "BaseEnvironment", spawned) -> bool: + """Publish ``spawned`` (None: the spawn failed); True when the exit fence went up meanwhile.""" + global _spawns_in_flight + with _live_foreground_cond: + _spawns_in_flight -= 1 + if spawned is not None: + _live_foreground[id(spawned)] = (env, spawned) + _live_foreground_cond.notify_all() + return _exit_fenced + + +def _quiet_kill(kill: Callable, proc) -> None: + try: + kill(proc) + except Exception: + logger.debug("exit-time kill of a foreground command failed", exc_info=True) def kill_live_foreground_processes(*, now: bool = False) -> int: """Kill every in-flight foreground command's process tree; returns how many were signalled. ``now=True`` is for a caller about to ``os._exit``: the graceful kill TERMs, waits and only then - KILLs, so a SIGTERM-ignoring command outlives a hard exit that lands inside that window.""" - with _live_foreground_lock: - live = list(_live_foreground.values()) - for env, proc in live: + KILLs, so a SIGTERM-ignoring command outlives a hard exit that lands inside that window. It also + raises the exit fence and waits for spawns already past it to register, so no command started + around the snapshot survives, and it never blocks past ``_HARD_KILL_BUDGET_S``: SDK cancels + (Modal, Daytona, Vercel) run on daemon threads under that one deadline.""" + global _exit_fenced + if not now: + with _live_foreground_cond: + live = list(_live_foreground.values()) + for env, proc in live: + _quiet_kill(env._kill_process, proc) + return len(live) + deadline = time.monotonic() + _HARD_KILL_BUDGET_S + _exit_fenced = True + if _live_foreground_cond.acquire(timeout=_HARD_KILL_BUDGET_S): try: - (env._force_kill_process if now else env._kill_process)(proc) - except Exception: - logger.debug("exit-time kill of a foreground command failed", exc_info=True) + _live_foreground_cond.wait_for(lambda: _spawns_in_flight == 0, max(0.0, deadline - time.monotonic())) + live = list(_live_foreground.values()) + finally: + _live_foreground_cond.release() + else: # the holder is stuck under our signal: a lock-free copy beats hanging the exit + live = list(_live_foreground.values()) + remote = [] + for env, proc in live: + if isinstance(proc, subprocess.Popen): # killpg/kill: never blocks + _quiet_kill(env._force_kill_process, proc) + else: + remote.append(threading.Thread(target=_quiet_kill, args=(env._force_kill_process, proc), daemon=True)) + remote[-1].start() + for t in remote: + t.join(max(0.0, deadline - time.monotonic())) return len(live) @@ -565,17 +620,23 @@ class BaseEnvironment(ABC): def _spawn_and_wait() -> dict: if parent_activity_cb is not None: set_activity_callback(parent_activity_cb) - spawned = self._run_bash(wrapped, login=login, timeout=effective_timeout, stdin_data=effective_stdin) + if not _enter_foreground_spawn(): + return {"output": "[host is exiting: command not started]", "returncode": 130} + spawned = None + try: + spawned = self._run_bash(wrapped, login=login, timeout=effective_timeout, stdin_data=effective_stdin) + finally: + fenced = _leave_foreground_spawn(self, spawned) proc_holder.append(spawned) - with _live_foreground_lock: - _live_foreground[id(spawned)] = (self, spawned) + if fenced: # the hard-exit kill may have stopped waiting for us before we registered + self._force_kill_process(spawned) try: return self._wait_for_process( spawned, timeout=effective_timeout, bounded_capture=bounded_capture, watch_interrupt_tid=parent_tid, **({"yield_handler": yield_handler} if yield_handler is not None else {})) finally: - with _live_foreground_lock: + with _live_foreground_cond: _live_foreground.pop(id(spawned), None) def _on_timeout() -> None: From b8e69858021e1fd286e12bc3948a05499327de5e Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 15:26:26 -0700 Subject: [PATCH 015/112] test(e2e): gateway exit mid-tool no longer a known break; the tree kill lands here --- .../core/chaos/test_tui_gateway_turn_liveness.py | 13 +------------ 1 file changed, 1 insertion(+), 12 deletions(-) diff --git a/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py b/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py index 5a1850d76f..c5b6c226a0 100644 --- a/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py +++ b/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py @@ -456,18 +456,7 @@ def scenario_futures(request: pytest.FixtureRequest, tmp_path_factory: pytest.Te fut.cancel() -# Real production bug on base (reported, not fixed here): when the gateway leaves mid-tool — -# client closes stdin or supervisor SIGTERMs — _shutdown_sessions() closes the agents but the -# in-flight foreground terminal command (its own process group) is never killed, so the -# `bash -c ...` + `sleep 3600` tree survives, reparented to init, and its tool_call is left with -# no result in state.db. strict: flips red once fixed. raises= names only the leftovers check's -# exception, so everything before it is asserted normally. -_ORPHANED_FOREGROUND_TOOL = pytest.mark.xfail( - strict=True, raises=ToolOutlivedGateway, - reason="tui_gateway exit (EOF/SIGTERM) orphans the running foreground terminal tool's process tree " - "and leaves its tool_call without a result") -KNOWN_BUGS = {"stdin_eof_during_hung_tool": _ORPHANED_FOREGROUND_TOOL, - "sigterm_during_hung_tool": _ORPHANED_FOREGROUND_TOOL} +KNOWN_BUGS: dict = {} @pytest.mark.parametrize("scn", [ From 4441dd34a52a4a00f7634207ed5c8dbcb398a53b Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:39:57 -0700 Subject: [PATCH 016/112] fix(tui-gateway): an ignored SIGINT no longer fences every later terminal command A server started as `cmd &` from a non-interactive shell inherits SIGINT as SIG_IGN. The exit-flush handler still ran _stop_turns_before_exit, which raises the one-way foreground-spawn fence, then chained to the ignored disposition and kept running: every later terminal command returned "[host is exiting: command not started]" rc 130. An ignored signal ends nothing, so the handler now returns before flushing or stopping turns. --- tests/tui_gateway/test_serve_exit_flush.py | 21 +++++++++++++++++++++ tui_gateway/session_reaper.py | 10 +++++++--- 2 files changed, 28 insertions(+), 3 deletions(-) diff --git a/tests/tui_gateway/test_serve_exit_flush.py b/tests/tui_gateway/test_serve_exit_flush.py index 44e7278309..7ea8976e9c 100644 --- a/tests/tui_gateway/test_serve_exit_flush.py +++ b/tests/tui_gateway/test_serve_exit_flush.py @@ -113,6 +113,27 @@ def test_sigterm_flushes_populated_session_into_state_db( assert any("survive the kill" in str(r.get("content", "")) for r in rows) +def test_ignored_sigint_leaves_terminal_commands_runnable(): + """SIGINT inherited as SIG_IGN (a server started as ``cmd &`` from a non-interactive shell) ends + nothing, so it must not raise the one-way exit fence: the process lives on and every later + terminal command would return 'host is exiting' rc 130.""" + from tools.environments.local import LocalEnvironment + + prev = {signal.SIGTERM: signal.getsignal(signal.SIGTERM), + signal.SIGINT: signal.signal(signal.SIGINT, signal.SIG_IGN)} + try: + assert server.install_exit_flush_signal_handlers() is True + os.kill(os.getpid(), signal.SIGINT) + time.sleep(0.05) + finally: + _restore_signal_state(prev) + env = LocalEnvironment(cwd=os.getcwd()) + try: + assert env.execute("echo still-alive", timeout=30)["returncode"] == 0 + finally: + env.cleanup() + + def test_exit_flush_is_bounded(registered_session): """A hung persist must never block exit longer than the budget.""" diff --git a/tui_gateway/session_reaper.py b/tui_gateway/session_reaper.py index 21e92dad12..8afc252a8d 100644 --- a/tui_gateway/session_reaper.py +++ b/tui_gateway/session_reaper.py @@ -151,17 +151,21 @@ _exit_flush_handlers_installed = False def _handle_exit_flush_signal(signum, frame) -> None: """Flush in-memory sessions, then hand off to the prior handler (uvicorn's graceful shutdown, a supervisor's handler, or the default disposition) — this only *prepends* a bounded flush.""" + import signal as _signal + prev = _exit_flush_prev_handlers.get(signum) + if prev is _signal.SIG_IGN: + # An inherited ignore (`cmd &` from a non-interactive shell) ends nothing: stopping turns here + # would raise the one-way exit fence in a process that keeps running and refuses every command. + return with contextlib.suppress(Exception): _flush_sessions_before_exit() # The group signal that stopped us never reaches a command in its own session: reap it now, # before a supervisor's SIGKILL can cut the graceful shutdown (and its atexit) short. with contextlib.suppress(Exception): _stop_turns_before_exit() - import signal as _signal - prev = _exit_flush_prev_handlers.get(signum) if callable(prev): prev(signum, frame) - elif prev is not _signal.SIG_IGN: + else: # Default disposition: restore it and re-raise so the process dies with the correct signal (exit status # visible to supervisors). try: From 37169f1dab62b59957b862da022931acdbdadbfc Mon Sep 17 00:00:00 2001 From: brooklyn! Date: Wed, 23 Sep 2026 19:21:19 -0500 Subject: [PATCH 017/112] fix(desktop): retire sealed live bubbles from any turn whose fold carries them A tool-using turn streams as several sealed bubbles but hydrates as one folded row. The post-turn reconcile only matched bubbles of the latest turn, only against text parts, and only when settled, so three shapes fell through to the tail: - Codex Responses commentary stored in `reasoning` (#119716): no text part ever matched, and the progress updates with their tool calls landed below the final reply. - An earlier turn's bubbles after a new prompt, when no submit receipt had moved the acknowledged boundary past that turn yet: they resurfaced under the newer prompt (#119511, the stale commentary tail in #119362). - Narration still marked pending when the rehydrate landed (#118228). Locate a bubble's folds by its own turn: rows sharing a tool call id with that turn (durable identity), plus the rows after its prompt's stored twin. Accept its text from a text part or as whole lines of Thinking, and treat a sealed interim as final even while pending. The latest turn keeps its positional fallback. A final answer the fold has not committed is still preserved. Co-authored-by: tian0cai Co-authored-by: xpk84 <36138499+xpk84@users.noreply.github.com> Co-authored-by: Benjamin Brumbaugh --- .../hooks/use-session-actions/utils.test.ts | 122 ++++++++++++++++++ .../hooks/use-session-actions/utils.ts | 97 ++++++++++---- 2 files changed, 193 insertions(+), 26 deletions(-) diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts b/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts index 63e320ba2f..dcd9877659 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts @@ -1231,6 +1231,128 @@ describe('preserveLocalPendingTurnMessages', () => { ) }) + // A Codex Responses turn: an acknowledgement, two progress updates between + // tool rounds, then the answer. Live, each seals as its own bubble; history + // folds them into one row and may keep the public commentary only in + // `reasoning` (#119716). Tool call ids are the durable identity either way. + const tool = (toolCallId: string) => + ({ type: 'tool-call', toolCallId, toolName: 'terminal', result: 'ok' }) as ChatMessagePart + + const sealed = (id: string, parts: ChatMessagePart[], extra: Partial = {}) => + ({ id, role: 'assistant', parts, pending: false, interim: true, ...extra }) as ChatMessage + + const lunaTurn = (prefix: string, callPrefix: string) => [ + sealed(`assistant-stream-${prefix}-ack`, [{ type: 'text', text: `${prefix}: on it, reading the logs.` }]), + sealed(`assistant-stream-${prefix}-progress-1`, [ + tool(`${callPrefix}-1`), + { type: 'text', text: `${prefix}: logs clean.` } + ]), + sealed(`assistant-stream-${prefix}-progress-2`, [ + tool(`${callPrefix}-2`), + { type: 'text', text: `${prefix}: config fixed.` } + ]), + sealed( + `assistant-stream-${prefix}-final`, + [tool(`${callPrefix}-3`), { type: 'text', text: `${prefix}: all done.` }], + { + interim: false + } + ) + ] + + const lunaFold = (id: string, prefix: string, callPrefix: string, commentary: 'reasoning' | 'text') => + ({ + id, + role: 'assistant', + parts: [ + ...[`${prefix}: on it, reading the logs.`, `${prefix}: logs clean.`, `${prefix}: config fixed.`].flatMap( + (text, at) => [ + commentary === 'text' + ? ({ type: 'text', text } as ChatMessagePart) + : ({ type: 'reasoning', text: `**Plan**\n\n${text}` } as ChatMessagePart), + tool(`${callPrefix}-${at + 1}`) + ] + ), + { type: 'text', text: `${prefix}: all done.` } + ] + }) as ChatMessage + + it.each(['reasoning', 'text'] as const)( + 'retires every sealed bubble of a folded turn whose commentary hydrated as %s', + commentary => { + const user = msg('1-user', 'user', 'fix it', { rowId: 1 }) + const next = [user, lunaFold('2-assistant', 'a', 'call-a', commentary)] + + expect(preserveLocalPendingTurnMessages(next, [user, ...lunaTurn('a', 'call-a')])).toBe(next) + } + ) + + // The previous turn's bubbles are not owned by the newest prompt, so they + // must not resurface under it (#119511, and the self-sustaining tail of + // stale commentary in #119362) — nor may they swallow the live reply. + it.each(['reasoning', 'text'] as const)( + 'does not re-append an earlier turn under a newer prompt when its commentary hydrated as %s', + commentary => { + const next = [ + msg('1-user', 'user', 'fix it', { rowId: 1 }), + lunaFold('2-assistant', 'a', 'call-a', commentary), + msg('3-user', 'user', 'and the other one', { rowId: 9 }) + ] + + const previous = [ + msg('user-1-a', 'user', 'fix it', { rowId: 1 }), + ...lunaTurn('a', 'call-a'), + // No submit receipt yet, so no acknowledged boundary past turn a. + msg('user-2-b', 'user', 'and the other one'), + sealed('assistant-stream-live', [tool('call-b-1'), { type: 'text', text: 'b: still going.' }]) + ] + + expect(preserveLocalPendingTurnMessages(next, previous).map(message => message.id)).toEqual([ + '1-user', + '2-assistant', + '3-user', + 'assistant-stream-live' + ]) + } + ) + + // #118228: narration bubbles still marked pending when the rehydrate lands. + // A sealed interim's text is final, so the fold carrying it retires it. + it('retires pending interim narration the merged fold already carries, even with later turns stored', () => { + const user = msg('1-user', 'user', 'run the build', { rowId: 1 }) + + // More live bubbles than stored assistant rows: ordinal pairing runs out. + const turn = [ + ...lunaTurn('a', 'call-a') + .slice(0, 3) + .map(row => ({ ...row, pending: true })), + msg('assistant-stream-a-tail', 'assistant', 'a: all done.', { pending: true }) + ] + + const next = [ + user, + lunaFold('2-assistant', 'a', 'call-a', 'text'), + msg('3-system', 'system', 'Background Process Finished: bash build.sh'), + msg('4-assistant', 'assistant', 'build verified'), + msg('5-user', 'user', 'installed it, same problem', { rowId: 20 }), + msg('6-assistant', 'assistant', 'then it is not the line count') + ] + + expect(preserveLocalPendingTurnMessages(next, [user, ...turn])).toBe(next) + }) + + // The fold committed the tool rounds but not the answer yet: that bubble is + // the only copy and must survive, while the carried commentary retires. + it('keeps the final answer a fold has not committed yet', () => { + const user = msg('1-user', 'user', 'fix it', { rowId: 1 }) + const fold = lunaFold('2-assistant', 'a', 'call-a', 'reasoning') + const next = [user, { ...fold, parts: fold.parts.filter(part => part.type !== 'text') }] + + expect( + preserveLocalPendingTurnMessages(next, [user, ...lunaTurn('a', 'call-a')]).map(message => message.id) + ).toEqual(['1-user', '2-assistant', 'assistant-stream-a-final']) + }) + // The whole point of replacing rather than appending: one reply on screen, // and the committed history around the live turn untouched. it('does not duplicate or rewrite committed history around the live turn', () => { diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts index 2b7fc460a1..4014f3c05a 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts @@ -596,19 +596,71 @@ const textPartsOf = (message: ChatMessage) => return text ? [text] : [] }) +const isPrompt = (message: ChatMessage) => message.role === 'user' && !isGatewaySystemMarker(message) + +/** + * Committed rows folding the local turn around `index`, for ANY turn, not just + * the latest: rows sharing a tool call id with that turn, plus the assistant + * rows after its prompt's durable twin. Only the latest turn may fall back to + * position (everything after the last stored prompt). + */ +function committedFoldsOfLocalTurn(candidates: ChatMessage[], previous: ChatMessage[], index: number): ChatMessage[] { + const start = previous.findLastIndex((row, at) => at < index && isPrompt(row)) + const end = previous.findIndex((row, at) => at > index && isPrompt(row)) + + const turnToolIds = new Set( + previous + .slice(start + 1, end < 0 ? undefined : end) + .flatMap(toolCallIdsOf) + .filter(Boolean) + ) + + const owner = previous[start] + const ownerRowIds = owner ? transcriptRowIds(owner) : [] + + const anchor = owner + ? candidates.findIndex(row => row.id === owner.id || transcriptRowIds(row).some(id => ownerRowIds.includes(id))) + : -1 + + const from = anchor >= 0 ? anchor : end < 0 ? candidates.findLastIndex(isPrompt) : candidates.length + const until = candidates.findIndex((row, at) => at > from && isPrompt(row)) + const segment = new Set(candidates.slice(from + 1, until < 0 ? undefined : until)) + + return candidates.filter( + row => + row.role === 'assistant' && + !isLiveTailRow(row) && + (segment.has(row) || toolCallIdsOf(row).some(id => turnToolIds.has(id))) + ) +} + +const hasWholeLines = (haystack: string, needle: string) => `\n${haystack}\n`.includes(`\n${needle}\n`) + +/** + * A fold carries sealed text verbatim as a text part, or inside Thinking when + * the provider stored public commentary in `reasoning` (Codex Responses, #119716). + */ +const foldCarriesText = (fold: ChatMessage, text: string) => + fold.parts.some(part => + part.type === 'text' + ? textWithoutReferenceLines(part.text).trim() === text + : part.type === 'reasoning' && hasWholeLines(part.text, text) + ) + /** * History folds a tool-heavy turn into one bubble, while the live stream sealed - * each interim segment and the final answer as bubbles of their own. A settled - * live bubble is that same occurrence when one committed fold holds every tool + * each interim segment and the final answer as bubbles of their own. A sealed + * live bubble is that same occurrence when its turn's folds hold every tool * call it ran (durable identity) and its text, either verbatim (a sealed middle * segment, #119540) or as the final answer, equal or extended (#118670). This * holds with a partial or missing completion receipt, where full-bubble * equality sees neither. */ -function durableFoldCoversLiveResponse(messages: ChatMessage[], live: ChatMessage): boolean { +function durableFoldCoversLiveResponse(folds: ChatMessage[], live: ChatMessage): boolean { const liveToolIds = toolCallIdsOf(live) + const sealed = live.pending !== true || live.interim === true - if (liveToolIds.length && (live.pending === true || liveToolIds.some(id => !id))) { + if (!folds.length || (liveToolIds.length && (!sealed || liveToolIds.some(id => !id)))) { return false } @@ -622,29 +674,24 @@ function durableFoldCoversLiveResponse(messages: ChatMessage[], live: ChatMessag return false } - const lastUser = messages.findLastIndex(message => message.role === 'user' && !isGatewaySystemMarker(message)) + const foldedToolIds = new Set(folds.flatMap(toolCallIdsOf)) - return messages.slice(lastUser + 1).some(message => { - if (message.role !== 'assistant' || isLiveTailRow(message)) { - return false - } + if (!liveToolIds.every(id => foldedToolIds.has(id))) { + return false + } - const foldedToolIds = new Set(toolCallIdsOf(message)) + if (sealed && liveTexts.every(text => folds.some(fold => foldCarriesText(fold, text)))) { + return true + } - if (!liveToolIds.every(id => foldedToolIds.has(id))) { - return false - } + return ( + Boolean(answer) && + folds.some(fold => { + const folded = lastFoldedResponseText(fold) - const foldedTexts = new Set(textPartsOf(message)) - - if (live.pending !== true && liveTexts.every(text => foldedTexts.has(text))) { - return true - } - - const folded = lastFoldedResponseText(message) - - return Boolean(answer) && (folded === answer || isStrictAnswerTextExtension(folded, answer)) - }) + return folded === answer || isStrictAnswerTextExtension(folded, answer) + }) + ) } export function preserveLocalPendingTurnMessages( @@ -715,7 +762,6 @@ export function preserveLocalPendingTurnMessages( // Authoritative id → richer local pending row. Replacing (not appending) // avoids painting both the empty inflight shell and the full stream bubble. const replacements = new Map() - const lastPreviousUser = previousMessages.findLastIndex(row => row.role === 'user' && !isGatewaySystemMarker(row)) let crossedUserBoundary = false for (const [index, message] of previousMessages.entries()) { @@ -902,8 +948,7 @@ export function preserveLocalPendingTurnMessages( if ( isPendingAssistant && - previousMessages.indexOf(message) > lastPreviousUser && - durableFoldCoversLiveResponse(candidates, message) + durableFoldCoversLiveResponse(committedFoldsOfLocalTurn(candidates, previousMessages, index), message) ) { continue } From b066034eaf691d7d0acbd36e068fb59a4bc631cf Mon Sep 17 00:00:00 2001 From: Xipong Date: Thu, 24 Sep 2026 03:30:45 +0300 Subject: [PATCH 018/112] fix(desktop): preserve interim messages across history and tool boundaries (#107386) * fix(desktop): preserve interim commentary across history and tool boundaries * fix(desktop): preserve interim commentary across history and tool boundaries * fix(desktop): keep public Codex commentary out of hydrated thinking Project profile-scoped, sanitized display commentary and reasoning for REST and gateway history without changing persisted replay items. Preserve canonical finals across stream recovery and keep tool-delimited interim text through settlement. --------- Co-authored-by: Xipong <217837358+Xipong@users.noreply.github.com> --- agent/history_commentary.py | 176 ++++++ agent/stream_delivery.py | 5 +- .../interim-history.test.tsx | 110 ++++ ...ages.codex-commentary-reprojection.test.ts | 251 +++++++++ .../chat-messages.codex-json-sidecar.test.ts | 13 +- ...t-messages.codex-sidecar-hydration.test.ts | 5 +- ...chat-messages.interim-preservation.test.ts | 251 +++++++++ apps/desktop/src/lib/chat-messages.test.ts | 11 +- .../src/lib/chat-messages/hydration.ts | 96 ++-- apps/desktop/src/lib/chat-messages/parts.ts | 41 +- apps/desktop/src/types/hermes.ts | 4 + hermes_cli/web_routers/sessions.py | 21 +- .../test_history_commentary_display.py | 504 ++++++++++++++++++ tui_gateway/compute_host.py | 2 +- tui_gateway/methods_session.py | 12 +- tui_gateway/methods_slash.py | 6 +- tui_gateway/server.py | 2 +- tui_gateway/session_history.py | 6 +- 18 files changed, 1442 insertions(+), 74 deletions(-) create mode 100644 agent/history_commentary.py create mode 100644 apps/desktop/src/app/session/hooks/use-message-stream/interim-history.test.tsx create mode 100644 apps/desktop/src/lib/chat-messages.codex-commentary-reprojection.test.ts create mode 100644 apps/desktop/src/lib/chat-messages.interim-preservation.test.ts create mode 100644 tests/hermes_cli/test_history_commentary_display.py diff --git a/agent/history_commentary.py b/agent/history_commentary.py new file mode 100644 index 0000000000..578ddc00eb --- /dev/null +++ b/agent/history_commentary.py @@ -0,0 +1,176 @@ +"""Display-only Codex commentary projection shared by REST and gateway history. + +Provider items and the stored reasoning string remain untouched for model replay. +Only the projected fields may be used for public transcript text. +""" + +import json +from contextlib import contextmanager +from typing import Any + +from agent.redact import redact_sensitive_text +from hermes_constants import reset_hermes_home_override, set_hermes_home_override +from utils import is_truthy_value + + +@contextmanager +def _owning_home(home): + token = set_hermes_home_override(str(home)) if home is not None else None + try: + yield + finally: + if token is not None: + reset_hermes_home_override(token) + + +def visible_commentary(text: str, *, strip_thinking=None) -> str: + """Use the same think stripping and secret redaction as live delivery.""" + if strip_thinking is None: + from agent.agent_runtime_helpers import strip_think_blocks + + strip_thinking = lambda value: strip_think_blocks(None, value) + + visible = strip_thinking(text).strip() + return redact_sensitive_text(visible) if visible else visible + + +def _phase_message_items(message: dict, phases: frozenset[str | None]) -> list[str]: + items = message.get("codex_message_items") + if isinstance(items, str): + try: + items = json.loads(items) + except (TypeError, ValueError): + return [] + if not isinstance(items, list): + return [] + result = [] + for item in items: + if ( + not isinstance(item, dict) + or item.get("type") != "message" + or item.get("role") != "assistant" + ): + continue + phase, content = item.get("phase"), item.get("content") + if phase is not None and not isinstance(phase, str): + continue + normalized_phase = phase.strip().lower() if isinstance(phase, str) else None + if normalized_phase not in phases or not isinstance(content, list): + continue + text = "".join( + part["text"] + for part in content + if isinstance(part, dict) + and part.get("type") == "output_text" + and isinstance(part.get("text"), str) + and part["text"].strip() + ).strip() + if text: + result.append(text) + return result + + +def _commentary_items(message: dict) -> list[str]: + return _phase_message_items(message, frozenset({"commentary"})) + + +def _final_items(message: dict) -> list[str]: + # Unphased assistant messages are normalized as final content too. + return _phase_message_items(message, frozenset({"final", "final_answer", None})) + + +def _without_flattened_commentary(reasoning: str, commentary: list[str]) -> str: + """Omit exact whole-line public segments from a display copy, not source data. + + Current normalization separates parts with two newlines; older rows may + have one. If a private segment happens to be identical, origin is ambiguous: + omit both occurrences rather than leaking the public text or choosing the + wrong one. Surrounding reasoning is retained in its original line style. + """ + remaining = reasoning + for text in sorted(set(commentary), key=len, reverse=True): + search_from = 0 + while (at := remaining.find(text, search_from)) >= 0: + end = at + len(text) + if (at == 0 or remaining[at - 1] == "\n") and ( + end == len(remaining) or remaining[end] == "\n" + ): + before, after = remaining[:at], remaining[end:] + prefix, suffix = before.rstrip("\n"), after.lstrip("\n") + separator = ( + "\n\n" + if before.endswith("\n\n") or after.startswith("\n\n") + else "\n" + ) + remaining = prefix + (separator if prefix and suffix else "") + suffix + search_from = 0 + else: + search_from = end + return remaining + + +def _project_one(message: dict, *, enabled: bool) -> dict: + if message.get("role") != "assistant" or message.get("display_kind") == "hidden": + return message + raw = _commentary_items(message) + if not raw: + return message + projected = dict(message) + # A disabled profile must not recover raw sidecar text on the frontend. + projected["display_commentary"] = [] + if enabled: + projected["display_commentary"] = [ + part for text in raw if (part := visible_commentary(text)) + ] + reasoning = ( + message.get("reasoning") + or message.get("reasoning_content") + or message.get("reasoning_details") + or "" + ) + if isinstance(reasoning, str): + projected["display_reasoning"] = _without_flattened_commentary(reasoning, raw) + + # A nonempty canonical answer can also be filled by stream recovery after + # the provider sidecar was captured. Without a matching final item there + # is no way to distinguish "commentary + final" from a real final that + # merely starts with the same words. Never delete canonical answer bytes. + # Sanitize the display copy when it contains raw commentary and avoid + # duplicating a leading commentary item in a separate Desktop bubble. + content = message.get( + "display_content", message.get("content", message.get("text")) + ) + if isinstance(content, str) and any(text in content for text in raw): + finals = _final_items(message) + if not (finals and content.strip() == "\n".join(finals)): + projected["display_content"] = visible_commentary(content) + if any(content.startswith(text) for text in raw): + projected["display_commentary"] = [] + return projected + + +def project_history_commentary(messages: list[dict], *, home: Any = None) -> list[dict]: + """Project a batch inside the owning profile's config and redaction scope.""" + if not any( + isinstance(message, dict) and message.get("codex_message_items") + for message in messages + ): + return messages + with _owning_home(home): + from hermes_cli.config import load_config + + try: + display = load_config().get("display") or {} + enabled = is_truthy_value( + display.get("show_commentary"), default=True + ) and is_truthy_value( + display.get("interim_assistant_messages"), default=True + ) + except Exception: + enabled = False # unreadable policy must not publish raw provider items + return [ + _project_one(message, enabled=enabled) + if isinstance(message, dict) + else message + for message in messages + ] diff --git a/agent/stream_delivery.py b/agent/stream_delivery.py index 13a8af8eba..2d88432e64 100644 --- a/agent/stream_delivery.py +++ b/agent/stream_delivery.py @@ -10,7 +10,7 @@ from typing import Any, Dict, List from agent.memory_manager import sanitize_context from agent.message_content import flatten_message_text -from agent.redact import redact_sensitive_text +from agent.history_commentary import visible_commentary # Same logger name as the origin module so log records / caplog filters are unchanged. logger = logging.getLogger("run_agent") @@ -135,8 +135,7 @@ class StreamDeliveryMixin: def _visible_commentary(self, text: str) -> str: """Think-stripped, redacted commentary text ("" when nothing visible remains).""" - visible = self._strip_think_blocks(text).strip() - return redact_sensitive_text(visible) if visible else visible + return visible_commentary(text, strip_thinking=getattr(self, "_strip_think_blocks")) def _extract_codex_interim_visible_text(self, assistant_msg: Dict[str, Any]) -> str: """All visible Codex commentary joined, for comparison/fallback.""" diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/interim-history.test.tsx b/apps/desktop/src/app/session/hooks/use-message-stream/interim-history.test.tsx new file mode 100644 index 0000000000..24993107fb --- /dev/null +++ b/apps/desktop/src/app/session/hooks/use-message-stream/interim-history.test.tsx @@ -0,0 +1,110 @@ +import type { GatewayEvent } from '@hermes/shared' +import { act, cleanup } from '@testing-library/react' +import { afterEach, describe, expect, it } from 'vitest' + +import { chatMessageText, toChatMessages } from '@/lib/chat-messages' +import type { SessionMessage } from '@/types/hermes' + +import { renderMessageStream } from './test-harness' + +const SID = 'interim-preservation' + +afterEach(cleanup) + +describe('intermediate assistant text survives Desktop lifecycle boundaries', () => { + it.each(['message.complete', 'message.interim'] as const)( + 'keeps pre-tool text when %s settles the next response', + async terminal => { + const stream = renderMessageStream(SID) + + const emit = async (type: GatewayEvent['type'], payload: GatewayEvent['payload']) => { + await act(() => stream.handleEvent({ type, payload, session_id: SID })) + } + + await emit('message.start', {}) + await emit('message.delta', { text: 'Checking the files.', timestamp: 1 }) + // Missing/disabled interim delivery must not make this older model response + // disposable when the later model response is finalized. + await emit('tool.start', { tool_id: 'call-1', name: 'terminal', args: { command: 'pwd' }, timestamp: 2 }) + await emit('tool.complete', { tool_id: 'call-1', name: 'terminal', result: 'ok', timestamp: 3 }) + await emit('message.delta', { text: 'Partial answer', timestamp: 4 }) + await emit(terminal, { text: 'The result is ready.', timestamp: 5 }) + const messages = stream.state().messages + const visible = messages.map(chatMessageText).join('\n') + expect(visible).toContain('Checking the files.') + expect(visible).toContain('The result is ready.') + expect(visible).not.toContain('Partial answer') + expect(messages.flatMap(message => message.parts).filter(part => part.type === 'tool-call')).toHaveLength(1) + expect(stream.state().busy).toBe(terminal === 'message.interim') + } + ) + + it.each(['rpc', 'rest'] as const)( + 'preserves the same public commentary on live → %s history projection', + async transport => { + const stream = renderMessageStream(SID) + + const emit = async (type: GatewayEvent['type'], payload: GatewayEvent['payload']) => { + await act(() => stream.handleEvent({ type, payload, session_id: SID })) + } + + await emit('message.start', {}) + await emit('reasoning.delta', { text: 'Real reasoning.', timestamp: 1 }) + await emit('message.interim', { text: 'I will inspect the files.', already_streamed: false, timestamp: 2 }) + await emit('tool.start', { tool_id: 'call-1', name: 'terminal', args: { command: 'pwd' }, timestamp: 3 }) + await emit('tool.complete', { tool_id: 'call-1', name: 'terminal', result: 'ok', timestamp: 4 }) + await emit('message.delta', { text: 'All checks passed.', timestamp: 5 }) + await emit('message.complete', { text: 'All checks passed.', timestamp: 6 }) + + const items = [ + { + type: 'message', + role: 'assistant', + phase: 'commentary', + content: [{ type: 'output_text', text: 'I will inspect the files.' }] + } + ] + + const rows: SessionMessage[] = [ + { + role: 'assistant', + content: '', + timestamp: 1, + reasoning: 'Real reasoning.\n\nI will inspect the files.', + display_reasoning: 'Real reasoning.', + display_commentary: ['I will inspect the files.'], + codex_message_items: transport === 'rest' ? JSON.stringify(items) : items, + tool_calls: [ + { id: 'call-1', type: 'function', function: { name: 'terminal', arguments: '{"command":"pwd"}' } } + ] + }, + { role: 'tool', content: 'ok', tool_call_id: 'call-1', name: 'terminal', timestamp: 4 }, + { role: 'assistant', content: 'All checks passed.', timestamp: 5 } + ] + + const texts = (messages: ReturnType) => + messages + .flatMap(message => message.parts.flatMap(part => (part.type === 'text' ? [part.text.trim()] : []))) + .filter(Boolean) + + const live = texts(stream.state().messages) + const restored = texts(toChatMessages(rows)) + expect(live).toEqual(['I will inspect the files.', 'All checks passed.']) + expect(restored).toEqual(live) + expect(restored.join('')).not.toContain('Real reasoning.') + + const restoredReasoning = toChatMessages(rows) + .flatMap(message => message.parts) + .filter(part => part.type === 'reasoning') + .map(part => part.text) + + expect(restoredReasoning).toEqual(['Real reasoning.']) + + expect( + toChatMessages(rows) + .flatMap(message => message.parts) + .filter(part => part.type === 'tool-call') + ).toHaveLength(1) + } + ) +}) diff --git a/apps/desktop/src/lib/chat-messages.codex-commentary-reprojection.test.ts b/apps/desktop/src/lib/chat-messages.codex-commentary-reprojection.test.ts new file mode 100644 index 0000000000..ca17143370 --- /dev/null +++ b/apps/desktop/src/lib/chat-messages.codex-commentary-reprojection.test.ts @@ -0,0 +1,251 @@ +import { describe, expect, it } from 'vitest' + +import { chatMessageText, toChatMessages } from '@/lib/chat-messages' +import type { SessionMessage } from '@/types/hermes' + +describe('public Codex commentary after transcript hydration', () => { + it('keeps a tool-call preamble in assistant text, not the Thinking disclosure', () => { + // Real row shape: the Responses adapter persists a phase=commentary item + // alongside an empty content column and a reasoning blob containing both + // the genuine summary and the already-delivered public preamble. + const preamble = 'I will inspect the repository before changing code.' + + const row: SessionMessage = { + id: 310, + role: 'assistant', + content: '', + reasoning: `**Checking workflow docs**\n\n${preamble}`, + display_reasoning: '**Checking workflow docs**', + display_commentary: [preamble], + tool_calls: [{ id: 'call-read', type: 'function', function: { name: 'read_file', arguments: '{}' } }], + codex_message_items: [ + { + type: 'message', + role: 'assistant', + phase: 'commentary', + content: [{ type: 'output_text', text: preamble }] + } + ], + timestamp: 2 + } + + const [message] = toChatMessages([row]) + const thoughts = message.parts.filter(part => part.type === 'reasoning') + + expect(chatMessageText(message)).toBe(preamble) + expect(thoughts).toHaveLength(1) + expect(thoughts[0]).toMatchObject({ text: '**Checking workflow docs**' }) + }) + + it('removes only exact commentary segments, preserving surrounding and incidental reasoning', () => { + const commentary = 'Inspect the files.\n\nThen run tests.' + + const row: SessionMessage = { + id: 311, + role: 'assistant', + content: 'Final reply.', + reasoning: `Summary before.\n\n${commentary}\n\nSummary after.`, + display_reasoning: 'Summary before.\n\nSummary after.', + display_commentary: [commentary], + codex_message_items: [ + { + type: 'message', + role: 'assistant', + phase: 'commentary', + content: [{ type: 'output_text', text: commentary }] + }, + { type: 'message', role: 'assistant', phase: 'analysis', content: [{ type: 'output_text', text: 'Private.' }] } + ], + timestamp: 3 + } + + const [message] = toChatMessages([row]) + + expect(message.parts.filter(part => part.type === 'text').map(part => part.text)).toEqual([ + commentary, + 'Final reply.' + ]) + expect(message.parts.filter(part => part.type === 'reasoning').map(part => part.text)).toEqual([ + 'Summary before.\n\nSummary after.' + ]) + + const incidental: SessionMessage = { + ...row, + reasoning: `Summary mentioning ${commentary} as an example.`, + display_reasoning: `Summary mentioning ${commentary} as an example.` + } + + const [unchanged] = toChatMessages([incidental]) + + expect(unchanged.parts.filter(part => part.type === 'reasoning').map(part => part.text)).toEqual([ + incidental.reasoning + ]) + }) + + it('honors disabled interim commentary without suppressing a separate final reply', () => { + const commentary = 'I will inspect the files.' + + const row: SessionMessage = { + id: 312, + role: 'assistant', + content: 'Here is the final result.', + reasoning: `Private summary.\n\n${commentary}`, + display_reasoning: 'Private summary.', + display_commentary: [], + codex_message_items: [ + { + type: 'message', + role: 'assistant', + phase: 'commentary', + content: [{ type: 'output_text', text: commentary }] + } + ], + timestamp: 4 + } + + const [message] = toChatMessages([row]) + + expect(chatMessageText(message)).toBe('Here is the final result.') + expect(message.parts.filter(part => part.type === 'reasoning').map(part => part.text)).toEqual(['Private summary.']) + }) + + it('uses only backend-authorized display commentary, never raw replay items', () => { + const raw = 'My key is «redacted:sk-…». private' + + const row: SessionMessage = { + id: 314, + role: 'assistant', + content: '', + reasoning: `Private summary.\n\n${raw}`, + display_reasoning: 'Private summary.', + display_commentary: ['My key is redacted.'], + codex_message_items: [ + { type: 'message', role: 'assistant', phase: 'commentary', content: [{ type: 'output_text', text: raw }] } + ], + timestamp: 5 + } + + expect(chatMessageText(toChatMessages([row])[0])).toBe('My key is redacted.') + expect( + toChatMessages([row])[0] + .parts.filter(part => part.type === 'reasoning') + .map(part => part.text) + ).toEqual(['Private summary.']) + expect(chatMessageText(toChatMessages([{ ...row, display_commentary: undefined }])[0])).toBe('') + }) + + it('keeps a sanitized canonical answer intact when its sidecar origin is ambiguous', () => { + const raw = 'Checking. hidden key=' + 'sk-' + 'demo0123456789abcdef' + + const row: SessionMessage = { + id: 315, + role: 'assistant', + content: `${raw}\n\nFinal result.`, + display_content: 'Checking. key=«redacted»\n\nFinal result.', + display_commentary: [], + reasoning: `Private.\n\n${raw}`, + display_reasoning: 'Private.', + codex_message_items: [ + { type: 'message', role: 'assistant', phase: 'commentary', content: [{ type: 'output_text', text: raw }] } + ], + timestamp: 6 + } + + const [shown] = toChatMessages([row]) + expect(shown.parts.filter(part => part.type === 'text').map(part => part.text)).toEqual([ + 'Checking. key=«redacted»\n\nFinal result.' + ]) + expect(shown.parts.filter(part => part.type === 'reasoning').map(part => part.text)).toEqual(['Private.']) + expect(chatMessageText(toChatMessages([{ ...row, display_commentary: [] }])[0])).toBe( + 'Checking. key=«redacted»\n\nFinal result.' + ) + }) + + it('keeps an explicit empty display content authoritative over a stale final sidecar', () => { + const row: SessionMessage = { + id: 316, + role: 'assistant', + content: 'Working', + display_content: '', + display_commentary: [], + reasoning: 'Private.\n\nWorking', + display_reasoning: 'Private.', + codex_message_items: [ + { + type: 'message', + role: 'assistant', + phase: 'commentary', + content: [{ type: 'output_text', text: 'Working' }] + }, + { + type: 'message', + role: 'assistant', + phase: 'final', + content: [{ type: 'output_text', text: 'Stale sidecar final' }] + } + ], + timestamp: 7 + } + + expect(chatMessageText(toChatMessages([row])[0])).toBe('') + expect( + toChatMessages([row])[0] + .parts.filter(part => part.type === 'reasoning') + .map(part => part.text) + ).toEqual(['Private.']) + // Older backends without a display projection still use final sidecars. + expect(chatMessageText(toChatMessages([{ ...row, display_content: undefined }])[0])).toBe('Working') + }) + + it('keeps a final answer that starts with or equals the commentary text', () => { + for (const final of ['Checking. The answer is 42.', 'Checking.']) { + const row: SessionMessage = { + id: 317, + role: 'assistant', + content: final, + display_commentary: ['Checking.'], + display_reasoning: 'Private.', + codex_message_items: [ + { + type: 'message', + role: 'assistant', + phase: 'commentary', + content: [{ type: 'output_text', text: 'Checking.' }] + }, + { type: 'message', role: 'assistant', phase: 'final', content: [{ type: 'output_text', text: final }] } + ], + timestamp: 8 + } + + const shown = toChatMessages([row])[0] + .parts.filter(part => part.type === 'text') + .map(part => part.text) + + expect(shown).toEqual(final === 'Checking.' ? [final] : ['Checking.', final]) + expect(chatMessageText(toChatMessages([{ ...row, display_commentary: [] }])[0])).toBe(final) + } + }) + + it('uses each message owner’s display projection, not a foreground global setting', () => { + const row: SessionMessage = { + id: 313, + role: 'assistant', + content: '', + reasoning: 'Checking files.', + codex_message_items: [ + { + type: 'message', + role: 'assistant', + phase: 'commentary', + content: [{ type: 'output_text', text: 'Checking files.' }] + } + ], + timestamp: 5 + } + + expect(chatMessageText(toChatMessages([{ ...row, display_commentary: [] }])[0])).toBe('') + expect(chatMessageText(toChatMessages([{ ...row, display_commentary: ['Checking files.'] }])[0])).toBe( + 'Checking files.' + ) + }) +}) diff --git a/apps/desktop/src/lib/chat-messages.codex-json-sidecar.test.ts b/apps/desktop/src/lib/chat-messages.codex-json-sidecar.test.ts index 984d193be1..88aeff32ae 100644 --- a/apps/desktop/src/lib/chat-messages.codex-json-sidecar.test.ts +++ b/apps/desktop/src/lib/chat-messages.codex-json-sidecar.test.ts @@ -30,9 +30,16 @@ it('hydrates the SQLite JSON-text sidecar returned by REST like the decoded RPC expect(toChatMessages([{ ...empty, codex_message_items: '{invalid' }])).toEqual([]) expect(toChatMessages([{ ...empty, display_kind: 'hidden', codex_message_items: JSON.stringify(items) }])).toEqual([]) - for (const phase of ['analysis', 'commentary']) { - expect(toChatMessages([{ ...empty, codex_message_items: JSON.stringify([{ ...items[0], phase }]) }])).toEqual([]) - } + expect( + toChatMessages([{ ...empty, codex_message_items: JSON.stringify([{ ...items[0], phase: 'analysis' }]) }]) + ).toEqual([]) + const rawCommentary = JSON.stringify([{ ...items[0], phase: 'commentary' }]) + expect(toChatMessages([{ ...empty, codex_message_items: rawCommentary }])).toEqual([]) + expect( + chatMessageText( + toChatMessages([{ ...empty, codex_message_items: rawCommentary, display_commentary: ['Durable reply'] }])[0] + ) + ).toBe('Durable reply') expect( chatMessageText( diff --git a/apps/desktop/src/lib/chat-messages.codex-sidecar-hydration.test.ts b/apps/desktop/src/lib/chat-messages.codex-sidecar-hydration.test.ts index 4429b6eab0..438d0797c6 100644 --- a/apps/desktop/src/lib/chat-messages.codex-sidecar-hydration.test.ts +++ b/apps/desktop/src/lib/chat-messages.codex-sidecar-hydration.test.ts @@ -31,6 +31,7 @@ describe('#68321 assistant rows whose reply persisted only in codex_message_item content: '', reasoning: null, reasoning_content: null, + display_commentary: ['Working through the approach...'], codex_message_items: [ { type: 'message', @@ -66,8 +67,8 @@ describe('#68321 assistant rows whose reply persisted only in codex_message_item expect(messages.map(m => m.role)).toEqual(['user', 'assistant', 'user']) // The final-answer text is painted as the bubble's reply text... expect(chatMessageText(messages[1])).toContain('Here is the full response you saw live.') - // ...and commentary / analysis narration (reasoning channel on the backend) is not. - expect(chatMessageText(messages[1])).not.toContain('Working through the approach...') + // ...and public commentary survives too, without promoting analysis. + expect(chatMessageText(messages[1])).toContain('Working through the approach...') expect(chatMessageText(messages[1])).not.toContain('Scratchpad thoughts.') }) }) diff --git a/apps/desktop/src/lib/chat-messages.interim-preservation.test.ts b/apps/desktop/src/lib/chat-messages.interim-preservation.test.ts new file mode 100644 index 0000000000..e0f64e70f1 --- /dev/null +++ b/apps/desktop/src/lib/chat-messages.interim-preservation.test.ts @@ -0,0 +1,251 @@ +import { describe, expect, it } from 'vitest' + +import type { SessionMessage } from '@/types/hermes' + +import { + assistantTextPart, + type ChatMessagePart, + chatMessageText, + mergeFinalAssistantText, + reasoningPart, + toChatMessages, + upsertToolPart +} from './chat-messages' + +const item = (phase: string, text: string) => ({ + type: 'message', + role: 'assistant', + phase, + content: [{ type: 'output_text', text }] +}) + +const textParts = (parts: ChatMessagePart[]) => parts.flatMap(part => (part.type === 'text' ? [part.text] : [])) + +const assistantText = (rows: SessionMessage[]) => + toChatMessages(rows) + .filter(message => message.role === 'assistant') + .map(chatMessageText) + .join('\n') + +const withTool = (parts: ChatMessagePart[], id = 'call-1', timestamp = 2) => + upsertToolPart( + parts, + { tool_id: id, name: 'terminal', args: { command: 'pwd' }, result: 'ok' }, + 'complete', + timestamp + ) + +describe('stored assistant commentary preservation', () => { + it.each(['rpc', 'rest'] as const)('restores commentary before the canonical reply through %s history', transport => { + const items = [ + item('commentary', 'I will inspect the files.'), + item('analysis', 'Private analysis.'), + item('final_answer', 'Stale sidecar final.') + ] + + const row: SessionMessage = { + role: 'assistant', + content: 'Canonical final.', + timestamp: 2, + reasoning: 'Real reasoning.', + display_commentary: ['I will inspect the files.'], + codex_message_items: transport === 'rest' ? JSON.stringify(items) : items + } + + const [message] = toChatMessages([row]) + expect(textParts(message.parts)).toEqual(['I will inspect the files.', 'Canonical final.']) + expect(chatMessageText(message)).not.toContain('Private analysis.') + expect(chatMessageText(message)).not.toContain('Stale sidecar final.') + expect(message.parts.filter(part => part.type === 'reasoning')).toEqual([reasoningPart('Real reasoning.', 2)]) + }) + + it.each(['rpc', 'rest'] as const)('keeps a commentary-only assistant row through %s history', transport => { + const items = [item('commentary', 'Still checking.')] + + const rows: SessionMessage[] = [ + { role: 'user', content: 'Check it.', timestamp: 1 }, + { + role: 'assistant', + content: '', + timestamp: 2, + display_commentary: ['Still checking.'], + codex_message_items: transport === 'rest' ? JSON.stringify(items) : items + }, + { role: 'user', content: 'Continue.', timestamp: 3 } + ] + + expect(toChatMessages(rows).map(message => message.role)).toEqual(['user', 'assistant', 'user']) + expect(assistantText(rows)).toBe('Still checking.') + }) + + it('keeps separate commentary items in order and joins only chunks of the same item', () => { + const first = item('commentary', 'First ') + first.content.push({ type: 'output_text', text: 'update.' }) + + const [message] = toChatMessages([ + { + role: 'assistant', + content: '', + timestamp: 2, + display_commentary: ['First update.', 'Second update.'], + codex_message_items: [first, item('commentary', 'Second update.'), item('final_answer', 'Final answer.')] + } + ]) + + expect(textParts(message.parts)).toEqual(['First update.', 'Second update.', 'Final answer.']) + }) + + it('keeps backend-authorized commentary without exposing analysis', () => { + expect( + assistantText([ + { + role: 'assistant', + content: '', + timestamp: 2, + display_commentary: ['Visible update.'], + codex_message_items: [ + item(' Commentary ', 'Visible update.'), + item(' ANALYSIS ', 'Hidden analysis.'), + item('final_answer', 'Answer.') + ] + } + ]) + ).toBe('Visible update.Answer.') + }) + + it('does not duplicate commentary also present in canonical content', () => { + const items = [item('commentary', 'First update.'), item('commentary', 'Second update.')] + expect( + assistantText([ + { + role: 'assistant', + content: 'First update.\n\nSecond update.', + timestamp: 2, + display_commentary: ['First update.', 'Second update.'], + codex_message_items: items + } + ]) + ).toBe('First update.\n\nSecond update.') + }) + + it('keeps canonical text authoritative when it equals one commentary item', () => { + expect( + assistantText([ + { + role: 'assistant', + content: 'Same text.', + timestamp: 2, + display_commentary: ['Same text.'], + codex_message_items: [item('commentary', 'Same text.'), item('final_answer', 'Stale text.')] + } + ]) + ).toBe('Same text.') + }) + + it('respects explicitly hidden and non-assistant rows', () => { + const codex_message_items = [item('commentary', 'Do not surface this.')] + expect( + toChatMessages([{ role: 'assistant', content: '', timestamp: 2, display_kind: 'hidden', codex_message_items }]) + ).toEqual([]) + expect(assistantText([{ role: 'user', content: 'User text.', timestamp: 2, codex_message_items }])).toBe('') + }) + + it.each(['{broken', 'null', '{}', '42'])( + 'ignores malformed sidecar %s without losing canonical content', + codex_message_items => { + expect(assistantText([{ role: 'assistant', content: 'Keep me.', timestamp: 2, codex_message_items }])).toBe( + 'Keep me.' + ) + } + ) + + it('does not promote a reasoning-only row into an answer', () => { + const [message] = toChatMessages([{ role: 'assistant', content: '', timestamp: 2, reasoning: 'Only reasoning.' }]) + expect(chatMessageText(message)).toBe('') + expect(message.parts).toEqual([reasoningPart('Only reasoning.', 2)]) + }) + + it('never promotes raw sidecar commentary without backend authorization', () => { + expect( + assistantText([ + { + role: 'assistant', + content: '', + timestamp: 2, + codex_message_items: JSON.stringify([ + null, + 1, + [], + { ...item('commentary', 'Foreign text.'), role: 'user' }, + { ...item('commentary', 'Wrong type.'), type: 'reasoning' }, + { ...item('commentary', 'Bad content.'), content: 'not an array' }, + { ...item('commentary', ''), content: [null, 1, [], { type: 'output_text', text: 42 }] }, + item('commentary', 'Valid update.') + ]) + } + ]) + ).toBe('') + }) +}) + +describe('authoritative final text is scoped to the latest tool-delimited response', () => { + it('does not erase commentary from before a tool when the interim frame is absent', () => { + const earlier = withTool([assistantTextPart('Checking the repository.', 1)]) + const parts = [...earlier, assistantTextPart('Partial final', 3)] + const before = structuredClone(parts) + const result = mergeFinalAssistantText(parts, 'Final answer.', 4) + expect(textParts(result)).toEqual(['Checking the repository.', 'Final answer.']) + expect(result.slice(0, earlier.length)).toEqual(earlier) + expect(parts).toEqual(before) + }) + + it('keeps all earlier tool-delimited updates, not just the latest one', () => { + const first = withTool([assistantTextPart('First update.', 1)]) + const second = withTool([...first, assistantTextPart('Second update.', 3)], 'call-2', 4) + const result = mergeFinalAssistantText([...second, assistantTextPart('Draft', 5)], 'Done.', 6) + expect(textParts(result)).toEqual(['First update.', 'Second update.', 'Done.']) + expect(result.filter(part => part.type === 'tool-call')).toHaveLength(2) + }) + + it('does not duplicate an exact cumulative final prefix while preserving the tool boundary', () => { + const parts = [...withTool([assistantTextPart('Earlier update.', 1)]), assistantTextPart('Partial', 3)] + const result = mergeFinalAssistantText(parts, 'Earlier update.Final answer.', 4) + expect(textParts(result)).toEqual(['Earlier update.', 'Final answer.']) + expect(result.findIndex(part => part.type === 'tool-call')).toBe(1) + }) + + it('keeps a longer earlier update when the final is only its short prefix', () => { + const parts = withTool([assistantTextPart('Done. The investigation details follow.', 1)]) + expect(textParts(mergeFinalAssistantText(parts, 'Done.', 3))).toEqual([ + 'Done. The investigation details follow.', + 'Done.' + ]) + }) + + it('drops a later provisional draft when the cumulative final equals the earlier response exactly', () => { + const earlier = withTool([assistantTextPart('Earlier update.', 1)]) + const parts = [...earlier, reasoningPart('Checking.', 3), assistantTextPart('Unfinished draft', 4)] + + const result = mergeFinalAssistantText(parts, 'Earlier update.', 5) + + expect(textParts(result)).toEqual(['Earlier update.']) + expect(result.slice(0, earlier.length)).toEqual(earlier) + expect(result.find(part => part.type === 'reasoning')?.text).toBe('Checking.') + }) + + it('still replaces provisional text within one response, even across reasoning parts', () => { + const parts = [ + assistantTextPart('Wrong draft.', 1), + reasoningPart('Reconsidering.', 2), + assistantTextPart('Another draft.', 3) + ] + + expect(textParts(mergeFinalAssistantText(parts, 'Correct final.', 4))).toEqual(['Correct final.']) + }) + + it('keeps a confirmed full stream and an empty terminal frame unchanged', () => { + const parts = [...withTool([assistantTextPart('Earlier.', 1)]), assistantTextPart('Final.', 3)] + expect(mergeFinalAssistantText(parts, 'Earlier.Final.', 4)).toBe(parts) + expect(mergeFinalAssistantText(parts, ' ', 4)).toBe(parts) + }) +}) diff --git a/apps/desktop/src/lib/chat-messages.test.ts b/apps/desktop/src/lib/chat-messages.test.ts index d9c18c0923..6b2be68a38 100644 --- a/apps/desktop/src/lib/chat-messages.test.ts +++ b/apps/desktop/src/lib/chat-messages.test.ts @@ -1173,7 +1173,9 @@ describe('mergeFinalAssistantText', () => { expect(result[0]).toMatchObject({ text: 'final answer', timestamp: 12.5, type: 'text' }) }) - it('removes all text parts and appends the final text', () => { + it('preserves pre-tool text and appends the later final response', () => { + // These deltas precede the tool call: they belong to an earlier model + // response, not to the provisional draft of the final being settled. const parts = [ { type: 'text' as const, text: 'streamed delta 1' }, { type: 'text' as const, text: 'streamed delta 2' }, @@ -1182,8 +1184,11 @@ describe('mergeFinalAssistantText', () => { const result = mergeFinalAssistantText(parts, 'final answer') - expect(result.filter(p => p.type === 'text')).toHaveLength(1) - expect(result.filter(p => p.type === 'text')[0]).toMatchObject({ text: 'final answer' }) + expect(result.filter(p => p.type === 'text').map(p => p.text)).toEqual([ + 'streamed delta 1', + 'streamed delta 2', + 'final answer' + ]) expect(result.some(p => p.type === 'tool-call')).toBe(true) }) diff --git a/apps/desktop/src/lib/chat-messages/hydration.ts b/apps/desktop/src/lib/chat-messages/hydration.ts index ecf79177ed..31e886a2ca 100644 --- a/apps/desktop/src/lib/chat-messages/hydration.ts +++ b/apps/desktop/src/lib/chat-messages/hydration.ts @@ -4,7 +4,14 @@ import { extractImageRefs } from '@/lib/embedded-images' import { dedupeGeneratedImageEchoesInParts } from '@/lib/generated-images' import type { MessageReaction, SessionMessage } from '@/types/hermes' -import { assistantTextPart, chatMessageText, dedupeRepeatedTextInParts, reasoningPart, textPart } from './parts' +import { + assistantTextPart, + chatMessageText, + dedupeRepeatedTextInParts, + reasoningPart, + renderMediaTags, + textPart +} from './parts' import { applyStoredToolResult, applyStoredToolResultToParts, @@ -26,29 +33,32 @@ const DISCORD_TRIGGERING_NOTE_RE = /(^|\n)\[Triggering message id: `[^`\n]*` — use as `message_id` for reply\/react\/pin via the discord tools\.\]\n*/ /** - * Reply text from a Responses-API `codex_message_items` sidecar (#68321), for rows - * whose `content` persisted empty. `commentary` / `analysis` items are mid-turn - * narration the backend routes to the reasoning channel - * (codex_responses_adapter `_OutputScan._message`); the remaining phases are the reply. + * Backend history projection authorizes/sanitizes public commentary before it + * reaches Desktop. Raw Responses sidecars are used only for final-answer fallback; + * phase=analysis and raw phase=commentary are never promoted to assistant text. */ -function codexMessageItemText(message: SessionMessage): string { +function codexMessageItemText(message: SessionMessage): { commentary: string[]; reply: string } { let items = message.codex_message_items + const commentary = Array.isArray(message.display_commentary) + ? message.display_commentary.filter((part): part is string => typeof part === 'string' && Boolean(part.trim())) + : [] + + const replies: string[] = [] + // REST carries SQLite JSON text; RPC history carries the decoded list. if (typeof items === 'string') { try { items = JSON.parse(items) } catch { - return '' + return { commentary, reply: '' } } } if (!Array.isArray(items)) { - return '' + return { commentary, reply: '' } } - const texts: string[] = [] - for (const item of items) { if (!item || typeof item !== 'object' || Array.isArray(item)) { continue @@ -60,37 +70,34 @@ function codexMessageItemText(message: SessionMessage): string { continue } - if (record.phase === 'commentary' || record.phase === 'analysis') { + const phase = typeof record.phase === 'string' ? record.phase.trim().toLowerCase() : '' + + if (phase === 'analysis' || phase === 'commentary' || !Array.isArray(record.content)) { continue } - const content = record.content + const chunks: string[] = [] - if (!Array.isArray(content)) { - continue - } - - for (const part of content) { + for (const part of record.content) { if (!part || typeof part !== 'object' || Array.isArray(part)) { continue } const partRecord = part as Record - const partType = partRecord.type - if (partType !== 'output_text' && partType !== 'text') { - continue + if ((partRecord.type === 'output_text' || partRecord.type === 'text') && typeof partRecord.text === 'string') { + chunks.push(partRecord.text) } + } - const text = partRecord.text + const text = chunks.join('') - if (typeof text === 'string' && text.length > 0) { - texts.push(text) - } + if (text) { + replies.push(text) } } - return texts.join('') + return { commentary, reply: replies.join('') } } function displayContentForMessage(role: SessionMessage['role'], content: unknown): string { @@ -368,31 +375,40 @@ export function toChatMessages(messages: SessionMessage[]): ChatMessage[] { const sourceHasTools = Array.isArray(message.tool_calls) && message.tool_calls.length > 0 const durableComplete = sourceHasTools ? false : rowId !== undefined ? true : undefined - const reasoning = + const codexText = + displayRole === 'assistant' && message.display_kind !== 'hidden' ? codexMessageItemText(message) : null + + const commentary = codexText?.commentary ?? [] + + const rawReasoning = message.reasoning || message.reasoning_content || (typeof message.reasoning_details === 'string' ? message.reasoning_details : '') + const reasoning = message.display_reasoning !== undefined ? message.display_reasoning : rawReasoning + if (reasoning && message.role === 'assistant') { parts.push(reasoningPart(reasoning, message.timestamp)) } - if (displayContent) { - parts.push( - displayRole === 'assistant' - ? assistantTextPart(displayContent, message.timestamp) - : textPart(displayContent, message.timestamp) - ) + const reply = message.display_content !== undefined ? displayContent : displayContent || codexText?.reply + // Some providers also persist the joined commentary as canonical content. + // Keep that authoritative copy once, without treating unrelated final text + // as a reason to discard the earlier public messages. + const normalized = (value: string) => renderMediaTags(value).replace(/\s+/g, ' ').trim() + + const commentaryIsReply = Boolean( + reply && commentary.length && normalized(commentary.join('\n\n')) === normalized(reply) + ) + + if (!commentaryIsReply) { + parts.push(...commentary.map(text => assistantTextPart(text, message.timestamp))) } - // Reply text can live only in the sidecar alongside reasoning or tool parts. - // Those parts are not a substitute for the answer; canonical content still wins. - if (message.role === 'assistant' && message.display_kind !== 'hidden' && !displayContent) { - const codexText = codexMessageItemText(message) - - if (codexText) { - parts.push(assistantTextPart(codexText, message.timestamp)) - } + if (reply) { + parts.push( + displayRole === 'assistant' ? assistantTextPart(reply, message.timestamp) : textPart(reply, message.timestamp) + ) } if (message.role === 'assistant' && Array.isArray(message.tool_calls)) { diff --git a/apps/desktop/src/lib/chat-messages/parts.ts b/apps/desktop/src/lib/chat-messages/parts.ts index eedf425e5e..144d7c53f7 100644 --- a/apps/desktop/src/lib/chat-messages/parts.ts +++ b/apps/desktop/src/lib/chat-messages/parts.ts @@ -266,8 +266,10 @@ export function dedupeRepeatedTextInParts(parts: ChatMessagePart[]): ChatMessage /** * Merge the final assistant text into a message's parts. * - * - Removes all existing `text` parts (they were streamed deltas, now superseded - * by the authoritative final response). + * - Preserves earlier tool-delimited responses: a missed interim frame must + * not make their public text disposable. + * - Replaces provisional text only in the latest response with its authoritative + * final text, retaining confirmed text/reasoning boundaries. * - Keeps `reasoning` parts, but drops one that the final text fully covers * (reasoning ⊆ final) — the final restates it. A short final ("Done.") must * NOT swallow a longer reasoning block that merely starts with it (#61447). @@ -300,14 +302,41 @@ export function mergeFinalAssistantText( return parts } + // A tool call is an explicit model-response boundary even when no + // message.interim frame sealed the earlier text into a separate bubble. + // Only the suffix after the last call belongs to this authoritative final. + const lastToolIndex = parts.findLastIndex(part => part.type === 'tool-call') + + if (lastToolIndex >= 0) { + const earlier = parts.slice(0, lastToolIndex + 1) + + const earlierText = earlier + .filter((part): part is Extract => part.type === 'text') + .map(part => part.text) + .join('') + + // Some terminal frames carry cumulative text. Strip only an exact prefix; + // fuzzy similarity is not proof that two assistant messages are the same. + const responseText = + earlierText && finalText.startsWith(earlierText) ? finalText.slice(earlierText.length) : finalText + + const suffix = parts.slice(lastToolIndex + 1) + + // A cumulative final can stop exactly at the pre-tool update. The suffix + // draft is still provisional; the ordinary empty-final path keeps drafts. + if (earlierText && finalText === earlierText) { + return [...earlier, ...suffix.filter(part => part.type !== 'text')] + } + + return [...earlier, ...mergeFinalAssistantText(suffix, responseText, fallbackTimestamp)] + } + const previousText = parts.findLast(part => part.type === 'text') const kept = parts.filter(part => { if (part.type === 'text') { - // Sealed text parts were already finalized into their own bubbles — - // this filter only runs on the LAST streaming bubble, so there are no - // sealed parts here. All text parts are streamed deltas that get - // replaced by the authoritative final text. + // The tool-delimited prefix was retained above. This suffix is + // provisional text from the response being finalized. return false } diff --git a/apps/desktop/src/types/hermes.ts b/apps/desktop/src/types/hermes.ts index efc0cde6df..6a91ec97de 100644 --- a/apps/desktop/src/types/hermes.ts +++ b/apps/desktop/src/types/hermes.ts @@ -628,6 +628,10 @@ export interface SessionMessage { content: unknown /** Backend-projected user-visible content when a physical row also carries internal model scaffolding. */ display_content?: unknown + /** Sanitized, profile-authorized public commentary supplied by the history backend. Never recover this from raw replay. */ + display_commentary?: string[] + /** Display-only reasoning after removing exact public commentary; stored reasoning remains unmodified. */ + display_reasoning?: string context?: unknown name?: string reasoning?: null | string diff --git a/hermes_cli/web_routers/sessions.py b/hermes_cli/web_routers/sessions.py index 97a852b22a..68d47f9cb7 100644 --- a/hermes_cli/web_routers/sessions.py +++ b/hermes_cli/web_routers/sessions.py @@ -551,9 +551,20 @@ def _with_tool_call_labels(message: dict) -> dict: return {**message, "tool_call_labels": labels} if labels else message -def _project_for_display(messages: list) -> list: +def _history_profile_home(profile): + if profile: + return _cron_profile_home(profile)[1] + # An omitted profile reads this process's DB (including custom HERMES_HOME), + # not necessarily the registered default profile used by cron routes. + from hermes_cli.config import get_hermes_home + + return get_hermes_home() + + +def _project_for_display(messages: list, *, home=None) -> list: from agent.compaction_display import project_compaction_message_for_display from agent.context_compressor import is_compaction_summary_message + from agent.history_commentary import project_history_commentary from agent.turn_failure_copy import untyped_failed_turn_display_kind projected_messages = [] @@ -579,7 +590,7 @@ def _project_for_display(messages: list) -> list: projected["display_content"] = display_view.get("content") projected.pop("display_kind", None) projected_messages.append(projected) - return projected_messages + return project_history_commentary(projected_messages, home=home) @manage_router.get("/api/sessions/{session_id}/messages") @@ -608,7 +619,8 @@ async def get_session_messages( if result is None: raise HTTPException(status_code=404, detail=_NOT_FOUND) sid, _limit, messages = result - projected_messages = _project_for_display(messages) + projected_messages = await asyncio.to_thread( + _project_for_display, messages, home=_history_profile_home(profile)) return { "session_id": sid, # The same stamp list rows carry, so the Desktop keys a page under the @@ -676,7 +688,8 @@ async def get_session_messages_around( return {"session_id": sid, "profile": owner, **page} result = await asyncio.to_thread(_with_db, profile, _read, read_only=True) - result["messages"] = _project_for_display(result["messages"]) + result["messages"] = await asyncio.to_thread( + _project_for_display, result["messages"], home=_history_profile_home(profile)) return result diff --git a/tests/hermes_cli/test_history_commentary_display.py b/tests/hermes_cli/test_history_commentary_display.py new file mode 100644 index 0000000000..7bccd70440 --- /dev/null +++ b/tests/hermes_cli/test_history_commentary_display.py @@ -0,0 +1,504 @@ +"""Public Codex commentary is a display projection, never a replay mutation.""" + +from hermes_cli.web_routers.sessions import _project_for_display +from tui_gateway.server import _history_to_messages +from agent.history_commentary import visible_commentary +from hermes_constants import get_hermes_home_override + +import asyncio +import pytest + + +def _row(text="I will inspect the files."): + return { + "role": "assistant", + "content": "", + "reasoning": f"Private summary.\n\n{text}", + "reasoning_content": f"Private summary.\n\n{text}", + "codex_message_items": [ + { + "type": "message", + "role": "assistant", + "phase": "commentary", + "content": [{"type": "output_text", "text": text}], + }, + { + "type": "message", + "role": "assistant", + "phase": "analysis", + "content": [{"type": "output_text", "text": "private analysis"}], + }, + ], + "tool_calls": [ + { + "id": "call-1", + "type": "function", + "function": {"name": "terminal", "arguments": "{}"}, + } + ], + "timestamp": 1, + } + + +def test_rest_history_separates_public_commentary_without_changing_source(): + row = _row() + projected = _project_for_display([row])[0] + + assert projected["display_commentary"] == ["I will inspect the files."] + assert projected["display_reasoning"] == "Private summary." + assert projected["codex_message_items"] == row["codex_message_items"] + assert projected["tool_calls"] == row["tool_calls"] + assert row["reasoning"] == "Private summary.\n\nI will inspect the files." + + +def test_gateway_history_uses_same_display_projection(): + row = _row() + projected = _history_to_messages([row])[0] + + assert projected["display_commentary"] == ["I will inspect the files."] + assert projected["display_reasoning"] == "Private summary." + assert projected["codex_message_items"] == row["codex_message_items"] + assert row["reasoning"].endswith("I will inspect the files.") + + +def test_history_applies_live_stripping_and_redaction_without_changing_raw_items( + monkeypatch, +): + monkeypatch.setattr("agent.redact._redact_enabled", lambda: True) + raw = ( + "Start. private scratchpad Key=" + "sk-" + "demo0123456789abcdef" + ) + row = _row(raw) + + for projected in (_project_for_display([row])[0], _history_to_messages([row])[0]): + assert projected["display_commentary"] == [visible_commentary(raw)] + assert "" not in projected["display_commentary"][0] + assert "sk-demo0123456789abcdef" not in projected["display_commentary"][0] + assert projected["display_reasoning"] == "Private summary." + assert projected["codex_message_items"] == row["codex_message_items"] + + +@pytest.mark.parametrize( + "disabled", + [ + {"show_commentary": False}, + {"show_commentary": "false"}, + {"interim_assistant_messages": False}, + ], +) +def test_owner_profile_settings_apply_to_rest_and_rpc(monkeypatch, tmp_path, disabled): + from hermes_cli import config as config_mod + + enabled_home, disabled_home = tmp_path / "enabled", tmp_path / "disabled" + observed = [] + + def config(): + home = get_hermes_home_override() + observed.append(str(home)) + return {"display": {} if str(home) == str(enabled_home) else disabled} + + monkeypatch.setattr(config_mod, "load_config", config) + row = _row() + for adapter in ( + lambda home: _project_for_display([row], home=home)[0], + lambda home: _history_to_messages([row], profile_home=home)[0], + ): + assert adapter(enabled_home)["display_commentary"] == [ + "I will inspect the files." + ] + hidden = adapter(disabled_home) + assert hidden["display_commentary"] == [] + assert hidden["display_reasoning"] == "Private summary." + assert hidden["reasoning"] == row["reasoning"] + assert adapter(enabled_home)["display_reasoning"] == "Private summary." + assert observed == [str(enabled_home), str(disabled_home), str(enabled_home)] * 2 + + +def test_rest_pages_bind_the_history_owner_for_messages_and_around( + monkeypatch, tmp_path +): + from hermes_cli import config as config_mod + from hermes_cli.web_routers import sessions + import hermes_state_timeline + + row = _row() + homes = {name: tmp_path / name for name in ("visible", "hidden")} + monkeypatch.setattr( + sessions, "_cron_profile_home", lambda profile: (profile, homes[profile]) + ) + monkeypatch.setattr( + config_mod, + "load_config", + lambda: { + "display": { + "show_commentary": str(get_hermes_home_override()) + == str(homes["visible"]) + } + }, + ) + + class FakeDB: + def resolve_session_id(self, sid): + return sid + + def resolve_resume_session_id(self, sid): + return sid + + def get_messages(self, *args, **kwargs): + return [row] + + monkeypatch.setattr( + sessions, "_with_db", lambda profile, fn, read_only: fn(FakeDB()) + ) + monkeypatch.setattr(sessions, "_timeline_session_id", lambda db, sid, owner: sid) + monkeypatch.setattr( + hermes_state_timeline, + "get_session_messages_around", + lambda *args, **kwargs: {"messages": [row], "pagination": {}}, + ) + + async def exercise(): + for profile, expected in ( + ("visible", ["I will inspect the files."]), + ("hidden", []), + ): + page = await sessions.get_session_messages( + "sid", + profile=profile, + limit=120, + offset=0, + order="latest", + include_compacted=True, + ) + around = await sessions.get_session_messages_around( + "sid", row_id=1, profile=profile, limit=120 + ) + assert page["messages"][0]["display_commentary"] == expected + assert around["messages"][0]["display_commentary"] == expected + assert page["profile"] == around["profile"] == profile + + asyncio.run(exercise()) + + +def test_unscoped_rest_history_uses_custom_home_of_its_database(monkeypatch, tmp_path): + from hermes_cli import config as config_mod + from hermes_cli.web_routers import sessions + from hermes_constants import reset_hermes_home_override, set_hermes_home_override + import hermes_state_timeline + + custom, default = tmp_path / "custom", tmp_path / "default" + monkeypatch.setattr( + sessions, "_cron_profile_home", lambda profile: ("default", default) + ) + monkeypatch.setattr( + config_mod, + "load_config", + lambda: { + "display": { + "show_commentary": str(get_hermes_home_override()) != str(custom) + } + }, + ) + row = _row() + + class FakeDB: + def resolve_session_id(self, sid): + return sid + + def resolve_resume_session_id(self, sid): + return sid + + def get_messages(self, *args, **kwargs): + return [row] + + monkeypatch.setattr( + sessions, "_with_db", lambda profile, fn, read_only: fn(FakeDB()) + ) + monkeypatch.setattr(sessions, "_timeline_session_id", lambda db, sid, owner: sid) + monkeypatch.setattr( + hermes_state_timeline, + "get_session_messages_around", + lambda *args, **kwargs: {"messages": [row], "pagination": {}}, + ) + + async def exercise(): + page = await sessions.get_session_messages( + "sid", + profile=None, + limit=120, + offset=0, + order="latest", + include_compacted=True, + ) + around = await sessions.get_session_messages_around( + "sid", + row_id=1, + profile=None, + limit=120, + ) + assert page["messages"][0]["display_commentary"] == [] + assert around["messages"][0]["display_commentary"] == [] + + token = set_hermes_home_override(str(custom)) + try: + asyncio.run(exercise()) + finally: + reset_hermes_home_override(token) + + +def test_cold_resume_uses_the_resumed_session_home(monkeypatch, tmp_path): + from hermes_cli import config as config_mod + from tui_gateway.server import _Resume + + home = tmp_path / "disabled" + monkeypatch.setattr( + config_mod, + "load_config", + lambda: { + "display": {"show_commentary": str(get_hermes_home_override()) != str(home)} + }, + ) + resume = _Resume.__new__(_Resume) + resume.omit_messages = False + resume.profile_home = home + + assert resume.messages([_row()])[0]["display_commentary"] == [] + + +def test_joined_canonical_commentary_uses_sanitized_display_copy(monkeypatch): + monkeypatch.setattr("agent.redact._redact_enabled", lambda: True) + raw = "Checking. Key=" + "sk-" + "demo0123456789abcdef" + row = _row(raw) + row["content"] = raw + + visible = _project_for_display([row])[0] + assert visible["display_content"] == visible_commentary(raw) + assert visible["content"] == raw + assert visible["display_reasoning"] == "Private summary." + + from agent.history_commentary import project_history_commentary + from hermes_cli import config as config_mod + + monkeypatch.setattr( + config_mod, "load_config", lambda: {"display": {"show_commentary": False}} + ) + hidden = project_history_commentary([row])[0] + assert hidden["display_content"] == visible_commentary(raw) + assert hidden["display_commentary"] == [] + + +def test_only_exact_delimiter_bounded_copy_is_removed(): + row = _row("Same phrase") + row["reasoning"] = "Summary mentioning Same phrase as an incidental substring." + projected = _project_for_display([row])[0] + assert projected["display_reasoning"] == row["reasoning"] + + row["reasoning"] = "Before.\n\nSame phrase\n\nAfter." + assert _project_for_display([row])[0]["display_reasoning"] == "Before.\n\nAfter." + + +def test_foreign_and_malformed_items_are_not_public(): + row = _row() + row["codex_message_items"][0]["role"] = "user" + assert "display_commentary" not in _project_for_display([row])[0] + row["codex_message_items"] = "{invalid JSON" + assert "display_commentary" not in _project_for_display([row])[0] + + +def test_mixed_canonical_content_is_sanitized_without_truncating_possible_final( + monkeypatch, +): + raw = "Checking. hidden key=" + "sk-" + "demo0123456789abcdef" + row = _row(raw) + row["content"] = raw + "\n\nFinal result." + monkeypatch.setattr("agent.redact._redact_enabled", lambda: True) + + for projected in (_project_for_display([row])[0], _history_to_messages([row])[0]): + assert projected["display_commentary"] == [] + assert projected["display_content"] == visible_commentary(row["content"]) + assert raw not in projected["display_reasoning"] + + from hermes_cli import config as config_mod + + monkeypatch.setattr( + config_mod, "load_config", lambda: {"display": {"show_commentary": False}} + ) + for projected in (_project_for_display([row])[0], _history_to_messages([row])[0]): + assert projected["display_commentary"] == [] + assert projected["display_content"] == visible_commentary(row["content"]) + assert row["content"].startswith(raw) + + +@pytest.mark.parametrize("final", ["Checking. The answer is 42.", "Checking."]) +def test_stream_recovered_final_without_final_sidecar_is_authoritative( + monkeypatch, final +): + from types import SimpleNamespace + from agent.turn_finalizer import _close_transcript_tail + from hermes_cli import config as config_mod + + row = _row("Checking.") + agent = SimpleNamespace(_db_flush_scan_prefix=None) + _close_transcript_tail(agent, [row], final, False, True) + assert row["content"] == final + for visible in (True, False): + monkeypatch.setattr( + config_mod, + "load_config", + lambda visible=visible: {"display": {"show_commentary": visible}}, + ) + for projected in ( + _project_for_display([row])[0], + _history_to_messages([row])[0], + ): + assert ( + projected.get( + "display_content", projected.get("content", projected.get("text")) + ) + == final + ) + + +@pytest.mark.parametrize("final", ["Checking. The answer is 42.", "Checking."]) +@pytest.mark.parametrize("phase", ["final", "final_answer", None]) +def test_canonical_content_matching_final_item_is_never_stripped( + monkeypatch, final, phase +): + from hermes_cli import config as config_mod + + row = _row("Checking.") + row["content"] = final + item = { + "type": "message", + "role": "assistant", + "content": [{"type": "output_text", "text": final}], + } + if phase is not None: + item["phase"] = phase + row["codex_message_items"].append(item) + for visible in (True, False): + monkeypatch.setattr( + config_mod, + "load_config", + lambda visible=visible: {"display": {"show_commentary": visible}}, + ) + for projected in ( + _project_for_display([row])[0], + _history_to_messages([row])[0], + ): + assert ( + projected.get( + "display_content", projected.get("content", projected.get("text")) + ) + == final + ) + assert projected["display_commentary"] == (["Checking."] if visible else []) + + +def test_mixed_canonical_preserves_final_repetition_of_a_commentary_phrase(monkeypatch): + from hermes_cli import config as config_mod + + row = _row("Checking") + row["content"] = "Checking\n\nFinal: Checking completed." + for visible in (True, False): + monkeypatch.setattr( + config_mod, + "load_config", + lambda visible=visible: {"display": {"show_commentary": visible}}, + ) + for projected in ( + _project_for_display([row])[0], + _history_to_messages([row])[0], + ): + assert projected["display_content"] == row["content"] + assert projected["display_commentary"] == [] + + row["codex_message_items"].append({ + "type": "message", + "role": "assistant", + "phase": "commentary", + "content": [{"type": "output_text", "text": "Second update"}], + }) + row["content"] = "Checking\n\nSecond update\n\nFinal: Checking completed." + monkeypatch.setattr( + config_mod, "load_config", lambda: {"display": {"show_commentary": True}} + ) + for projected in (_project_for_display([row])[0], _history_to_messages([row])[0]): + assert projected["display_content"] == row["content"] + assert projected["display_commentary"] == [] + + +def test_legacy_newline_and_ambiguous_repeats_do_not_leak_public_text_into_thinking(): + row = _row("Public update.") + row["reasoning"] = "Private before.\nPublic update.\nPrivate after." + assert ( + _project_for_display([row])[0]["display_reasoning"] + == "Private before.\nPrivate after." + ) + + row["reasoning"] = ( + "Private before.\n\nPublic update.\n\nPrivate after.\n\nPublic update." + ) + assert ( + _project_for_display([row])[0]["display_reasoning"] + == "Private before.\n\nPrivate after." + ) + assert row["reasoning"].endswith("Public update.") + + +def test_overlapping_public_items_do_not_leave_partial_text_in_final_or_thinking( + monkeypatch, +): + from hermes_cli import config as config_mod + + row = _row("Start") + row["codex_message_items"].insert( + 1, + { + "type": "message", + "role": "assistant", + "phase": "commentary", + "content": [{"type": "output_text", "text": "Start\nTail public"}], + }, + ) + row["reasoning"] = "Private.\n\nStart\nTail public" + row["content"] = "Start\nTail public\n\nFinal answer." + + for visible in (True, False): + monkeypatch.setattr( + config_mod, + "load_config", + lambda visible=visible: {"display": {"show_commentary": visible}}, + ) + for projected in ( + _project_for_display([row])[0], + _history_to_messages([row])[0], + ): + assert projected["display_content"] == row["content"] + assert projected["display_reasoning"] == "Private." + assert projected["display_commentary"] == [] + + # Without an attributable leading commentary span, this belongs to the final. + row["content"] = "Leading Start\nTail public trailing" + for visible in (True, False): + monkeypatch.setattr( + config_mod, + "load_config", + lambda visible=visible: {"display": {"show_commentary": visible}}, + ) + for projected in ( + _project_for_display([row])[0], + _history_to_messages([row])[0], + ): + assert projected["display_content"] == "Leading Start\nTail public trailing" + assert projected["display_commentary"] == ( + ["Start", "Start\nTail public"] if visible else [] + ) + + row["content"] = "Start\nTail publicFinal answer." + monkeypatch.setattr( + config_mod, "load_config", lambda: {"display": {"show_commentary": True}} + ) + for projected in (_project_for_display([row])[0], _history_to_messages([row])[0]): + assert projected["display_content"] == row["content"] + assert projected["display_commentary"] == [] diff --git a/tui_gateway/compute_host.py b/tui_gateway/compute_host.py index 201fff5303..f68f7d2df0 100644 --- a/tui_gateway/compute_host.py +++ b/tui_gateway/compute_host.py @@ -462,7 +462,7 @@ class ComputeHost: else: output = server._mirror_slash_side_effects(sid, session, command) if command else "" with session["history_lock"]: - messages = server._history_to_messages(list(session.get("history") or [])) + messages = server._history_to_messages(list(session.get("history") or []), profile_home=session.get("profile_home")) ack = {"output": output, **_history_meta(session), "messages": messages} ack["session_info"] = server._session_info(session.get("agent"), session) return ack diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index 05d03d305c..87af3176b1 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -399,7 +399,7 @@ def _(rid, params: dict) -> dict: _schedule_session_cap_enforcement() # trim detached idle sessions over the cap cwd = _sessions[sid]["cwd"] override = session_model_override or {} - messages = _history_to_messages(history) # hidden seed rows are not on the wire; count what is (as resume does) + messages = _history_to_messages(history, profile_home=profile_home) # hidden seed rows are not on the wire; count what is (as resume does) return _ok(rid, { "session_id": sid, "stored_session_id": key, "message_count": len(messages), "messages": messages, # Reflect the override now so the client doesn't clobber its sticky pick. @@ -572,7 +572,7 @@ class _Resume: return self.db.get_messages_as_conversation(self.target, repair_alternation=repair, include_row_ids=True) def messages(self, display: list) -> list: - return [] if self.omit_messages else _history_to_messages(display) + return [] if self.omit_messages else _history_to_messages(display, profile_home=self.profile_home) def read_history(self) -> tuple: """One lineage SELECT, two projections: model-fed copy alternation-repaired (healed once @@ -1794,7 +1794,7 @@ def _(rid, params: dict, session: dict) -> dict: # use. See #87059. history = db.get_messages_as_conversation( session["session_key"], include_ancestors=True, include_row_ids=True) - return _ok(rid, {"count": len(history), "messages": _history_to_messages(history)}) + return _ok(rid, {"count": len(history), "messages": _history_to_messages(history, profile_home=session.get("profile_home"))}) @_session_method("session.undo", live=True) @@ -1869,7 +1869,7 @@ def _compress_via_compute_host(rid, params: dict, session: dict) -> dict: "status": "compressed", "turn_isolation": True, # `messages` goes top-level for the transcript replacement; don't duplicate it in the ack. "host_ack": {key: value for key, value in ack.items() if key != "messages"}, "info": host_info, - "messages": _history_to_messages(ack.get("messages")) if isinstance(ack.get("messages"), list) else [], + "messages": _history_to_messages(ack.get("messages"), profile_home=session.get("profile_home")) if isinstance(ack.get("messages"), list) else [], "usage": host_info.get("usage") if isinstance(host_info.get("usage"), dict) else {}}) @@ -1914,7 +1914,7 @@ def _compress_live(rid, sid: str, session: dict, focus_topic: str) -> dict: "status": "aborted" if summary["aborted"] else "compressed", "removed": removed, "before_messages": before_count, "after_messages": len(messages), "before_tokens": before_tokens, "after_tokens": after_tokens, "summary": summary, - "usage": usage, "info": info, "messages": _history_to_messages(messages)}) + "usage": usage, "info": info, "messages": _history_to_messages(messages, profile_home=session.get("profile_home"))}) finally: # Always clear the pinned compressing status (success, no-op, or raise). _status_update(sid, "ready") @@ -2072,7 +2072,7 @@ def _(rid, params: dict, session: dict) -> dict: except Exception as e: return _err(rid, 5000, f"agent init failed on branch: {e}") return _ok(rid, {"session_id": new_sid, "stored_session_id": new_key, "title": title, "parent": old_key, - "message_count": len(history), "messages": _history_to_messages(history), + "message_count": len(history), "messages": _history_to_messages(history, profile_home=session.get("profile_home")), "info": _session_info(agent, _sessions.get(new_sid))}) diff --git a/tui_gateway/methods_slash.py b/tui_gateway/methods_slash.py index 138cf7fc4e..2bf17d90a1 100644 --- a/tui_gateway/methods_slash.py +++ b/tui_gateway/methods_slash.py @@ -99,7 +99,7 @@ def _format_live_history_output(sid: str, session: dict, arg: str) -> str: with session["history_lock"]: history = list(session.get("history", [])) db_history = _live_session_messages(session) - messages = _history_to_messages(history if db_history is None else db_history) + messages = _history_to_messages(history if db_history is None else db_history, profile_home=session.get("profile_home")) if not messages: return "No conversation history yet." lines = ["Conversation History", "────────────────────────────────────────"] @@ -128,12 +128,12 @@ def _format_live_prompt_output(sid: str, session: dict, arg: str) -> str: def _format_live_context_output(sid: str, session: dict, arg: str) -> str: from collections import Counter try: - messages = _history_to_messages(_live_session_messages(session) or []) + messages = _history_to_messages(_live_session_messages(session) or [], profile_home=session.get("profile_home")) except Exception: messages = [] # malformed db rows fall back to the live history below if not messages: with session["history_lock"]: - messages = _history_to_messages(list(session.get("history", []))) + messages = _history_to_messages(list(session.get("history", [])), profile_home=session.get("profile_home")) usage = _session_usage_snapshot(session) mirror = _metadata_mirror(session) lines = [f"Conversation: {len(messages)} messages" if messages else "Conversation is empty (no messages yet)."] diff --git a/tui_gateway/server.py b/tui_gateway/server.py index b232f7343f..88ae448518 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -2895,7 +2895,7 @@ def _live_session_payload( history = _live_visible_history(session, db, in_memory_history) # message_count follows _resume_response: the stored size when messages are omitted, else the wire count # (a hidden seed row is in ``history`` but never on the wire). - messages = [] if omit_messages else _history_to_messages(history) + messages = [] if omit_messages else _history_to_messages(history, profile_home=session.get("profile_home")) payload = { "info": _fallback_session_info(session), "message_count": len(history) if omit_messages else len(messages), "messages": messages, diff --git a/tui_gateway/session_history.py b/tui_gateway/session_history.py index b01a4129a3..e322c24e20 100644 --- a/tui_gateway/session_history.py +++ b/tui_gateway/session_history.py @@ -197,7 +197,9 @@ _HISTORY_ASSISTANT_DETAIL_KEYS = ( _HISTORY_ROLES = frozenset({"user", "assistant", "tool", "system"}) -def _history_to_messages(history: list[dict]) -> list[dict]: +def _history_to_messages(history: list[dict], *, profile_home=None) -> list[dict]: + from agent.history_commentary import project_history_commentary + messages = [] tool_call_args = {} for m in history: @@ -263,7 +265,7 @@ def _history_to_messages(history: list[dict]) -> list[dict]: if m.get("display_metadata"): msg["display_metadata"] = m["display_metadata"] messages.append(msg) - return messages + return project_history_commentary(messages, home=profile_home) def _coerce_seed_history(value: Any) -> list[dict]: From 8a1272b814f039fc54f35a09a40ca7b015c25b51 Mon Sep 17 00:00:00 2001 From: "hermes-seaeye[bot]" <307254004+hermes-seaeye[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 00:37:55 +0000 Subject: [PATCH 019/112] fmt(js): `npm run fix` on merge (#120819) Co-authored-by: github-actions[bot] --- apps/desktop/src/plugins/hermes-bots/group-turns.ts | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/apps/desktop/src/plugins/hermes-bots/group-turns.ts b/apps/desktop/src/plugins/hermes-bots/group-turns.ts index 00af6bb52c..e859437943 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-turns.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-turns.ts @@ -226,7 +226,9 @@ export function retainedGroupTurnError(state: GroupSessionSnapshot | null | unde * carries no start time, so two identical consecutive failures differ only * by `turn_started_at`. */ function retainedGroupTurnKey(state: GroupSessionSnapshot | null | undefined): null | string { - return retainedGroupTurnError(state) === null ? null : JSON.stringify([state?.turn_started_at ?? null, state?.inflight]) + return retainedGroupTurnError(state) === null + ? null + : JSON.stringify([state?.turn_started_at ?? null, state?.inflight]) } /** Does a user row follow the stranded turn's own prompt? Then a later turn @@ -1378,7 +1380,9 @@ export async function harvestStrandedGroupReply(group: string, member: GroupMemb const retained = laterTurnAfterStranded(messages, strandedBefore) ? null : retainedGroupTurnError(state) const pick = - retained === null && messages.length > strandedBefore ? pickStrandedGroupTurnReply(messages, strandedBefore) : null + retained === null && messages.length > strandedBefore + ? pickStrandedGroupTurnReply(messages, strandedBefore) + : null const reply = typeof pick === 'string' ? pick : null const failedNotice = typeof pick === 'string' ? null : (pick?.failedNotice ?? null) From 0a2e1732a0a207871b82900b50a19e4242ffdb26 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 10:55:04 -0700 Subject: [PATCH 020/112] fix(plugins): validator accepts the config_schema types the loader and renderer accept MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `hermes plugins validate` kept a private `_CONFIG_TYPES` copy that never learned `secret` (or `object`) when the manifest loader and the Desktop settings renderer did, so catalog admission (`pinned-source-validate`) went red for any plugin that declares a `type: secret` setting — the documented way to surface an API key in the Plugins tab (#120088 web-search-plus 4.3.0, #120492 browserclaw 3.1.0). Derive the admission set from `plugins_manifest._CONFIG_SCHEMA_TYPES` so the three cannot drift again; `mapping`/`map`, which only the validator ever accepted (the loader warned "unknown type" on every load), are no longer admitted. --- hermes_cli/plugin_validate.py | 8 ++++---- tests/hermes_cli/test_plugin_validate.py | 20 ++++++++++++++++++++ 2 files changed, 24 insertions(+), 4 deletions(-) diff --git a/hermes_cli/plugin_validate.py b/hermes_cli/plugin_validate.py index 490c412099..04185f8229 100644 --- a/hermes_cli/plugin_validate.py +++ b/hermes_cli/plugin_validate.py @@ -24,12 +24,12 @@ from pathlib import Path from typing import Any, Dict, List, Optional, Tuple from hermes_cli.plugin_validate_desktop import check_desktop_surface +from hermes_cli.plugins_manifest import _CONFIG_SCHEMA_TYPES _UPPER_SNAKE_RE = re.compile(r"^[A-Z][A-Z0-9_]*$") -_CONFIG_TYPES = { - "str", "string", "int", "integer", "float", "number", - "bool", "boolean", "list", "array", "dict", "mapping", "map", -} +# Admission accepts exactly the ``config_schema`` types the loader type-checks at load time (and the +# Desktop settings renderer keys its field table on) — a private copy drifted and rejected ``secret``. +_CONFIG_TYPES = frozenset(_CONFIG_SCHEMA_TYPES) _PROBE_TIMEOUT = 30 _PROBE_SENTINEL = "HERMES_VALIDATE_JSON:" diff --git a/tests/hermes_cli/test_plugin_validate.py b/tests/hermes_cli/test_plugin_validate.py index ecfd09db4c..0569ccb03a 100644 --- a/tests/hermes_cli/test_plugin_validate.py +++ b/tests/hermes_cli/test_plugin_validate.py @@ -87,6 +87,26 @@ def test_requires_hermes_spec_is_validated(tmp_path): assert any(name == "requires_hermes" and ok for name, ok, _ in report.checks) +def test_config_schema_admits_every_type_the_loader_and_renderer_accept(tmp_path): + """A ``type:`` the Desktop settings renderer/loader accept (``secret`` + ``env:``, ``object``) must + pass admission — the catalog validator rejecting a documented type blocks pins of plugins that + declare a secret setting.""" + from hermes_cli.plugins_manifest import _CONFIG_SCHEMA_TYPES + from hermes_cli.plugins_settings import _FIELD_TYPES + + assert set(_FIELD_TYPES) == set(_CONFIG_SCHEMA_TYPES) + schema = {f"k_{t}": {"type": t} for t in _FIELD_TYPES} + schema["api_key"] = {"type": "secret", "env": "FIXTURE_API_KEY", "description": "token"} + d = _make_plugin(tmp_path, manifest=dict(BASE_MANIFEST, config_schema=schema)) + + report = validate_plugin_dir(d) + + assert ("config schema", True, "shape valid") in report.checks, report.failures + bad = validate_plugin_dir(_make_plugin( + tmp_path / "bad", manifest=dict(BASE_MANIFEST, config_schema={"x": {"type": "mapping"}}))) + assert [ok for n, ok, _ in bad.checks if n == "config schema"] == [False], bad.checks + + def test_admission_runs_the_install_scanner(tmp_path): """Admission and install must agree: a tree the installer would hard-block (dangerous) fails validation; caution findings are surfaced to the reviewer as warnings without failing.""" From e50b3a393300b6d139611d36fd5d3b0baa2d2e27 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:32:13 -0700 Subject: [PATCH 021/112] fix(update): checkout_contains reads the build stamp on a no-.git image, so the pending-restart catch-up stops contradicting itself On a Docker / Hermes Cloud image there is no .git; the code identity is the baked build stamp (build_info.get_code_identity, source == "build-file"). checkout_contains walked git merge-base regardless and was therefore always False on an image, which made _marker_only_restart_obsolete and _live_fleet_covers_receipt disagree with the fleet matrix: the update catch-up printed "Every running gateway already serves the checkout code" and "gateways are still off the checkout code" in the same run (found in the s6 update-tail audit for #120516). With no history to walk, contained collapses to equal-to-the-stamp (either side may be the short form an older writer recorded). A source install keeps asking git, so a carried hotfix past the pulled SHA still counts (#119367). Red on main / green here: tests/hermes_cli/test_update_fleet_checkout_build_stamp.py --- hermes_cli/update_cmd_fleet_checkout.py | 14 +++++++- .../test_update_fleet_checkout_build_stamp.py | 36 +++++++++++++++++++ 2 files changed, 49 insertions(+), 1 deletion(-) create mode 100644 tests/hermes_cli/test_update_fleet_checkout_build_stamp.py diff --git a/hermes_cli/update_cmd_fleet_checkout.py b/hermes_cli/update_cmd_fleet_checkout.py index 2d7dfffc1b..c758f678b9 100644 --- a/hermes_cli/update_cmd_fleet_checkout.py +++ b/hermes_cli/update_cmd_fleet_checkout.py @@ -14,11 +14,23 @@ logger = logging.getLogger(__name__) def checkout_contains(sha: str) -> bool: - """True when ``sha`` is an ancestor of (or equal to) the checkout HEAD; False on any probe failure. + """True when ``sha`` is an ancestor of (or equal to) the code this checkout runs; False on any + probe failure. Fail-closed on purpose: an unknown ancestry is not evidence that the fleet serves the update. + + A Docker/Cloud image has no ``.git``; its identity is the baked build stamp + (``build_info.get_code_identity`` → ``source == "build-file"``). There is no history to walk, so + "contained" collapses to "equal to the stamp" — without this the probe was always False on an + image and the pending-restart catch-up printed "every gateway serves the checkout" and "still + off the checkout code" in the same breath. """ + from hermes_cli.build_info import get_code_identity from hermes_cli.update_cmd import _m + identity = get_code_identity() or {} + if identity.get("source") == "build-file": + stamped = str(identity.get("sha") or "") + return bool(stamped) and (stamped == sha or stamped.startswith(sha) or sha.startswith(stamped)) try: result = subprocess.run( ["git", "merge-base", "--is-ancestor", sha, "HEAD"], diff --git a/tests/hermes_cli/test_update_fleet_checkout_build_stamp.py b/tests/hermes_cli/test_update_fleet_checkout_build_stamp.py new file mode 100644 index 0000000000..26a163235a --- /dev/null +++ b/tests/hermes_cli/test_update_fleet_checkout_build_stamp.py @@ -0,0 +1,36 @@ +"""``checkout_contains`` on a build-stamped image (no ``.git``): ancestry collapses to stamp equality. + +On a Docker/Cloud image ``git merge-base`` has nothing to walk, so the probe was always False and the +pending-restart catch-up printed "every gateway already serves the checkout" and "still off the checkout +code" in the same run. The stamp IS the checkout there. +""" + +from hermes_cli import update_cmd_fleet_checkout as chk + + +def _identity(monkeypatch, sha, source): + import hermes_cli.build_info as bi + monkeypatch.setattr(bi, "get_code_identity", lambda refresh=False: {"sha": sha, "short_sha": sha[:8], "source": source}) + + +def test_build_stamped_image_contains_exactly_its_stamp(monkeypatch): + stamped = "b936546561aa0d2e6d0f7c3d1a9c5e8f2b4d6a70" + _identity(monkeypatch, stamped, "build-file") + # no git call may be attempted on an image: make one blow up if it is + monkeypatch.setattr(chk.subprocess, "run", lambda *a, **k: (_ for _ in ()).throw(AssertionError("git probe on a build-stamped image"))) + assert chk.checkout_contains(stamped) is True + assert chk.checkout_contains(stamped[:12]) is True # short form recorded by an older writer + assert chk.checkout_contains("0000000000aa0d2e6d0f7c3d1a9c5e8f2b4d6a70") is False + + +def test_git_checkout_still_walks_ancestry(monkeypatch): + """Control: a source install keeps asking git, so a carried hotfix past the pulled SHA still counts.""" + _identity(monkeypatch, "deadbeef" * 5, "git") + calls = [] + + class _R: + returncode = 0 + + monkeypatch.setattr(chk.subprocess, "run", lambda cmd, **k: calls.append(cmd) or _R()) + assert chk.checkout_contains("cafebabe" * 5) is True + assert calls and calls[0][:3] == ["git", "merge-base", "--is-ancestor"] From 8fb196b3b6fad09425ffbdb1e4079d0afd817e7a Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:08:37 -0700 Subject: [PATCH 022/112] test(desktop-core): transcript oracle across every chat transition (C2) Class: duplicate / vanishing / reordered messages in the Desktop (#120005 and its dupes #119131 #119566 #119540 #118934; completed-reply refresh, warm-resume, steer and reconnect regressions). One real Electron app + one real hermes serve; only the LLM is faked by a lane-private scripted provider keyed by each turn's own marker (no shared trigger soup / global counters). After every transition - stream, tool-call turn, reasoning turn, steer, queued follow-up, session switch mid-stream, warm resume, reload, WebSocket drop+reconnect mid-stream (loopback proxy the test controls), a non-default profile's chat, a forced second socket to the same backend, final reload - the oracle asserts: persisted rows render exactly once, in order, nothing unpersisted renders, no marker ever renders twice even transiently, and the backend's concatenated deltas equal what the provider streamed. Why: the Desktop E2E job is disabled, so the 67 legacy specs run nowhere; this is the small deterministic lane meant to be required (retries 0, no fixed sleeps, one worker). --- apps/desktop/e2e/core/README.md | 42 ++ apps/desktop/e2e/core/harness.ts | 580 ++++++++++++++++++ apps/desktop/e2e/core/oracle.ts | 439 +++++++++++++ apps/desktop/e2e/core/playwright.config.ts | 31 + apps/desktop/e2e/core/provider.ts | 317 ++++++++++ .../e2e/core/transcript-integrity.spec.ts | 292 +++++++++ 6 files changed, 1701 insertions(+) create mode 100644 apps/desktop/e2e/core/README.md create mode 100644 apps/desktop/e2e/core/harness.ts create mode 100644 apps/desktop/e2e/core/oracle.ts create mode 100644 apps/desktop/e2e/core/playwright.config.ts create mode 100644 apps/desktop/e2e/core/provider.ts create mode 100644 apps/desktop/e2e/core/transcript-integrity.spec.ts diff --git a/apps/desktop/e2e/core/README.md b/apps/desktop/e2e/core/README.md new file mode 100644 index 0000000000..7dd68d731a --- /dev/null +++ b/apps/desktop/e2e/core/README.md @@ -0,0 +1,42 @@ +# Desktop core suite (required CI lane) + +A small, deterministic Electron suite that guards two issue classes end to end: + +- **C2 transcript integrity** — `transcript-integrity.spec.ts`: one real app + + one real `hermes serve`, only the LLM faked (`provider.ts`, scripted per turn + by a unique marker, every streamed chunk recorded). After every transition — + stream, tool-call turn, reasoning turn, steer, queued follow-up, session + switch mid-stream, warm resume, reload, WebSocket drop + reconnect mid-stream, + a non-default profile's chat, a forced second socket to the same backend, + final reload — `oracle.ts` asserts: + - every persisted user/assistant message is rendered exactly once, in order, + and nothing unpersisted is rendered; + - no marker is ever rendered twice, even transiently (in-page + MutationObserver sampler — the #120005 garble healed on its own in the + final DOM, so a final-state check alone misses it); + - backend stream integrity: each turn's concatenated `message.delta` / + `reasoning.delta` equals what the provider streamed, and + `message.complete` equals the final completion. +- **C5 boot / process lifecycle** — `boot-lifecycle.spec.ts`: interactive + composer + first turn, exactly one backend; `kill -9` backend → exactly one + supervised respawn and a working turn; quit mid-turn with a running tool + child → zero sandbox processes (/proc census on the sandbox `HERMES_HOME`, + so orphans reparented to init are counted); relaunch the same home 3× → one + backend per boot, zero after each quit, transcript cold-hydrates once. + +Rules the suite keeps (why the old lane was disabled): no fixed sleeps as +synchronisation (every wait is on a frame, pid, DOM state or persisted row +with a deadline), no shared mock state between scenarios (replies are keyed +by the turn's own marker), no visual baselines, `retries: 0`, one worker, +sandboxed `HOME`/`HERMES_HOME`/user-data per test, all `HERMES_*` and +credential env stripped from the spawned app. + +Run locally (Linux, after `npm run build` in `apps/desktop`): + +```sh +cd apps/desktop +xvfb-run -a npx playwright test -c e2e/core/playwright.config.ts +``` + +`HERMES_E2E_CORE_ROOT` picks the sandbox parent dir (default: OS tmpdir); +`HERMES_E2E_CORE_KEEP=1` keeps sandboxes for post-mortem. diff --git a/apps/desktop/e2e/core/harness.ts b/apps/desktop/e2e/core/harness.ts new file mode 100644 index 0000000000..8042a45e6b --- /dev/null +++ b/apps/desktop/e2e/core/harness.ts @@ -0,0 +1,580 @@ +/** + * Lane-private harness for the core Desktop suite: an isolated sandbox with a + * fake HOME, a launch helper, a per-socket WebSocket recorder, a /proc-based + * process census for the backend, and the transcript oracle. + * + * Synchronisation rule for every helper here: wait on an observable fact + * (a frame, a DOM state, a persisted row, a pid) with a deadline — never a + * fixed sleep. + */ + +import * as fs from 'node:fs' +import * as net from 'node:net' +import * as os from 'node:os' +import * as path from 'node:path' +import { DatabaseSync } from 'node:sqlite' + +import { _electron, type ElectronApplication, expect, type Page } from '@playwright/test' + +import { resolveElectronBinary } from '../electron-binary' + +export const DESKTOP_ROOT = path.resolve(import.meta.dirname, '..', '..') +export const REPO_ROOT = path.resolve(DESKTOP_ROOT, '..', '..') + +// ─── Sandbox ──────────────────────────────────────────────────────────── + +export interface CoreSandbox { + root: string + /** Prepended to PATH: external-platform fakes (see createCoreSandbox). */ + bin: string + home: string + hermesHome: string + userDataDir: string + cleanup: () => void +} + +/** + * HOME is faked too, not just HERMES_HOME: profile roots are anchored to + * `Path.home()/.hermes`, so a sandbox HERMES_HOME under the real ~/.hermes + * would read and write the real install's profiles/. + */ +export function createCoreSandbox(label: string): CoreSandbox { + const parent = process.env.HERMES_E2E_CORE_ROOT || os.tmpdir() + fs.mkdirSync(parent, { recursive: true }) + const root = fs.mkdtempSync(path.join(parent, `core-${label}-`)) + const home = path.join(root, 'home') + const hermesHome = path.join(home, '.hermes') + const userDataDir = path.join(root, 'user-data') + fs.mkdirSync(hermesHome, { recursive: true }) + fs.mkdirSync(userDataDir, { recursive: true }) + fs.writeFileSync( + path.join(userDataDir, 'window-state.json'), + JSON.stringify({ x: 0, y: 0, width: 1280, height: 860, isMaximized: false }) + ) + fs.writeFileSync(path.join(userDataDir, 'zoom-state.json'), JSON.stringify({ zoomLevel: 0 })) + // External platform fake: a logged-out `gh`. The backend probes `gh auth + // token` for GitHub credentials; with a sandbox HOME a real gh can block on + // the desktop keyring for ~60 s, and a probe in flight at quit outlives the + // backend (reported as a finding) — which would make the orphan census + // depend on the runner's keyring rather than on Hermes. + const bin = path.join(root, 'bin') + fs.mkdirSync(bin, { recursive: true }) + fs.writeFileSync(path.join(bin, 'gh'), '#!/bin/sh\necho "no oauth token found for github.com" >&2\nexit 1\n', { mode: 0o755 }) + + return { + root, + bin, + home, + hermesHome, + userDataDir, + cleanup: () => { + if (!process.env.HERMES_E2E_CORE_KEEP) { + fs.rmSync(root, { recursive: true, force: true }) + } + } + } +} + +export function providerConfigYaml(providerUrl: string, extra = ''): string { + return `model: + default: mock-model + provider: mock +providers: + mock: + api: ${providerUrl}/v1 + name: Mock + api_mode: chat_completions + key_env: MOCK_API_KEY + models: + mock-model: {} + context_length: 64000 +auxiliary: + title_generation: + enabled: false +approvals: + mode: "off" +${extra}` +} + +export function writeProviderHome(dir: string, providerUrl: string, extra = ''): void { + fs.mkdirSync(dir, { recursive: true }) + fs.writeFileSync(path.join(dir, 'config.yaml'), providerConfigYaml(providerUrl, extra)) + fs.writeFileSync(path.join(dir, '.env'), 'MOCK_API_KEY=core-e2e-key\n') +} + +const CREDENTIAL_RE = /(_API_KEY|_TOKEN|_SECRET|_PASSWORD|_CREDENTIALS|_ACCESS_KEY|_PRIVATE_KEY|_BASE_URL)$/ + +/** + * The runner's own env minus credentials and every HERMES_* knob: an agent + * shell exports HERMES_YOLO_MODE / _HERMES_GATEWAY, which the spawned backend + * would inherit (auto-approving every command, changing the run under test). + */ +export function coreAppEnv(sandbox: CoreSandbox, extra: Record = {}): Record { + const env: Record = {} + + for (const [key, value] of Object.entries(process.env)) { + if (!value || CREDENTIAL_RE.test(key) || /^_?HERMES_/.test(key) || key === 'VIRTUAL_ENV') { + continue + } + + env[key] = value + } + + return { + ...env, + PATH: `${sandbox.bin}${path.delimiter}${env.PATH ?? ''}`, + HOME: sandbox.home, + HERMES_HOME: sandbox.hermesHome, + HERMES_DESKTOP_USER_DATA_DIR: sandbox.userDataDir, + HERMES_DESKTOP_IGNORE_EXISTING: '1', + HERMES_DESKTOP_HERMES_ROOT: REPO_ROOT, + HERMES_DESKTOP_APP_NAME: `HermesCoreE2E-${path.basename(sandbox.root)}`, + HERMES_DESKTOP_SKIP_QUIT_CONFIRM: '1', + HERMES_DESKTOP_CDP_PORT: 'off', + ...extra + } +} + +export async function launchCoreApp(env: Record): Promise<{ app: ElectronApplication; page: Page }> { + if (!fs.existsSync(path.join(DESKTOP_ROOT, 'dist', 'electron-main.mjs'))) { + throw new Error("Desktop dist not built: run 'npm run build' in apps/desktop first") + } + + const app = await _electron.launch({ + executablePath: resolveElectronBinary([DESKTOP_ROOT, REPO_ROOT]), + args: [DESKTOP_ROOT, '--disable-gpu', '--no-sandbox'], + env, + cwd: DESKTOP_ROOT + }) + + const page = await app.firstWindow() + + return { app, page } +} + +// ─── Process census ───────────────────────────────────────────────────── + +export interface ProcInfo { + pid: number + ppid: number + cmdline: string +} + +function readProc(pid: number): null | { environ: string; cmdline: string; ppid: number } { + try { + const environ = fs.readFileSync(`/proc/${pid}/environ`, 'utf8') + const cmdline = fs.readFileSync(`/proc/${pid}/cmdline`, 'utf8').split('\0').join(' ').trim() + const stat = fs.readFileSync(`/proc/${pid}/stat`, 'utf8') + // Field 4 (ppid) follows the parenthesised comm, which may contain spaces. + const ppid = Number(stat.slice(stat.lastIndexOf(')') + 2).split(' ')[1]) + + return { environ, cmdline, ppid } + } catch { + return null + } +} + +/** Every live process whose environment carries this sandbox's HERMES_HOME (orphans included). */ +export function sandboxProcesses(sandbox: CoreSandbox): ProcInfo[] { + const needle = `HERMES_HOME=${sandbox.hermesHome}\0` + const out: ProcInfo[] = [] + + for (const entry of fs.readdirSync('/proc')) { + const pid = Number(entry) + + if (!Number.isInteger(pid) || pid === process.pid) { + continue + } + + const info = readProc(pid) + + if (!info || !(info.environ + '\0').includes(needle)) { + continue + } + + // Zombies have an empty cmdline and are already dead for our purposes. + if (!info.cmdline) { + continue + } + + out.push({ pid, ppid: info.ppid, cmdline: info.cmdline }) + } + + return out +} + +/** The `hermes serve` backend(s) spawned for this sandbox. */ +export function backendProcesses(sandbox: CoreSandbox): ProcInfo[] { + return sandboxProcesses(sandbox).filter( + proc => / serve( |$)/.test(proc.cmdline) && !/electron/i.test(proc.cmdline.split(' ')[0]) + ) +} + +// ─── WebSocket recorder ───────────────────────────────────────────────── + +export interface GatewayEventFrame { + socket: number + type: string + sessionId: string + seq: null | number + payload: any +} + +export interface WsRecorder { + sockets: { id: number; url: string; closed: boolean }[] + events: GatewayEventFrame[] + sent: { socket: number; method: string; params: any }[] +} + +export function recordWebSockets(page: Page): WsRecorder { + const rec: WsRecorder = { sockets: [], events: [], sent: [] } + + page.on('websocket', ws => { + if (!ws.url().includes('/api/ws')) { + return + } + + const id = rec.sockets.length + const entry = { id, url: ws.url(), closed: false } + rec.sockets.push(entry) + ws.on('close', () => { + entry.closed = true + }) + ws.on('framereceived', frame => { + try { + const msg = JSON.parse(String(frame.payload)) + + if (msg?.method === 'event' && msg.params) { + rec.events.push({ + socket: id, + type: String(msg.params.type ?? ''), + sessionId: String(msg.params.session_id ?? ''), + seq: typeof msg.params.seq === 'number' ? msg.params.seq : null, + payload: msg.params.payload + }) + } + } catch { + /* binary / non-JSON frames carry no gateway event */ + } + }) + ws.on('framesent', frame => { + try { + const msg = JSON.parse(String(frame.payload)) + + if (typeof msg?.method === 'string') { + rec.sent.push({ socket: id, method: msg.method, params: msg.params }) + } + } catch { + /* ignore */ + } + }) + }) + + return rec +} + +// ─── Network fault injection ──────────────────────────────────────────── + +export interface TcpProxy { + port: number + /** Live client connections through the proxy. */ + connections: () => number + /** Reset every live connection (the renderer sees an abnormal close, as on a network drop). */ + dropAll: () => number + close: () => Promise +} + +/** A loopback TCP proxy in the test process; the renderer's primary socket is routed through it. */ +export function startTcpProxy(targetPort: number): Promise { + const live = new Set() + + const server = net.createServer(client => { + const upstream = net.connect(targetPort, '127.0.0.1') + live.add(client) + const forget = () => { + live.delete(client) + client.destroy() + upstream.destroy() + } + client.on('error', forget) + upstream.on('error', forget) + client.on('close', forget) + upstream.on('close', forget) + client.pipe(upstream) + upstream.pipe(client) + }) + + return new Promise((resolve, reject) => { + server.on('error', reject) + server.listen(0, '127.0.0.1', () => { + resolve({ + port: (server.address() as net.AddressInfo).port, + connections: () => live.size, + dropAll: () => { + const count = live.size + + for (const socket of [...live]) { + socket.resetAndDestroy() + } + + return count + }, + close: () => + new Promise(done => { + for (const socket of [...live]) { + socket.destroy() + } + + server.close(() => done()) + }) + }) + }) + }) +} + +/** + * Rewrite the PRIMARY gateway WebSocket URL main hands the renderer (initial + * connection and every reconnect mint) so it dials `proxyPort`. REST stays on + * main's IPC bridge. Takes effect on the renderer's next dial (reload/reconnect). + */ +export async function routePrimaryWebSocket(app: ElectronApplication, backendPort: number, proxyPort: number) { + await app.evaluate( + ({ ipcMain }, { backendPort, proxyPort }) => { + const handlers = (ipcMain as any)._invokeHandlers as Map Promise> + const from = `ws://127.0.0.1:${backendPort}/` + const to = `ws://127.0.0.1:${proxyPort}/` + const rewrite = (value: any): any => { + if (typeof value === 'string') { + return value.startsWith(from) ? to + value.slice(from.length) : value + } + + if (Array.isArray(value)) { + return value.map(rewrite) + } + + if (value && typeof value === 'object') { + return Object.fromEntries(Object.entries(value).map(([key, inner]) => [key, rewrite(inner)])) + } + + return value + } + + for (const channel of ['hermes:connection', 'hermes:gateway:ws-url']) { + const original = handlers.get(channel) + + if (!original) { + throw new Error(`no ipc handler ${channel}`) + } + + ipcMain.removeHandler(channel) + ipcMain.handle(channel, async (event: unknown, profile?: unknown, ...rest: unknown[]) => { + const result = await original(event, profile, ...rest) + const primary = profile === undefined || profile === null || profile === '' || profile === 'default' + + return primary ? rewrite(result) : result + }) + } + }, + { backendPort, proxyPort } + ) +} + +/** + * Make main stop tagging `profile`'s route as served by the shared primary + * backend (while still pointing at that same backend). The renderer then dials + * a SECOND WebSocket to the same process for that profile's session calls — + * the #120005 topology — while calls made earlier keep the primary joined. + */ +export async function splitProfileRoute(app: ElectronApplication, profile: string): Promise { + await app.evaluate(({ ipcMain }, profile) => { + const handlers = (ipcMain as any)._invokeHandlers as Map Promise> + const original = handlers.get('hermes:connection:for') + + if (!original) { + throw new Error('no ipc handler hermes:connection:for') + } + + ipcMain.removeHandler('hermes:connection:for') + ipcMain.handle('hermes:connection:for', async (event: unknown, payload: any) => { + const result = await original(event, payload) + + if (payload?.profile === profile && result && typeof result === 'object') { + const { sharedPrimary: _shared, ...rest } = result + + return rest + } + + return result + }) + }, profile) +} + +// ─── Renderer helpers ─────────────────────────────────────────────────── + +export function composer(page: Page) { + return page.locator('[data-slot="composer-root"] [contenteditable="true"]').filter({ visible: true }).first() +} + +/** Composer mounted, no full-viewport overlay above it, window visible. */ +export async function waitForInteractive(app: ElectronApplication, page: Page, timeout = 180_000): Promise { + await expect(composer(page)).toBeVisible({ timeout }) + await page.waitForFunction( + () => { + const el = document.elementFromPoint(window.innerWidth / 2, window.innerHeight / 2) + let node: Element | null = el + + if (!el) { + return false + } + + while (node) { + const cs = window.getComputedStyle(node) + + if (cs.position === 'fixed') { + const r = node.getBoundingClientRect() + + if (r.left <= 0 && r.top <= 0 && r.right >= window.innerWidth && r.bottom >= window.innerHeight) { + return false + } + } + + node = node.parentElement + } + + return true + }, + undefined, + { timeout, polling: 250 } + ) + await expect + .poll( + () => + app + .evaluate(({ BrowserWindow }) => BrowserWindow.getAllWindows()[0]?.isVisible() ?? false) + .catch(() => false), + { timeout, intervals: [250] } + ) + .toBe(true) +} + +/** + * Enter submits (and steers a live turn); Control+Enter queues a follow-up + * behind it. While the gateway is reconnecting the composer keeps the draft + * and ignores Enter (by design), so the press is repeated until the draft is + * accepted — but only while the draft is still in the composer AND no + * prompt.submit carrying it went out on any socket, so a retry can never + * double-submit. + */ +export async function send( + page: Page, + text: string, + key: 'Control+Enter' | 'Enter' = 'Enter', + ws?: WsRecorder +): Promise { + const box = composer(page) + const probe = text.slice(0, 12) + await box.click() + await box.fill(text) + await expect(box).toContainText(probe) + const submitted = () => (ws?.sent ?? []).some(frame => JSON.stringify(frame.params ?? '').includes(probe)) + + await expect + .poll( + async () => { + const draft = (await box.textContent().catch(() => '')) ?? '' + + if (!draft.includes(probe)) { + return 'accepted' + } + + if (!submitted()) { + await box.press(key) + } + + return 'pending' + }, + { timeout: 120_000, intervals: [1_000, 2_000, 4_000], message: `composer accepted ${probe}` } + ) + .toBe('accepted') +} + +export async function currentSessionId(page: Page): Promise { + return page.evaluate(() => decodeURIComponent(location.hash.replace(/^#\/?/, '').split('?')[0] ?? '')) +} + +export interface PersistedMessage { + role: string + content: string +} + +/** + * The stored session id holding a user message with `marker`, read straight + * from the profile's state.db (read-only) — independent of any renderer or + * REST projection. + */ +export function storedSessionForMarker(sandbox: CoreSandbox, profile: string, marker: string): null | string { + const dbPath = + profile === 'default' + ? path.join(sandbox.hermesHome, 'state.db') + : path.join(sandbox.hermesHome, 'profiles', profile, 'state.db') + + if (!fs.existsSync(dbPath)) { + return null + } + + const db = new DatabaseSync(dbPath, { readOnly: true }) + + try { + const row = db + .prepare("SELECT session_id FROM messages WHERE role = 'user' AND content LIKE ? ORDER BY id LIMIT 1") + .get(`%${marker}%`) as undefined | { session_id: string } + + return row?.session_id ?? null + } catch { + // Mid-WAL-checkpoint reads can fail transiently; the caller polls. + return null + } finally { + db.close() + } +} + +/** The backend's persisted display transcript (REST, the same read the renderer hydrates from). */ +export async function persistedTranscript(page: Page, sessionId: string, profile?: string): Promise { + const query = profile ? `&profile=${encodeURIComponent(profile)}` : '' + + const result = await page.evaluate( + async ({ sessionId, query }) => + (window as any).hermesDesktop.api({ path: `/api/sessions/${sessionId}/messages?order=oldest&limit=500${query}` }), + { sessionId, query } + ) + + return (result?.messages ?? []).map((m: any) => ({ + role: String(m.role ?? ''), + content: typeof m.content === 'string' ? m.content : JSON.stringify(m.content ?? '') + })) +} + +export interface RenderedMessage { + role: 'assistant' | 'user' + text: string +} + +/** Every user/assistant bubble in the visible thread, in DOM order. */ +export async function renderedTranscript(page: Page): Promise<{ bubbles: RenderedMessage[]; fullText: string }> { + return page.evaluate(() => { + const viewport = document.querySelector('[data-slot="aui_thread-viewport"]') + + if (!viewport) { + return { bubbles: [], fullText: '' } + } + + const bubbles = [ + ...viewport.querySelectorAll('[data-slot="aui_user-message-root"], [data-slot="aui_assistant-message-root"]') + ].map(el => ({ + role: (el.getAttribute('data-slot') === 'aui_user-message-root' ? 'user' : 'assistant') as 'assistant' | 'user', + text: (el as HTMLElement).innerText.replace(/\s+/g, ' ').trim() + })) + + return { bubbles, fullText: (viewport as HTMLElement).innerText.replace(/\s+/g, ' ') } + }) +} diff --git a/apps/desktop/e2e/core/oracle.ts b/apps/desktop/e2e/core/oracle.ts new file mode 100644 index 0000000000..0680e71144 --- /dev/null +++ b/apps/desktop/e2e/core/oracle.ts @@ -0,0 +1,439 @@ +/** + * The transcript oracle. + * + * Invariant, checked after every transition: each persisted user/assistant + * message is rendered EXACTLY ONCE, in persisted order; nothing scenario-made + * is rendered that is not persisted; the persisted assistant rows are exactly + * what the provider streamed; and on the wire, each turn's concatenated + * message.delta text equals what the provider streamed for that turn (and its + * message.complete text equals the final completion). + * + * Every scenario text carries a unique marker (`U-` for the user, + * `A-` / `Ai-` for replies, `R-` for reasoning), + * so "exactly once" is a count over rendered text, independent of how the + * renderer groups bubbles (live vs hydrated grouping legitimately differs). + * + * Final-state checks converge with a deadline (hydration is asynchronous); + * transient duplicates are caught separately by an in-page MutationObserver + * sampler that is never allowed to see a marker twice. + */ + +import { expect, type Page } from '@playwright/test' + +import type { GatewayEventFrame, PersistedMessage, WsRecorder } from './harness' +import { persistedTranscript } from './harness' +import type { RecordedCompletion, ScriptedProvider } from './provider' + +export const ANY_MARKER_RE = /\b[UAR]\d+i?-[a-z0-9]{3,}\b/g +const FIRST_MARKER_RE = /\b[UAR]\d+i?-[a-z0-9]{3,}\b/ + +const norm = (text: string) => text.replace(/\s+/g, ' ').trim() + +function countOccurrences(haystack: string, needle: string): number { + if (!needle) { + return 0 + } + + let count = 0 + let at = haystack.indexOf(needle) + + while (at !== -1) { + count++ + at = haystack.indexOf(needle, at + needle.length) + } + + return count +} + +function countMarker(haystack: string, marker: string): number { + return (haystack.match(new RegExp(`\\b${marker}\\b`, 'g')) ?? []).length +} + +/** The user marker a reply/reasoning marker answers: A3-x / A3i-x / R3-x → U3-x. */ +export function userMarkerFor(marker: string): string { + return marker.replace(/^[AR](\d+)i?-/, 'U$1-') +} + +// ─── Transient-duplicate sampler ──────────────────────────────────────── + +/** Idempotent; re-run after every reload (the observer dies with the document). */ +export async function installDuplicateSampler(page: Page): Promise { + await page.evaluate(source => { + const w = window as any + + if (w.__coreSampler) { + return + } + + const re = new RegExp(source, 'g') + w.__coreSampler = { samples: 0, violations: [] as { marker: string; count: number; text: string }[] } + let scheduled = false + + const sample = () => { + scheduled = false + const viewports = [...document.querySelectorAll('[data-slot="aui_thread-viewport"]')] as HTMLElement[] + + for (const viewport of viewports) { + if (viewport.getClientRects().length === 0) { + continue + } + + const text = viewport.innerText + w.__coreSampler.samples++ + const counts = new Map() + + for (const match of text.match(re) ?? []) { + counts.set(match, (counts.get(match) ?? 0) + 1) + } + + for (const [marker, count] of counts) { + if (count > 1 && w.__coreSampler.violations.length < 20) { + w.__coreSampler.violations.push({ marker, count, text: text.replace(/\s+/g, ' ').slice(0, 600) }) + } + } + } + } + + new MutationObserver(() => { + if (!scheduled) { + scheduled = true + requestAnimationFrame(sample) + } + }).observe(document.body, { childList: true, subtree: true, characterData: true }) + }, ANY_MARKER_RE.source) +} + +async function samplerViolations(page: Page): Promise<{ samples: number; violations: any[] }> { + return page.evaluate(() => (window as any).__coreSampler ?? { samples: 0, violations: [] }) +} + +// ─── Rendered view ────────────────────────────────────────────────────── + +interface RenderedView { + text: string + userBubbles: string[] +} + +async function renderedView(page: Page): Promise { + return page.evaluate(() => { + const viewport = ([...document.querySelectorAll('[data-slot="aui_thread-viewport"]')] as HTMLElement[]).find( + el => el.getClientRects().length > 0 + ) + + if (!viewport) { + return { text: '', userBubbles: [] } + } + + return { + text: viewport.innerText.replace(/\s+/g, ' '), + userBubbles: ([...viewport.querySelectorAll('[data-slot="aui_user-message-root"]')] as HTMLElement[]).map(el => + el.innerText.replace(/\s+/g, ' ').trim() + ) + } + }) +} + +// ─── Wire check ───────────────────────────────────────────────────────── + +interface WireTurn { + sessionId: string + deltas: string + reasoning: string + complete: null | string +} + +/** + * Backend-truth turns: events deduplicated by (runtime session, seq) across + * every socket (the backend stamps seq once before fan-out), split at + * message.start, in seq order. + */ +export function wireTurns(events: GatewayEventFrame[]): WireTurn[] { + const unique = new Map() + + for (const event of events) { + if (event.seq === null || !event.sessionId) { + continue + } + + const key = `${event.sessionId}#${event.seq}` + + if (!unique.has(key)) { + unique.set(key, event) + } + } + + const bySession = new Map() + + for (const event of unique.values()) { + const list = bySession.get(event.sessionId) ?? [] + list.push(event) + bySession.set(event.sessionId, list) + } + + const turns: WireTurn[] = [] + + for (const [sessionId, list] of bySession) { + list.sort((a, b) => (a.seq ?? 0) - (b.seq ?? 0)) + let turn: null | WireTurn = null + + for (const event of list) { + if (event.type === 'message.start') { + turn = { sessionId, deltas: '', reasoning: '', complete: null } + turns.push(turn) + } else if (!turn) { + continue + } else if (event.type === 'message.delta') { + turn.deltas += String(event.payload?.text ?? '') + } else if (event.type === 'reasoning.delta') { + turn.reasoning += String(event.payload?.text ?? '') + } else if (event.type === 'message.complete') { + turn.complete = String(event.payload?.text ?? '') + } + } + } + + return turns +} + +/** + * Split a turn's streamed text at scenario markers: every scripted completion + * starts with its own marker, so each segment is one completion's output. + */ +function segments(text: string, re: RegExp): { marker: string; text: string }[] { + const out: { marker: string; text: string }[] = [] + const flat = norm(text) + const hits = [...flat.matchAll(new RegExp(re.source, 'g'))] + + hits.forEach((hit, i) => { + const end = i + 1 < hits.length ? hits[i + 1]!.index : flat.length + out.push({ marker: hit[0], text: flat.slice(hit.index, end).trim() }) + }) + + return out +} + +/** + * Backend stream integrity per turn: the concatenated message.delta (and + * reasoning.delta) text is exactly the provider's output, segment by segment — + * each completion once, whole (or a prefix when the backend itself hung up on + * it, e.g. a steer), in order — and message.complete equals the final + * completion. Only whitespace between tool-iteration segments may differ. + */ +function isSubsequence(words: string[], of: string[]): boolean { + let i = 0 + + for (const word of of) { + if (i < words.length && words[i] === word) { + i++ + } + } + + return i === words.length +} + +function wireViolations( + ws: WsRecorder, + provider: ScriptedProvider, + userMarkers: Set, + lossy: Set +): string[] { + const problems: string[] = [] + const byOpening = new Map() + + for (const completion of provider.completions) { + const opening = FIRST_MARKER_RE.exec(completion.sentText)?.[0] + const reasoningOpening = FIRST_MARKER_RE.exec(completion.sentReasoning)?.[0] + + if (opening) { + byOpening.set(opening, completion) + } + + if (reasoningOpening) { + byOpening.set(reasoningOpening, completion) + } + } + + for (const turn of wireTurns(ws.events)) { + const textSegments = segments(turn.deltas, /\bA\d+i?-[a-z0-9]{3,}\b/) + + if (textSegments.length === 0 || !userMarkers.has(userMarkerFor(textSegments.at(-1)!.marker))) { + continue + } + + // A turn that straddled an injected socket drop may have frames that went + // to the dead socket (the renderer recovers them from the persisted + // transcript, which the render oracle checks). Its frames must still be an + // in-order subsequence of what was streamed — never doubled or foreign. + const gapped = lossy.has(userMarkerFor(textSegments.at(-1)!.marker)) + + if (turn.complete === null && !gapped) { + problems.push(`wire ${textSegments.at(-1)!.marker}: turn never completed`) + continue + } + + const seen = new Set() + const check = (kind: string, segs: { marker: string; text: string }[], pick: (c: RecordedCompletion) => string) => { + segs.forEach((seg, i) => { + const completion = byOpening.get(seg.marker) + + if (!completion) { + problems.push(`wire: ${kind} segment ${seg.marker} was never streamed by the provider`) + return + } + + if (seen.has(`${kind}:${seg.marker}`)) { + problems.push(`wire: ${kind} segment ${seg.marker} delivered twice in one turn`) + } + + seen.add(`${kind}:${seg.marker}`) + const sent = norm(pick(completion)) + const partialAllowed = completion.aborted && i < segs.length - 1 + + if (gapped) { + if (!isSubsequence(seg.text.split(' '), sent.split(' '))) { + problems.push(`wire: ${kind} ${JSON.stringify(seg.text)} is not an in-order subsequence of ${JSON.stringify(sent)}`) + } + } else if (partialAllowed ? !sent.startsWith(seg.text) : seg.text !== sent) { + problems.push(`wire: ${kind} ${JSON.stringify(seg.text)} != provider ${JSON.stringify(sent)}${completion.aborted ? ' (aborted)' : ''}`) + } + }) + } + + check('message.delta', textSegments, c => c.sentText) + check('reasoning.delta', segments(turn.reasoning, /\bR\d+-[a-z0-9]{3,}\b/), c => c.sentReasoning) + + const final = byOpening.get(textSegments.at(-1)!.marker) + + if (final && turn.complete !== null && norm(turn.complete) !== norm(final.sentText)) { + problems.push(`wire: message.complete ${JSON.stringify(turn.complete)} != final completion ${JSON.stringify(final.sentText)}`) + } + } + + return problems +} + +// ─── The oracle ───────────────────────────────────────────────────────── + +export interface OracleTarget { + /** Stored session id (the route id). */ + sessionId: string + profile?: string + /** User markers this session must contain (guards against an empty/wrong session passing vacuously). */ + expectUserMarkers: string[] + /** User markers whose turn straddled an injected socket drop (wire frames may be gapped, never doubled). */ + lossyWire?: string[] +} + +function transcriptViolations(persisted: PersistedMessage[], view: RenderedView, target: OracleTarget): string[] { + const problems: string[] = [] + const rows = persisted.filter(m => (m.role === 'user' || m.role === 'assistant') && norm(m.content)) + const persistedMarkers = new Set() + + for (const marker of target.expectUserMarkers) { + if (!rows.some(row => row.role === 'user' && countMarker(row.content, marker) === 1)) { + problems.push(`persisted: user message ${marker} missing from the session`) + } + } + + for (const marker of target.expectUserMarkers) { + if (rows.filter(row => row.role === 'user' && row.content.includes(marker)).length > 1) { + problems.push(`persisted: user message ${marker} stored more than once`) + } + } + + let cursor = -1 + + for (const row of rows) { + const content = norm(row.content) + + for (const marker of content.match(ANY_MARKER_RE) ?? []) { + persistedMarkers.add(marker) + } + + const occurrences = countOccurrences(view.text, content) + + if (occurrences !== 1) { + problems.push(`rendered ${occurrences}x (want 1): ${row.role} ${JSON.stringify(content.slice(0, 80))}`) + continue + } + + const at = view.text.indexOf(content) + + if (at < cursor) { + problems.push(`order: ${row.role} ${JSON.stringify(content.slice(0, 40))} rendered before an earlier message`) + } + + cursor = at + } + + const renderedMarkers = view.text.match(ANY_MARKER_RE) ?? [] + const counts = new Map() + + for (const marker of renderedMarkers) { + counts.set(marker, (counts.get(marker) ?? 0) + 1) + } + + for (const [marker, count] of counts) { + if (count > 1) { + problems.push(`marker ${marker} rendered ${count}x`) + } + + // Reasoning is not part of the display content; every other marker must be persisted. + if (!marker.startsWith('R') && !persistedMarkers.has(marker)) { + problems.push(`rendered but not persisted: ${marker}`) + } + } + + const userRows = rows.filter(row => row.role === 'user').map(row => norm(row.content)) + + if (view.userBubbles.length !== userRows.length) { + problems.push(`user bubbles ${view.userBubbles.length} != persisted user rows ${userRows.length}`) + } + + return problems +} + +/** + * Assert the invariant for the session currently on screen. Converges on the + * final state (deadline), then requires zero transient duplicates since the + * sampler was installed and a clean wire. + */ +export async function assertTranscriptOracle( + page: Page, + ws: WsRecorder, + provider: ScriptedProvider, + target: OracleTarget, + label: string +): Promise { + let last: { problems: string[]; persisted: PersistedMessage[]; view: RenderedView } = { + problems: ['not evaluated'], + persisted: [], + view: { text: '', userBubbles: [] } + } + + await expect + .poll( + async () => { + const persisted = await persistedTranscript(page, target.sessionId, target.profile) + const view = await renderedView(page) + last = { problems: transcriptViolations(persisted, view, target), persisted, view } + + return last.problems + }, + { timeout: 60_000, intervals: [250, 500, 1000], message: `transcript oracle [${label}]` } + ) + .toEqual([]) + .catch(error => { + throw new Error( + `transcript oracle [${label}] failed:\n ${last.problems.join('\n ')}\n` + + `persisted: ${JSON.stringify(last.persisted.map(m => [m.role, m.content.slice(0, 60)]))}\n` + + `rendered: ${JSON.stringify(last.view.text.slice(0, 1500))}\n(${(error as Error).message.split('\n')[0]})` + ) + }) + + const transient = await samplerViolations(page) + expect(transient.violations, `transient duplicate render during [${label}] (${transient.samples} samples)`).toEqual([]) + + const wire = wireViolations(ws, provider, new Set(target.expectUserMarkers), new Set(target.lossyWire ?? [])) + expect(wire, `wire integrity [${label}]`).toEqual([]) +} diff --git a/apps/desktop/e2e/core/playwright.config.ts b/apps/desktop/e2e/core/playwright.config.ts new file mode 100644 index 0000000000..3a28ac90b8 --- /dev/null +++ b/apps/desktop/e2e/core/playwright.config.ts @@ -0,0 +1,31 @@ +import '../fix-electron-tracing' + +import { defineConfig } from '@playwright/test' + +/** + * The core Desktop suite: a small, deterministic, REQUIRED lane. + * + * Deliberately different from ../../playwright.config.ts: + * - retries: 0 — a required job that retries hides exactly the flake it + * should expose (the old lane retried and still went red for weeks). + * - no visual baselines / always-on screenshots; artifacts only on failure. + * - one worker: every spec owns a real Electron + `hermes serve`; running + * them concurrently on a loaded runner is the timing margin we refuse. + * - generous per-test timeout; every wait inside is event-driven with its + * own deadline, so a long timeout never slows a green run. + */ +export default defineConfig({ + testDir: '.', + testMatch: '*.spec.ts', + timeout: 600_000, + expect: { timeout: 60_000 }, + retries: 0, + workers: 1, + fullyParallel: false, + reporter: [['list'], ['html', { open: 'never', outputFolder: '../../playwright-report/core' }]], + outputDir: '../../test-results/core', + use: { + screenshot: 'only-on-failure', + trace: 'retain-on-failure' + } +}) diff --git a/apps/desktop/e2e/core/provider.ts b/apps/desktop/e2e/core/provider.ts new file mode 100644 index 0000000000..15d6105721 --- /dev/null +++ b/apps/desktop/e2e/core/provider.ts @@ -0,0 +1,317 @@ +/** + * Scripted, recording OpenAI-compatible provider for the core Desktop suite. + * + * Unlike the shared tests-js mock (trigger keywords + module-global counters, + * one of the reasons the old lane drifted), every reply here is keyed by the + * unique marker in the turn's own user message, and the step within a turn is + * derived from the request itself (assistant messages after the last user + * message). Two scenarios can never consume each other's script, and a replay + * or retry of the same request gets the same answer. + * + * Every chunk actually written is recorded so the oracle can assert + * "rendered == persisted == what the provider streamed". + */ + +import http from 'node:http' +import type { AddressInfo } from 'node:net' + +export interface Gate { + open: () => void + opened: Promise +} + +export function gate(): Gate { + let open = () => {} + const opened = new Promise(resolve => { + open = resolve + }) + + return { open, opened } +} + +export interface ToolCallSpec { + name: string + args: Record +} + +/** One provider completion. Chunks are streamed in order: reasoning, text, tool calls. */ +export interface Step { + reasoning?: string[] + text?: string[] + toolCalls?: ToolCallSpec[] + /** Hold the stream after the first text (or reasoning) chunk until the gate opens. */ + holdAfterFirstChunk?: Gate +} + +export interface RecordedCompletion { + marker: null | string + step: number + stream: boolean + sentText: string + sentReasoning: string + toolCalls: string[] + finished: boolean + /** The client (the backend) hung up before the stream ended, e.g. an interrupt/steer. */ + aborted: boolean + body: any +} + +export interface ScriptedProvider { + url: string + /** Register the steps a turn keyed by `marker` answers with (step i = i-th completion of that turn). */ + script: (marker: string, steps: Step[]) => void + completions: RecordedCompletion[] + /** Resolves once the first chunk of `marker`'s step has been written to the wire. */ + streamStarted: (marker: string, step?: number) => Promise + close: () => Promise +} + +const MARKER_RE = /\bU\d+-[a-z0-9]+\b/ + +function textOf(content: unknown): string { + if (typeof content === 'string') { + return content + } + + if (Array.isArray(content)) { + return content.map(part => (typeof part?.text === 'string' ? part.text : '')).join('') + } + + return '' +} + +function turnPosition(messages: any[]): { marker: null | string; step: number } { + let lastUser = -1 + + for (let i = messages.length - 1; i >= 0; i--) { + if (messages[i]?.role === 'user') { + lastUser = i + break + } + } + + if (lastUser < 0) { + return { marker: null, step: 0 } + } + + const marker = MARKER_RE.exec(textOf(messages[lastUser].content))?.[0] ?? null + const step = messages.slice(lastUser + 1).filter(m => m?.role === 'assistant').length + + return { marker, step } +} + +function chunk(model: string, delta: Record, finish: null | string = null): string { + return `data: ${JSON.stringify({ + id: 'core-e2e', + object: 'chat.completion.chunk', + created: 0, + model, + choices: [{ index: 0, delta, finish_reason: finish }] + })}\n\n` +} + +const tick = () => new Promise(resolve => setTimeout(resolve, 15)) + +export function startScriptedProvider(): Promise { + const scripts = new Map() + const completions: RecordedCompletion[] = [] + const started = new Map() + + const startedGate = (key: string) => { + let g = started.get(key) + + if (!g) { + g = gate() + started.set(key, g) + } + + return g + } + + async function streamStep(res: http.ServerResponse, model: string, step: Step, rec: RecordedCompletion) { + res.writeHead(200, { 'Content-Type': 'text/event-stream', 'Cache-Control': 'no-cache', Connection: 'keep-alive' }) + res.on('close', () => { + if (!res.writableFinished) { + rec.aborted = true + } + }) + let first = true + + const afterChunk = async () => { + if (first) { + first = false + startedGate(`${rec.marker}#${rec.step}`).open() + + if (step.holdAfterFirstChunk) { + await step.holdAfterFirstChunk.opened + } + } + + await tick() + } + + for (const piece of step.reasoning ?? []) { + if (rec.aborted) { + return + } + + res.write(chunk(model, { reasoning_content: piece })) + rec.sentReasoning += piece + await afterChunk() + } + + for (const piece of step.text ?? []) { + if (rec.aborted) { + return + } + + res.write(chunk(model, { content: piece })) + rec.sentText += piece + await afterChunk() + } + + const calls = step.toolCalls ?? [] + + if (rec.aborted) { + return + } + + if (calls.length > 0) { + res.write( + chunk(model, { + tool_calls: calls.map((call, index) => ({ + index, + id: `call_core_${rec.marker}_${rec.step}_${index}`, + type: 'function', + function: { name: call.name, arguments: JSON.stringify(call.args) } + })) + }) + ) + rec.toolCalls.push(...calls.map(call => call.name)) + await afterChunk() + } + + res.write(chunk(model, {}, calls.length > 0 ? 'tool_calls' : 'stop')) + res.write('data: [DONE]\n\n') + res.end() + rec.finished = true + } + + function plainJson(res: http.ServerResponse, model: string, step: Step, rec: RecordedCompletion) { + const text = (step.text ?? []).join('') + const calls = step.toolCalls ?? [] + rec.sentText = text + rec.sentReasoning = (step.reasoning ?? []).join('') + rec.toolCalls.push(...calls.map(call => call.name)) + rec.finished = true + startedGate(`${rec.marker}#${rec.step}`).open() + res.writeHead(200, { 'Content-Type': 'application/json' }) + res.end( + JSON.stringify({ + id: 'core-e2e', + object: 'chat.completion', + created: 0, + model, + choices: [ + { + index: 0, + finish_reason: calls.length > 0 ? 'tool_calls' : 'stop', + message: { + role: 'assistant', + content: text || null, + ...(rec.sentReasoning ? { reasoning_content: rec.sentReasoning } : {}), + ...(calls.length > 0 + ? { + tool_calls: calls.map((call, index) => ({ + id: `call_core_${rec.marker}_${rec.step}_${index}`, + type: 'function', + function: { name: call.name, arguments: JSON.stringify(call.args) } + })) + } + : {}) + } + } + ], + usage: { prompt_tokens: 10, completion_tokens: 10, total_tokens: 20 } + }) + ) + } + + const server = http.createServer((req, res) => { + if (req.method === 'GET' && req.url?.startsWith('/v1/models')) { + res.writeHead(200, { 'Content-Type': 'application/json' }) + res.end(JSON.stringify({ object: 'list', data: [{ id: 'mock-model', object: 'model', created: 0, owned_by: 'core' }] })) + + return + } + + if (req.method !== 'POST' || !req.url?.startsWith('/v1/chat/completions')) { + res.writeHead(404, { 'Content-Type': 'application/json' }) + res.end(JSON.stringify({ error: 'not found' })) + + return + } + + let raw = '' + req.on('data', (data: Buffer) => { + raw += data.toString() + }) + req.on('end', () => { + let body: any = {} + + try { + body = JSON.parse(raw) + } catch { + body = {} + } + + const messages: any[] = Array.isArray(body.messages) ? body.messages : [] + const { marker, step: stepIndex } = turnPosition(messages) + const steps = marker ? scripts.get(marker) : undefined + // Unscripted traffic (auxiliary calls, a marker-less prompt) gets a + // fixed short answer; it is recorded, never matched to a scenario. + const step: Step = steps?.[Math.min(stepIndex, steps.length - 1)] ?? { text: ['ok'] } + const model = typeof body.model === 'string' ? body.model : 'mock-model' + + const rec: RecordedCompletion = { + marker: steps ? marker : null, + step: stepIndex, + stream: body.stream === true, + sentText: '', + sentReasoning: '', + toolCalls: [], + finished: false, + aborted: false, + body + } + + completions.push(rec) + + if (rec.stream) { + void streamStep(res, model, step, rec).catch(() => res.destroy()) + } else { + plainJson(res, model, step, rec) + } + }) + }) + + return new Promise((resolve, reject) => { + server.on('error', reject) + server.listen(0, '127.0.0.1', () => { + const { port } = server.address() as AddressInfo + resolve({ + url: `http://127.0.0.1:${port}`, + script: (marker, steps) => { + scripts.set(marker, steps) + }, + completions, + streamStarted: (marker, step = 0) => startedGate(`${marker}#${step}`).opened, + close: () => + new Promise(done => { + server.closeAllConnections?.() + server.close(() => done()) + }) + }) + }) + }) +} diff --git a/apps/desktop/e2e/core/transcript-integrity.spec.ts b/apps/desktop/e2e/core/transcript-integrity.spec.ts new file mode 100644 index 0000000000..e8cc0333d5 --- /dev/null +++ b/apps/desktop/e2e/core/transcript-integrity.spec.ts @@ -0,0 +1,292 @@ +/** + * C2 core: transcript integrity across every transition that has shipped a + * duplicate / vanishing / reordered message bug. + * + * One real Electron app + one real `hermes serve` backend (only the LLM is + * faked, by a scripted recording provider). After every transition the + * transcript oracle (./oracle.ts) asserts: each persisted user/assistant + * message is rendered exactly once, in order, nothing unpersisted is + * rendered, no marker was ever rendered twice even transiently, and the + * backend's concatenated deltas equal what the provider streamed. + * + * Transitions (in order, sharing the app so each builds on the last): + * stream → tool-call turn → reasoning turn → steer mid-stream → + * queued follow-up → + * session switch mid-stream → warm resume → reload → + * WebSocket drop + reconnect mid-stream → second socket to the same + * backend (#120005 topology) → final reload re-checks every session. + */ + +import * as path from 'node:path' + +import { expect, type Page, test } from '@playwright/test' + +import { + coreAppEnv, + createCoreSandbox, + currentSessionId, + launchCoreApp, + recordWebSockets, + routePrimaryWebSocket, + send, + splitProfileRoute, + startTcpProxy, + storedSessionForMarker, + waitForInteractive, + writeProviderHome +} from './harness' +import { assertTranscriptOracle, installDuplicateSampler, type OracleTarget } from './oracle' +import { gate, startScriptedProvider } from './provider' + +const nonce = Math.random().toString(36).slice(2, 8).replace(/[^a-z0-9]/g, 'x').padEnd(4, 'q') +const U = (n: number) => `U${n}-${nonce}` +const A = (n: number) => `A${n}-${nonce}` +const AI = (n: number) => `A${n}i-${nonce}` +const R = (n: number) => `R${n}-${nonce}` + +const words = (marker: string, ...rest: string[]) => [`${marker} `, ...rest.map((w, i) => (i === rest.length - 1 ? w : `${w} `))] + +function viewport(page: Page) { + return page.locator('[data-slot="aui_thread-viewport"]').filter({ visible: true }).first() +} + +async function openSession(page: Page, sessionId: string, mustShow: string) { + await page.evaluate(id => { + window.location.hash = `#/${encodeURIComponent(id)}` + }, sessionId) + await expect.poll(() => currentSessionId(page)).toBe(sessionId) + await expect(viewport(page)).toContainText(mustShow, { timeout: 60_000 }) +} + +test('transcript oracle holds across every transition', async () => { + const provider = await startScriptedProvider() + const sandbox = createCoreSandbox('transcript') + writeProviderHome(sandbox.hermesHome, provider.url) + writeProviderHome(path.join(sandbox.hermesHome, 'profiles', 'p2'), provider.url) + const { app, page } = await launchCoreApp(coreAppEnv(sandbox)) + const ws = recordWebSockets(page) + const proxies: { close: () => Promise }[] = [] + + const finished = (marker: string, step = 0) => + expect + .poll(() => provider.completions.some(c => c.marker === marker && c.step === step && c.finished), { + timeout: 120_000, + message: `provider finished ${marker} step ${step}` + }) + .toBe(true) + + const sessionA: OracleTarget = { sessionId: '', expectUserMarkers: [] } + const sessionB: OracleTarget = { sessionId: '', expectUserMarkers: [] } + + try { + await waitForInteractive(app, page) + await installDuplicateSampler(page) + + await test.step('stream: multi-chunk reply', async () => { + provider.script(U(1), [{ text: words(A(1), 'alpha', 'bravo', 'charlie', 'delta', 'echo') }]) + await send(page, `${U(1)} hello`, 'Enter', ws) + await finished(U(1)) + await expect.poll(() => currentSessionId(page)).not.toBe('') + sessionA.sessionId = await currentSessionId(page) + sessionA.expectUserMarkers.push(U(1)) + await assertTranscriptOracle(page, ws, provider, sessionA, 'stream') + }) + + await test.step('tool-call turn: interim text + real terminal tool + final', async () => { + provider.script(U(2), [ + { text: words(AI(2), 'checking', 'first'), toolCalls: [{ name: 'terminal', args: { command: 'echo core-tool-ok' } }] }, + { text: words(A(2), 'tool', 'said', 'ok') } + ]) + await send(page, `${U(2)} run a tool`, 'Enter', ws) + await finished(U(2), 1) + sessionA.expectUserMarkers.push(U(2)) + await assertTranscriptOracle(page, ws, provider, sessionA, 'tool-call turn') + }) + + await test.step('reasoning turn: reasoning deltas never leak into or double the reply', async () => { + provider.script(U(3), [{ reasoning: words(R(3), 'weighing', 'options'), text: words(A(3), 'reasoned', 'answer') }]) + await send(page, `${U(3)} think first`, 'Enter', ws) + await finished(U(3)) + sessionA.expectUserMarkers.push(U(3)) + await assertTranscriptOracle(page, ws, provider, sessionA, 'reasoning turn') + }) + + await test.step('steer mid-stream: Enter while busy redirects the live turn', async () => { + const hold = gate() + provider.script(U(4), [{ text: words(A(4), 'slow', 'first', 'reply'), holdAfterFirstChunk: hold }]) + provider.script(U(5), [{ text: words(A(5), 'steered', 'reply') }]) + await send(page, `${U(4)} slow one`, 'Enter', ws) + await provider.streamStarted(U(4)) + await expect(viewport(page)).toContainText(A(4)) + await send(page, `${U(5)} change course`, 'Enter', ws) + await finished(U(5)) + hold.open() + sessionA.expectUserMarkers.push(U(4), U(5)) + await assertTranscriptOracle(page, ws, provider, sessionA, 'steer mid-stream') + }) + + await test.step('queued follow-up: Ctrl+Enter while busy runs after the live turn, once', async () => { + const hold = gate() + provider.script(U(11), [{ text: words(A(11), 'long', 'running', 'reply'), holdAfterFirstChunk: hold }]) + provider.script(U(12), [{ text: words(A(12), 'queued', 'reply') }]) + await send(page, `${U(11)} long one`, 'Enter', ws) + await provider.streamStarted(U(11)) + await expect(viewport(page)).toContainText(A(11)) + await send(page, `${U(12)} after that`, 'Control+Enter', ws) + hold.open() + await finished(U(11)) + await finished(U(12)) + // The queued message is a turn of its own, strictly after the first. + const u11 = provider.completions.findIndex(c => c.marker === U(11)) + const u12 = provider.completions.findIndex(c => c.marker === U(12)) + expect(u12).toBeGreaterThan(u11) + sessionA.expectUserMarkers.push(U(11), U(12)) + await assertTranscriptOracle(page, ws, provider, sessionA, 'queued follow-up') + }) + + await test.step('session switch mid-stream: the away session completes exactly once', async () => { + // NEW_CHAT_ROUTE ('/') — the route the sidebar's New session button opens. + await page.evaluate(() => { + window.location.hash = '#/' + }) + await expect.poll(() => currentSessionId(page)).toBe('') + await expect(viewport(page)).not.toContainText(U(1)) + const hold = gate() + provider.script(U(6), [{ text: words(A(6), 'finished', 'while', 'away'), holdAfterFirstChunk: hold }]) + await send(page, `${U(6)} in session b`, 'Enter', ws) + await provider.streamStarted(U(6)) + await expect.poll(() => currentSessionId(page)).not.toBe('') + sessionB.sessionId = await currentSessionId(page) + expect(sessionB.sessionId).not.toBe(sessionA.sessionId) + sessionB.expectUserMarkers.push(U(6)) + await openSession(page, sessionA.sessionId, A(12)) + hold.open() + await finished(U(6)) + await assertTranscriptOracle(page, ws, provider, sessionA, 'switch: session A while B streams') + await openSession(page, sessionB.sessionId, U(6)) + await assertTranscriptOracle(page, ws, provider, sessionB, 'switch: back to B after it completed away') + }) + + await test.step('warm resume: revisit cached sessions and continue one', async () => { + await openSession(page, sessionA.sessionId, A(12)) + await assertTranscriptOracle(page, ws, provider, sessionA, 'warm resume A') + await openSession(page, sessionB.sessionId, A(6)) + provider.script(U(7), [{ text: words(A(7), 'warm', 'continue') }]) + await send(page, `${U(7)} continue b`, 'Enter', ws) + await finished(U(7)) + sessionB.expectUserMarkers.push(U(7)) + await assertTranscriptOracle(page, ws, provider, sessionB, 'warm resume B + new turn') + }) + + // From here on the primary socket dials through a loopback proxy the test + // controls, so a network drop can be injected mid-stream. + const backendPort = Number(new URL(ws.sockets[0]!.url).port) + const proxy = await startTcpProxy(backendPort) + proxies.push(proxy) + await routePrimaryWebSocket(app, backendPort, proxy.port) + + await test.step('reload: hydrated transcript equals persisted', async () => { + await page.reload() + await waitForInteractive(app, page) + await installDuplicateSampler(page) + await expect.poll(() => proxy.connections()).toBeGreaterThan(0) + await openSession(page, sessionB.sessionId, A(7)) + await assertTranscriptOracle(page, ws, provider, sessionB, 'reload B') + await openSession(page, sessionA.sessionId, A(12)) + await assertTranscriptOracle(page, ws, provider, sessionA, 'reload A') + }) + + await test.step('WebSocket drop + reconnect mid-stream: no lost, doubled or replayed text', async () => { + const hold = gate() + provider.script(U(8), [{ text: words(A(8), 'survives', 'the', 'drop'), holdAfterFirstChunk: hold }]) + await send(page, `${U(8)} across a drop`, 'Enter', ws) + await provider.streamStarted(U(8)) + await expect(viewport(page)).toContainText(A(8)) + const socketsBefore = ws.sockets.length + expect(proxy.dropAll()).toBeGreaterThan(0) + // The renderer must dial a fresh socket (through the proxy) on its own. + await expect.poll(() => ws.sockets.length, { timeout: 60_000 }).toBeGreaterThan(socketsBefore) + await expect.poll(() => proxy.connections(), { timeout: 60_000 }).toBeGreaterThan(0) + hold.open() + await finished(U(8)) + sessionA.expectUserMarkers.push(U(8)) + sessionA.lossyWire = [U(8)] + await assertTranscriptOracle(page, ws, provider, sessionA, 'ws drop + reconnect') + }) + + // Sockets that delivered the message.complete of the turn whose reply opens with `marker`. + const completeSockets = (marker: string) => + new Set(ws.events.filter(e => e.type === 'message.complete' && String(e.payload?.text ?? '').startsWith(marker)).map(e => e.socket)) + + let sessionP2: OracleTarget = { sessionId: '', profile: 'p2', expectUserMarkers: [] } + + await test.step('non-default profile on the shared host backend: one socket per session', async () => { + await page.getByRole('button', { name: 'Bots', exact: true }).or(page.getByRole('tab', { name: 'Bots', exact: true })).first().click() + const row = page.locator('[data-slot="bots-roster"] [data-roster-key="local::p2"]') + await expect(row).toBeVisible({ timeout: 60_000 }) + await row.click() + await expect(page.getByRole('tab', { name: /Draft/ })).toBeVisible({ timeout: 60_000 }) + provider.script(U(9), [{ text: words(A(9), 'profile', 'two', 'first') }]) + await send(page, `${U(9)} on p2`, 'Enter', ws) + await finished(U(9)) + await expect(viewport(page)).toContainText(A(9)) + let p2Session: null | string = null + await expect + .poll(() => (p2Session = storedSessionForMarker(sandbox, 'p2', U(9))), { message: 'p2 turn persisted in the p2 state.db' }) + .not.toBeNull() + sessionP2 = { sessionId: p2Session!, profile: 'p2', expectUserMarkers: [U(9)] } + await assertTranscriptOracle(page, ws, provider, sessionP2, 'p2 first turn') + + // #120005: the SECOND message of a non-default-profile chat is where a + // session-owner call used to dial a second socket to the same process. + provider.script(U(13), [{ text: words(A(13), 'profile', 'two', 'second') }]) + await send(page, `${U(13)} again on p2`, 'Enter', ws) + await finished(U(13)) + sessionP2.expectUserMarkers.push(U(13)) + await assertTranscriptOracle(page, ws, provider, sessionP2, 'p2 second turn') + expect([...completeSockets(A(9))].length, 'p2 turn 1 delivered on exactly one socket').toBe(1) + expect([...completeSockets(A(13))].length, 'p2 turn 2 delivered on exactly one socket (no 2nd socket to one backend)').toBe(1) + }) + + await test.step('forced second socket to the same backend: every event still renders once', async () => { + // Force the #120005 topology regardless of routing: from the next + // session call on, the p2 route is served by a second socket to the same + // backend while the primary stays joined. The renderer must still apply + // every event exactly once (tool turn → interim bubble + final). + const socketsBefore = ws.sockets.length + await splitProfileRoute(app, 'p2') + provider.script(U(10), [ + { text: words(AI(10), 'looking', 'it', 'up'), toolCalls: [{ name: 'terminal', args: { command: 'echo core-two-sockets' } }] }, + { text: words(A(10), 'two', 'sockets', 'one', 'render') } + ]) + await send(page, `${U(10)} tool on p2`, 'Enter', ws) + await finished(U(10), 1) + await expect.poll(() => ws.sockets.length).toBeGreaterThan(socketsBefore) + // Precondition of the scenario: the turn really was fanned out to 2+ sockets. + await expect + .poll(() => completeSockets(A(10)).size, { timeout: 60_000, message: 'the forced turn reached more than one socket' }) + .toBeGreaterThan(1) + sessionP2.expectUserMarkers.push(U(10)) + await assertTranscriptOracle(page, ws, provider, sessionP2, 'p2 tool turn over two sockets') + }) + + await test.step('final reload: every session re-hydrates exactly once', async () => { + await page.reload() + await waitForInteractive(app, page) + await installDuplicateSampler(page) + await openSession(page, sessionA.sessionId, A(8)) + await assertTranscriptOracle(page, ws, provider, sessionA, 'final reload A') + await openSession(page, sessionB.sessionId, A(7)) + await assertTranscriptOracle(page, ws, provider, sessionB, 'final reload B') + }) + } finally { + await app.close().catch(() => undefined) + + for (const proxy of proxies) { + await proxy.close() + } + + await provider.close() + sandbox.cleanup() + } +}) From 75908c1d234e06d31c6e67f3ab143070a99565ab Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:08:37 -0700 Subject: [PATCH 023/112] test(desktop-core): boot handshake, supervised respawn, zero orphans (C5) Class: Desktop boot / backend lifecycle - stuck boots, crash-loop or double respawn, orphaned hermes serve / tool processes piling up after quit. Real app + real backend: interactive composer and first turn with exactly one backend; kill -9 the backend -> exactly one supervised replacement (background /proc census catches transient extra spawns) and a working turn; quit mid-turn with a running terminal-tool child -> zero processes carrying the sandbox HERMES_HOME (orphans reparented to init included); relaunch the same home 3x -> one backend per boot, zero after each quit, transcript cold-hydrates exactly once. --- apps/desktop/e2e/core/boot-lifecycle.spec.ts | 237 +++++++++++++++++++ 1 file changed, 237 insertions(+) create mode 100644 apps/desktop/e2e/core/boot-lifecycle.spec.ts diff --git a/apps/desktop/e2e/core/boot-lifecycle.spec.ts b/apps/desktop/e2e/core/boot-lifecycle.spec.ts new file mode 100644 index 0000000000..27c3628201 --- /dev/null +++ b/apps/desktop/e2e/core/boot-lifecycle.spec.ts @@ -0,0 +1,237 @@ +/** + * C5 core: boot handshake and backend process lifecycle. + * + * Real Electron + real `hermes serve`; only the LLM is faked. Process facts + * come from a /proc census of every process carrying this sandbox's + * HERMES_HOME (so an orphan reparented to init is still counted), sampled + * continuously in the background so a transient extra spawn is seen too. + * + * 1. boot: composer interactive, exactly one backend, first turn completes + * and passes the transcript oracle. + * 2. kill -9 the backend: the supervisor respawns EXACTLY one replacement + * (no crash-loop, no double spawn), and the app serves a new turn. + * 3. quit while a turn is streaming and a tool subprocess is running: zero + * sandbox processes remain — no backend, no tool child, no Electron helper. + * 4. relaunch the same HERMES_HOME repeatedly: each boot has exactly one + * backend, each quit leaves zero processes, and the transcript persisted + * by the first launch cold-hydrates exactly once every time. + */ + +import * as fs from 'node:fs' + +import { expect, test } from '@playwright/test' + +import { + backendProcesses, + coreAppEnv, + createCoreSandbox, + currentSessionId, + launchCoreApp, + type ProcInfo, + recordWebSockets, + sandboxProcesses, + send, + waitForInteractive, + writeProviderHome +} from './harness' +import { assertTranscriptOracle, installDuplicateSampler, type OracleTarget } from './oracle' +import { gate, startScriptedProvider } from './provider' + +const nonce = Math.random().toString(36).slice(2, 8).replace(/[^a-z0-9]/g, 'x').padEnd(4, 'q') +const U = (n: number) => `U${n}-${nonce}` +const A = (n: number) => `A${n}-${nonce}` +const TOOL_TAG = `core-orphan-${nonce}` + +/** Every live process whose command line carries `tag` (tool children may scrub HERMES_HOME). */ +function taggedProcesses(tag: string): ProcInfo[] { + const out: ProcInfo[] = [] + + for (const entry of fs.readdirSync('/proc')) { + const pid = Number(entry) + + if (!Number.isInteger(pid) || pid === process.pid) { + continue + } + + try { + const cmdline = fs.readFileSync(`/proc/${pid}/cmdline`, 'utf8').split('\0').join(' ').trim() + + if (cmdline.includes(tag)) { + out.push({ pid, ppid: 0, cmdline }) + } + } catch { + /* exited mid-scan */ + } + } + + return out +} + +test('boot handshake, supervised respawn, and zero orphans on quit', async () => { + const provider = await startScriptedProvider() + const sandbox = createCoreSandbox('boot') + writeProviderHome(sandbox.hermesHome, provider.url) + const { app, page } = await launchCoreApp(coreAppEnv(sandbox)) + const ws = recordWebSockets(page) + let closed = false + + // Background census: every backend pid ever observed. + const seenBackends = new Set() + const census = setInterval(() => { + for (const proc of backendProcesses(sandbox)) { + seenBackends.add(proc.pid) + } + }, 100) + + const finished = (marker: string, step = 0) => + expect + .poll(() => provider.completions.some(c => c.marker === marker && c.step === step && c.finished), { + timeout: 120_000, + message: `provider finished ${marker} step ${step}` + }) + .toBe(true) + + try { + const session: OracleTarget = { sessionId: '', expectUserMarkers: [] } + + await test.step('boot: interactive composer, one backend, first turn', async () => { + await waitForInteractive(app, page) + await installDuplicateSampler(page) + await expect.poll(() => backendProcesses(sandbox).length, { message: 'exactly one backend after boot' }).toBe(1) + provider.script(U(1), [{ text: [`${A(1)} `, 'booted ', 'and ', 'answered'] }]) + await send(page, `${U(1)} first`, 'Enter', ws) + await finished(U(1)) + await expect.poll(() => currentSessionId(page)).not.toBe('') + session.sessionId = await currentSessionId(page) + session.expectUserMarkers.push(U(1)) + await assertTranscriptOracle(page, ws, provider, session, 'boot first turn') + expect(seenBackends.size, `backend pids seen during boot: ${[...seenBackends]}`).toBe(1) + }) + + await test.step('kill -9 backend: exactly one supervised respawn, app serves again', async () => { + const [victim] = backendProcesses(sandbox) + expect(victim).toBeTruthy() + process.kill(victim!.pid, 'SIGKILL') + await expect + .poll(() => backendProcesses(sandbox).map(p => p.pid).filter(pid => pid !== victim!.pid).length, { + timeout: 120_000, + message: 'a replacement backend is spawned' + }) + .toBe(1) + await waitForInteractive(app, page) + provider.script(U(2), [{ text: [`${A(2)} `, 'after ', 'respawn'] }]) + await send(page, `${U(2)} still there`, 'Enter', ws) + await finished(U(2)) + session.expectUserMarkers.push(U(2)) + await assertTranscriptOracle(page, ws, provider, session, 'after backend respawn') + const alive = backendProcesses(sandbox) + expect(alive.length, `live backends after recovery: ${JSON.stringify(alive)}`).toBe(1) + // Initial + exactly one replacement, ever — a crash loop or a racing + // second spawn would add pids here even if they died again. + expect(seenBackends.size, `backend pids ever seen: ${[...seenBackends]}`).toBe(2) + }) + + await test.step('quit mid-turn with a running tool child: zero processes remain', async () => { + const hold = gate() + provider.script(U(3), [ + { text: [`${A(3)} `, 'starting ', 'tool'], toolCalls: [{ name: 'terminal', args: { command: `exec -a ${TOOL_TAG} sleep 3600` } }] }, + { text: [`${A(3)}b `, 'never ', 'reached'], holdAfterFirstChunk: hold } + ]) + await send(page, `${U(3)} long tool`, 'Enter', ws) + await expect.poll(() => taggedProcesses(TOOL_TAG).length, { timeout: 120_000, message: 'tool child running' }).toBeGreaterThan(0) + expect(sandboxProcesses(sandbox).length).toBeGreaterThan(0) + + await app.close() + closed = true + clearInterval(census) + await expect + .poll(() => [...sandboxProcesses(sandbox), ...taggedProcesses(TOOL_TAG)].map(p => `${p.pid} ${p.cmdline.slice(0, 120)}`), { + timeout: 60_000, + message: 'no sandbox process (backend, tool child, Electron helper) survives quit' + }) + .toEqual([]) + hold.open() + }) + } finally { + clearInterval(census) + + if (!closed) { + await app.close().catch(() => undefined) + } + + // Never leave a test-made process behind even on failure (only our own tag / sandbox). + for (const proc of [...sandboxProcesses(sandbox), ...taggedProcesses(TOOL_TAG)]) { + try { + process.kill(proc.pid, 'SIGKILL') + } catch { + /* already gone */ + } + } + + await provider.close() + sandbox.cleanup() + } +}) + +test('relaunching the same home: one backend per boot, zero after each quit, transcript intact', async () => { + const provider = await startScriptedProvider() + const sandbox = createCoreSandbox('relaunch') + writeProviderHome(sandbox.hermesHome, provider.url) + const session: OracleTarget = { sessionId: '', expectUserMarkers: [U(1)] } + let live: Awaited> | null = null + + try { + for (let launch = 1; launch <= 3; launch++) { + await test.step(`launch ${launch}`, async () => { + live = await launchCoreApp(coreAppEnv(sandbox)) + const { app, page } = live + const ws = recordWebSockets(page) + await waitForInteractive(app, page) + await installDuplicateSampler(page) + await expect.poll(() => backendProcesses(sandbox).length, { message: `one backend on launch ${launch}` }).toBe(1) + + if (launch === 1) { + provider.script(U(1), [{ text: [`${A(1)} `, 'persisted ', 'across ', 'launches'] }]) + await send(page, `${U(1)} remember me`, 'Enter', ws) + await expect + .poll(() => provider.completions.some(c => c.marker === U(1) && c.finished), { timeout: 120_000 }) + .toBe(true) + await expect.poll(() => currentSessionId(page)).not.toBe('') + session.sessionId = await currentSessionId(page) + } else { + await page.evaluate(id => { + window.location.hash = `#/${encodeURIComponent(id)}` + }, session.sessionId) + await expect(page.locator('[data-slot="aui_thread-viewport"]').filter({ visible: true }).first()).toContainText(A(1), { + timeout: 60_000 + }) + } + + await assertTranscriptOracle(page, ws, provider, session, `launch ${launch}`) + await app.close() + live = null + await expect + .poll(() => sandboxProcesses(sandbox).map(p => `${p.pid} ${p.cmdline.slice(0, 120)}`), { + timeout: 60_000, + message: `no sandbox process survives quit #${launch}` + }) + .toEqual([]) + }) + } + } finally { + if (live) { + await (live as Awaited>).app.close().catch(() => undefined) + } + + for (const proc of sandboxProcesses(sandbox)) { + try { + process.kill(proc.pid, 'SIGKILL') + } catch { + /* already gone */ + } + } + + await provider.close() + sandbox.cleanup() + } +}) From 2bfa29b697304fef4b27f7fd7cf2b5cc83446aeb Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:03:20 -0700 Subject: [PATCH 024/112] test(desktop-core): one live socket per backend; sampler names the duplicate bubbles The non-default-profile step now asserts the renderer holds exactly one live socket to the host backend (red when 0bb539b4725, #120006, is reverted); the earlier per-turn check alone missed that regression. Transient-duplicate violations now report which bubble ids carry the copies. --- apps/desktop/e2e/core/oracle.ts | 40 ++++++++++-- apps/desktop/e2e/core/provider.ts | 6 +- .../e2e/core/transcript-integrity.spec.ts | 63 ++++++++++++++++--- 3 files changed, 93 insertions(+), 16 deletions(-) diff --git a/apps/desktop/e2e/core/oracle.ts b/apps/desktop/e2e/core/oracle.ts index 0680e71144..09c5acc859 100644 --- a/apps/desktop/e2e/core/oracle.ts +++ b/apps/desktop/e2e/core/oracle.ts @@ -88,7 +88,25 @@ export async function installDuplicateSampler(page: Page): Promise { for (const [marker, count] of counts) { if (count > 1 && w.__coreSampler.violations.length < 20) { - w.__coreSampler.violations.push({ marker, count, text: text.replace(/\s+/g, ' ').slice(0, 600) }) + // Where the copies live: one bubble with doubled text vs two bubbles. + const bubbles = ( + [ + ...viewport.querySelectorAll( + '[data-slot="aui_user-message-root"], [data-slot="aui_assistant-message-root"]' + ) + ] as HTMLElement[] + ) + .filter(el => el.innerText.includes(marker)) + .map(el => `${el.getAttribute('data-slot')}#${el.getAttribute('data-message-id') ?? el.id ?? ''}`) + + w.__coreSampler.violations.push({ + marker, + count, + at: Math.round(performance.now()), + route: location.hash, + bubbles, + text: text.replace(/\s+/g, ' ').slice(0, 600) + }) } } } @@ -268,16 +286,19 @@ function wireViolations( if (turn.complete === null && !gapped) { problems.push(`wire ${textSegments.at(-1)!.marker}: turn never completed`) + continue } const seen = new Set() + const check = (kind: string, segs: { marker: string; text: string }[], pick: (c: RecordedCompletion) => string) => { segs.forEach((seg, i) => { const completion = byOpening.get(seg.marker) if (!completion) { problems.push(`wire: ${kind} segment ${seg.marker} was never streamed by the provider`) + return } @@ -291,10 +312,14 @@ function wireViolations( if (gapped) { if (!isSubsequence(seg.text.split(' '), sent.split(' '))) { - problems.push(`wire: ${kind} ${JSON.stringify(seg.text)} is not an in-order subsequence of ${JSON.stringify(sent)}`) + problems.push( + `wire: ${kind} ${JSON.stringify(seg.text)} is not an in-order subsequence of ${JSON.stringify(sent)}` + ) } } else if (partialAllowed ? !sent.startsWith(seg.text) : seg.text !== sent) { - problems.push(`wire: ${kind} ${JSON.stringify(seg.text)} != provider ${JSON.stringify(sent)}${completion.aborted ? ' (aborted)' : ''}`) + problems.push( + `wire: ${kind} ${JSON.stringify(seg.text)} != provider ${JSON.stringify(sent)}${completion.aborted ? ' (aborted)' : ''}` + ) } }) } @@ -305,7 +330,9 @@ function wireViolations( const final = byOpening.get(textSegments.at(-1)!.marker) if (final && turn.complete !== null && norm(turn.complete) !== norm(final.sentText)) { - problems.push(`wire: message.complete ${JSON.stringify(turn.complete)} != final completion ${JSON.stringify(final.sentText)}`) + problems.push( + `wire: message.complete ${JSON.stringify(turn.complete)} != final completion ${JSON.stringify(final.sentText)}` + ) } } @@ -354,6 +381,7 @@ function transcriptViolations(persisted: PersistedMessage[], view: RenderedView, if (occurrences !== 1) { problems.push(`rendered ${occurrences}x (want 1): ${row.role} ${JSON.stringify(content.slice(0, 80))}`) + continue } @@ -432,7 +460,9 @@ export async function assertTranscriptOracle( }) const transient = await samplerViolations(page) - expect(transient.violations, `transient duplicate render during [${label}] (${transient.samples} samples)`).toEqual([]) + expect(transient.violations, `transient duplicate render during [${label}] (${transient.samples} samples)`).toEqual( + [] + ) const wire = wireViolations(ws, provider, new Set(target.expectUserMarkers), new Set(target.lossyWire ?? [])) expect(wire, `wire integrity [${label}]`).toEqual([]) diff --git a/apps/desktop/e2e/core/provider.ts b/apps/desktop/e2e/core/provider.ts index 15d6105721..414988edd9 100644 --- a/apps/desktop/e2e/core/provider.ts +++ b/apps/desktop/e2e/core/provider.ts @@ -22,6 +22,7 @@ export interface Gate { export function gate(): Gate { let open = () => {} + const opened = new Promise(resolve => { open = resolve }) @@ -86,6 +87,7 @@ function turnPosition(messages: any[]): { marker: null | string; step: number } for (let i = messages.length - 1; i >= 0; i--) { if (messages[i]?.role === 'user') { lastUser = i + break } } @@ -240,7 +242,9 @@ export function startScriptedProvider(): Promise { const server = http.createServer((req, res) => { if (req.method === 'GET' && req.url?.startsWith('/v1/models')) { res.writeHead(200, { 'Content-Type': 'application/json' }) - res.end(JSON.stringify({ object: 'list', data: [{ id: 'mock-model', object: 'model', created: 0, owned_by: 'core' }] })) + res.end( + JSON.stringify({ object: 'list', data: [{ id: 'mock-model', object: 'model', created: 0, owned_by: 'core' }] }) + ) return } diff --git a/apps/desktop/e2e/core/transcript-integrity.spec.ts b/apps/desktop/e2e/core/transcript-integrity.spec.ts index e8cc0333d5..882508232e 100644 --- a/apps/desktop/e2e/core/transcript-integrity.spec.ts +++ b/apps/desktop/e2e/core/transcript-integrity.spec.ts @@ -38,13 +38,20 @@ import { import { assertTranscriptOracle, installDuplicateSampler, type OracleTarget } from './oracle' import { gate, startScriptedProvider } from './provider' -const nonce = Math.random().toString(36).slice(2, 8).replace(/[^a-z0-9]/g, 'x').padEnd(4, 'q') +const nonce = Math.random() + .toString(36) + .slice(2, 8) + .replace(/[^a-z0-9]/g, 'x') + .padEnd(4, 'q') const U = (n: number) => `U${n}-${nonce}` const A = (n: number) => `A${n}-${nonce}` const AI = (n: number) => `A${n}i-${nonce}` const R = (n: number) => `R${n}-${nonce}` -const words = (marker: string, ...rest: string[]) => [`${marker} `, ...rest.map((w, i) => (i === rest.length - 1 ? w : `${w} `))] +const words = (marker: string, ...rest: string[]) => [ + `${marker} `, + ...rest.map((w, i) => (i === rest.length - 1 ? w : `${w} `)) +] function viewport(page: Page) { return page.locator('[data-slot="aui_thread-viewport"]').filter({ visible: true }).first() @@ -94,7 +101,10 @@ test('transcript oracle holds across every transition', async () => { await test.step('tool-call turn: interim text + real terminal tool + final', async () => { provider.script(U(2), [ - { text: words(AI(2), 'checking', 'first'), toolCalls: [{ name: 'terminal', args: { command: 'echo core-tool-ok' } }] }, + { + text: words(AI(2), 'checking', 'first'), + toolCalls: [{ name: 'terminal', args: { command: 'echo core-tool-ok' } }] + }, { text: words(A(2), 'tool', 'said', 'ok') } ]) await send(page, `${U(2)} run a tool`, 'Enter', ws) @@ -104,7 +114,9 @@ test('transcript oracle holds across every transition', async () => { }) await test.step('reasoning turn: reasoning deltas never leak into or double the reply', async () => { - provider.script(U(3), [{ reasoning: words(R(3), 'weighing', 'options'), text: words(A(3), 'reasoned', 'answer') }]) + provider.script(U(3), [ + { reasoning: words(R(3), 'weighing', 'options'), text: words(A(3), 'reasoned', 'answer') } + ]) await send(page, `${U(3)} think first`, 'Enter', ws) await finished(U(3)) sessionA.expectUserMarkers.push(U(3)) @@ -216,12 +228,20 @@ test('transcript oracle holds across every transition', async () => { // Sockets that delivered the message.complete of the turn whose reply opens with `marker`. const completeSockets = (marker: string) => - new Set(ws.events.filter(e => e.type === 'message.complete' && String(e.payload?.text ?? '').startsWith(marker)).map(e => e.socket)) + new Set( + ws.events + .filter(e => e.type === 'message.complete' && String(e.payload?.text ?? '').startsWith(marker)) + .map(e => e.socket) + ) let sessionP2: OracleTarget = { sessionId: '', profile: 'p2', expectUserMarkers: [] } await test.step('non-default profile on the shared host backend: one socket per session', async () => { - await page.getByRole('button', { name: 'Bots', exact: true }).or(page.getByRole('tab', { name: 'Bots', exact: true })).first().click() + await page + .getByRole('button', { name: 'Bots', exact: true }) + .or(page.getByRole('tab', { name: 'Bots', exact: true })) + .first() + .click() const row = page.locator('[data-slot="bots-roster"] [data-roster-key="local::p2"]') await expect(row).toBeVisible({ timeout: 60_000 }) await row.click() @@ -232,7 +252,9 @@ test('transcript oracle holds across every transition', async () => { await expect(viewport(page)).toContainText(A(9)) let p2Session: null | string = null await expect - .poll(() => (p2Session = storedSessionForMarker(sandbox, 'p2', U(9))), { message: 'p2 turn persisted in the p2 state.db' }) + .poll(() => (p2Session = storedSessionForMarker(sandbox, 'p2', U(9))), { + message: 'p2 turn persisted in the p2 state.db' + }) .not.toBeNull() sessionP2 = { sessionId: p2Session!, profile: 'p2', expectUserMarkers: [U(9)] } await assertTranscriptOracle(page, ws, provider, sessionP2, 'p2 first turn') @@ -245,7 +267,22 @@ test('transcript oracle holds across every transition', async () => { sessionP2.expectUserMarkers.push(U(13)) await assertTranscriptOracle(page, ws, provider, sessionP2, 'p2 second turn') expect([...completeSockets(A(9))].length, 'p2 turn 1 delivered on exactly one socket').toBe(1) - expect([...completeSockets(A(13))].length, 'p2 turn 2 delivered on exactly one socket (no 2nd socket to one backend)').toBe(1) + expect([...completeSockets(A(13))].length, 'p2 turn 2 delivered on exactly one socket').toBe(1) + // One backend process, one socket: the host backend serves p2 too, so + // the renderer must not hold a second live socket to it (#120006). + const sameBackend = new Set([String(backendPort), String(proxy.port)]) + await expect + .poll( + () => + ws.sockets + .filter(s => !s.closed && sameBackend.has(new URL(s.url).port)) + .map(s => s.url.replace(/token=[^&]+/, 'token=…')), + { + timeout: 30_000, + message: 'live sockets to the one host backend' + } + ) + .toHaveLength(1) }) await test.step('forced second socket to the same backend: every event still renders once', async () => { @@ -256,7 +293,10 @@ test('transcript oracle holds across every transition', async () => { const socketsBefore = ws.sockets.length await splitProfileRoute(app, 'p2') provider.script(U(10), [ - { text: words(AI(10), 'looking', 'it', 'up'), toolCalls: [{ name: 'terminal', args: { command: 'echo core-two-sockets' } }] }, + { + text: words(AI(10), 'looking', 'it', 'up'), + toolCalls: [{ name: 'terminal', args: { command: 'echo core-two-sockets' } }] + }, { text: words(A(10), 'two', 'sockets', 'one', 'render') } ]) await send(page, `${U(10)} tool on p2`, 'Enter', ws) @@ -264,7 +304,10 @@ test('transcript oracle holds across every transition', async () => { await expect.poll(() => ws.sockets.length).toBeGreaterThan(socketsBefore) // Precondition of the scenario: the turn really was fanned out to 2+ sockets. await expect - .poll(() => completeSockets(A(10)).size, { timeout: 60_000, message: 'the forced turn reached more than one socket' }) + .poll(() => completeSockets(A(10)).size, { + timeout: 60_000, + message: 'the forced turn reached more than one socket' + }) .toBeGreaterThan(1) sessionP2.expectUserMarkers.push(U(10)) await assertTranscriptOracle(page, ws, provider, sessionP2, 'p2 tool turn over two sockets') From c62fe9e39fda42055049d481570521b1724927cc Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:03:20 -0700 Subject: [PATCH 025/112] test(desktop-core): switch-back hydrate race, both orders forced (C2) Parks the REST page read in main and the provider stream on gates so message.complete lands before / after the switch-back hydrate deterministically. complete-before-hydrate was 4/4 red on base (reply rendered twice) and is green with the renderer fix. --- .../desktop/e2e/core/switch-back-race.spec.ts | 154 ++++++++++++++++++ 1 file changed, 154 insertions(+) create mode 100644 apps/desktop/e2e/core/switch-back-race.spec.ts diff --git a/apps/desktop/e2e/core/switch-back-race.spec.ts b/apps/desktop/e2e/core/switch-back-race.spec.ts new file mode 100644 index 0000000000..234b93ce60 --- /dev/null +++ b/apps/desktop/e2e/core/switch-back-race.spec.ts @@ -0,0 +1,154 @@ +/** + * C2 core: switch back to a session whose reply completes while the switch-back + * hydrate is in flight. + * + * Session B streams (provider held) -> user opens A -> user returns to B. The + * return issues a REST page read (/api/sessions/B/messages) that races the + * live message.complete. Both orders are forced deterministically: the REST + * read is parked in main (ipcMain 'hermes:api' wrapper) until the test opens + * it, and the provider stream is parked on a gate. No sleeps. + * + * complete-before-hydrate was red on base (reply rendered twice: committed row + * plus a live `assistant-stream-*` row, sometimes frozen on its first chunk); + * fixed in "render a reply once when it completes during the switch-back + * hydrate". Oracle: every message exactly once, DOM == persisted, at every + * sampled frame. + */ + +import { expect, type Page, test } from '@playwright/test' + +import { + coreAppEnv, + createCoreSandbox, + currentSessionId, + launchCoreApp, + recordWebSockets, + send, + waitForInteractive, + writeProviderHome +} from './harness' +import { assertTranscriptOracle, installDuplicateSampler, type OracleTarget } from './oracle' +import { gate, startScriptedProvider } from './provider' + +const nonce = Math.random() + .toString(36) + .slice(2, 8) + .replace(/[^a-z0-9]/g, 'x') + .padEnd(4, 'q') +const U = (n: number) => `U${n}-${nonce}` +const A = (n: number) => `A${n}-${nonce}` + +function viewport(page: Page) { + return page.locator('[data-slot="aui_thread-viewport"]').filter({ visible: true }).first() +} + +for (const order of ['complete-before-hydrate', 'hydrate-before-complete'] as const) { + test(`switch back while away session completes: ${order}`, async () => { + const provider = await startScriptedProvider() + const sandbox = createCoreSandbox('race') + writeProviderHome(sandbox.hermesHome, provider.url) + const { app, page } = await launchCoreApp(coreAppEnv(sandbox)) + const ws = recordWebSockets(page) + + try { + await waitForInteractive(app, page) + await installDuplicateSampler(page) + provider.script(U(1), [{ text: [`${A(1)} `, 'one'] }]) + await send(page, `${U(1)} a`, 'Enter', ws) + await expect.poll(() => provider.completions.some(c => c.marker === U(1) && c.finished)).toBe(true) + await expect.poll(() => currentSessionId(page)).not.toBe('') + const a = await currentSessionId(page) + await expect(viewport(page)).toContainText(A(1)) + + await page.evaluate(() => { + window.location.hash = '#/' + }) + await expect.poll(() => currentSessionId(page)).toBe('') + const hold = gate() + provider.script(U(2), [{ text: [`${A(2)} `, 'finished ', 'while ', 'away'], holdAfterFirstChunk: hold }]) + await send(page, `${U(2)} b`, 'Enter', ws) + await provider.streamStarted(U(2)) + await expect.poll(() => currentSessionId(page)).not.toBe('') + const b = await currentSessionId(page) + await expect(viewport(page)).toContainText(A(2)) + + await page.evaluate(id => { + window.location.hash = `#/${id}` + }, a) + await expect(viewport(page)).toContainText(A(1)) + + { + const wrapped = await app.evaluate(({ ipcMain }, b) => { + const g = globalThis as any + g.__gated = [] + g.__gateOpen = false + g.__gateWaiters = [] as (() => void)[] + // Electron keeps invoke handlers in a private map; if that ever moves, + // fail loudly here rather than silently not gating. + const handlers = (ipcMain as any)._invokeHandlers as Map Promise> | undefined + const original = handlers?.get('hermes:api') + + if (!original) { + return false + } + ipcMain.removeHandler('hermes:api') + ipcMain.handle('hermes:api', async (event: any, request: any) => { + const p = String(request?.path ?? '') + + if (p.includes(`/api/sessions/${b}`) && !g.__gateOpen) { + g.__gated.push(p) + await new Promise(resolve => g.__gateWaiters.push(resolve)) + } + + return original(event, request) + }) + + return true + }, b) + + expect(wrapped, 'hermes:api handler wrapped for gating').toBe(true) + } + + await page.evaluate(id => { + window.location.hash = `#/${id}` + }, b) + await expect.poll(() => currentSessionId(page)).toBe(b) + + const openGate = () => + app.evaluate(() => { + const g = globalThis as any + g.__gateOpen = true + + for (const w of g.__gateWaiters) { + w() + } + + return g.__gated + }) + + const completed = () => + ws.events.some(e => e.type === 'message.complete' && String(e.payload?.text ?? '').startsWith(A(2))) + + const gatedCount = () => app.evaluate(() => ((globalThis as any).__gated ?? []).length as number) + + if (order === 'complete-before-hydrate') { + await expect.poll(gatedCount).toBeGreaterThan(0) + hold.open() + await expect.poll(completed).toBe(true) + await openGate() + } else if (order === 'hydrate-before-complete') { + await expect.poll(gatedCount).toBeGreaterThan(0) + await openGate() + hold.open() + await expect.poll(completed).toBe(true) + } + + const target: OracleTarget = { sessionId: b, expectUserMarkers: [U(2)] } + await assertTranscriptOracle(page, ws, provider, target, `switch-back race ${order}`) + } finally { + await app.close().catch(() => undefined) + await provider.close() + sandbox.cleanup() + } + }) +} From 0015e4038b769b3356bad97636230fa2328200da Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:03:21 -0700 Subject: [PATCH 026/112] test(desktop-core): clarify and approval round trips (C20) Clarify: one card, and the clicked choice is exactly what the model receives in the tool result. Approval: the gated command has not run before the click and has run after Run once; Deny never runs it and the turn still completes. --- apps/desktop/e2e/core/boot-lifecycle.spec.ts | 52 ++++-- apps/desktop/e2e/core/harness.ts | 30 +++- .../e2e/core/interactive-prompts.spec.ts | 166 ++++++++++++++++++ 3 files changed, 226 insertions(+), 22 deletions(-) create mode 100644 apps/desktop/e2e/core/interactive-prompts.spec.ts diff --git a/apps/desktop/e2e/core/boot-lifecycle.spec.ts b/apps/desktop/e2e/core/boot-lifecycle.spec.ts index 27c3628201..f6c641c61d 100644 --- a/apps/desktop/e2e/core/boot-lifecycle.spec.ts +++ b/apps/desktop/e2e/core/boot-lifecycle.spec.ts @@ -37,7 +37,11 @@ import { import { assertTranscriptOracle, installDuplicateSampler, type OracleTarget } from './oracle' import { gate, startScriptedProvider } from './provider' -const nonce = Math.random().toString(36).slice(2, 8).replace(/[^a-z0-9]/g, 'x').padEnd(4, 'q') +const nonce = Math.random() + .toString(36) + .slice(2, 8) + .replace(/[^a-z0-9]/g, 'x') + .padEnd(4, 'q') const U = (n: number) => `U${n}-${nonce}` const A = (n: number) => `A${n}-${nonce}` const TOOL_TAG = `core-orphan-${nonce}` @@ -77,6 +81,7 @@ test('boot handshake, supervised respawn, and zero orphans on quit', async () => // Background census: every backend pid ever observed. const seenBackends = new Set() + const census = setInterval(() => { for (const proc of backendProcesses(sandbox)) { seenBackends.add(proc.pid) @@ -113,10 +118,16 @@ test('boot handshake, supervised respawn, and zero orphans on quit', async () => expect(victim).toBeTruthy() process.kill(victim!.pid, 'SIGKILL') await expect - .poll(() => backendProcesses(sandbox).map(p => p.pid).filter(pid => pid !== victim!.pid).length, { - timeout: 120_000, - message: 'a replacement backend is spawned' - }) + .poll( + () => + backendProcesses(sandbox) + .map(p => p.pid) + .filter(pid => pid !== victim!.pid).length, + { + timeout: 120_000, + message: 'a replacement backend is spawned' + } + ) .toBe(1) await waitForInteractive(app, page) provider.script(U(2), [{ text: [`${A(2)} `, 'after ', 'respawn'] }]) @@ -134,21 +145,32 @@ test('boot handshake, supervised respawn, and zero orphans on quit', async () => await test.step('quit mid-turn with a running tool child: zero processes remain', async () => { const hold = gate() provider.script(U(3), [ - { text: [`${A(3)} `, 'starting ', 'tool'], toolCalls: [{ name: 'terminal', args: { command: `exec -a ${TOOL_TAG} sleep 3600` } }] }, + { + text: [`${A(3)} `, 'starting ', 'tool'], + toolCalls: [{ name: 'terminal', args: { command: `exec -a ${TOOL_TAG} sleep 3600` } }] + }, { text: [`${A(3)}b `, 'never ', 'reached'], holdAfterFirstChunk: hold } ]) await send(page, `${U(3)} long tool`, 'Enter', ws) - await expect.poll(() => taggedProcesses(TOOL_TAG).length, { timeout: 120_000, message: 'tool child running' }).toBeGreaterThan(0) + await expect + .poll(() => taggedProcesses(TOOL_TAG).length, { timeout: 120_000, message: 'tool child running' }) + .toBeGreaterThan(0) expect(sandboxProcesses(sandbox).length).toBeGreaterThan(0) await app.close() closed = true clearInterval(census) await expect - .poll(() => [...sandboxProcesses(sandbox), ...taggedProcesses(TOOL_TAG)].map(p => `${p.pid} ${p.cmdline.slice(0, 120)}`), { - timeout: 60_000, - message: 'no sandbox process (backend, tool child, Electron helper) survives quit' - }) + .poll( + () => + [...sandboxProcesses(sandbox), ...taggedProcesses(TOOL_TAG)].map( + p => `${p.pid} ${p.cmdline.slice(0, 120)}` + ), + { + timeout: 60_000, + message: 'no sandbox process (backend, tool child, Electron helper) survives quit' + } + ) .toEqual([]) hold.open() }) @@ -188,7 +210,9 @@ test('relaunching the same home: one backend per boot, zero after each quit, tra const ws = recordWebSockets(page) await waitForInteractive(app, page) await installDuplicateSampler(page) - await expect.poll(() => backendProcesses(sandbox).length, { message: `one backend on launch ${launch}` }).toBe(1) + await expect + .poll(() => backendProcesses(sandbox).length, { message: `one backend on launch ${launch}` }) + .toBe(1) if (launch === 1) { provider.script(U(1), [{ text: [`${A(1)} `, 'persisted ', 'across ', 'launches'] }]) @@ -202,7 +226,9 @@ test('relaunching the same home: one backend per boot, zero after each quit, tra await page.evaluate(id => { window.location.hash = `#/${encodeURIComponent(id)}` }, session.sessionId) - await expect(page.locator('[data-slot="aui_thread-viewport"]').filter({ visible: true }).first()).toContainText(A(1), { + await expect( + page.locator('[data-slot="aui_thread-viewport"]').filter({ visible: true }).first() + ).toContainText(A(1), { timeout: 60_000 }) } diff --git a/apps/desktop/e2e/core/harness.ts b/apps/desktop/e2e/core/harness.ts index 8042a45e6b..4037bfbb12 100644 --- a/apps/desktop/e2e/core/harness.ts +++ b/apps/desktop/e2e/core/harness.ts @@ -59,7 +59,9 @@ export function createCoreSandbox(label: string): CoreSandbox { // depend on the runner's keyring rather than on Hermes. const bin = path.join(root, 'bin') fs.mkdirSync(bin, { recursive: true }) - fs.writeFileSync(path.join(bin, 'gh'), '#!/bin/sh\necho "no oauth token found for github.com" >&2\nexit 1\n', { mode: 0o755 }) + fs.writeFileSync(path.join(bin, 'gh'), '#!/bin/sh\necho "no oauth token found for github.com" >&2\nexit 1\n', { + mode: 0o755 + }) return { root, @@ -75,7 +77,7 @@ export function createCoreSandbox(label: string): CoreSandbox { } } -export function providerConfigYaml(providerUrl: string, extra = ''): string { +export function providerConfigYaml(providerUrl: string, extra = '', approvals: 'manual' | 'off' = 'off'): string { return `model: default: mock-model provider: mock @@ -92,13 +94,18 @@ auxiliary: title_generation: enabled: false approvals: - mode: "off" + mode: "${approvals}" ${extra}` } -export function writeProviderHome(dir: string, providerUrl: string, extra = ''): void { +export function writeProviderHome( + dir: string, + providerUrl: string, + extra = '', + approvals: 'manual' | 'off' = 'off' +): void { fs.mkdirSync(dir, { recursive: true }) - fs.writeFileSync(path.join(dir, 'config.yaml'), providerConfigYaml(providerUrl, extra)) + fs.writeFileSync(path.join(dir, 'config.yaml'), providerConfigYaml(providerUrl, extra, approvals)) fs.writeFileSync(path.join(dir, '.env'), 'MOCK_API_KEY=core-e2e-key\n') } @@ -291,11 +298,13 @@ export function startTcpProxy(targetPort: number): Promise { const server = net.createServer(client => { const upstream = net.connect(targetPort, '127.0.0.1') live.add(client) + const forget = () => { live.delete(client) client.destroy() upstream.destroy() } + client.on('error', forget) upstream.on('error', forget) client.on('close', forget) @@ -343,6 +352,7 @@ export async function routePrimaryWebSocket(app: ElectronApplication, backendPor const handlers = (ipcMain as any)._invokeHandlers as Map Promise> const from = `ws://127.0.0.1:${backendPort}/` const to = `ws://127.0.0.1:${proxyPort}/` + const rewrite = (value: any): any => { if (typeof value === 'string') { return value.startsWith(from) ? to + value.slice(from.length) : value @@ -449,9 +459,7 @@ export async function waitForInteractive(app: ElectronApplication, page: Page, t await expect .poll( () => - app - .evaluate(({ BrowserWindow }) => BrowserWindow.getAllWindows()[0]?.isVisible() ?? false) - .catch(() => false), + app.evaluate(({ BrowserWindow }) => BrowserWindow.getAllWindows()[0]?.isVisible() ?? false).catch(() => false), { timeout, intervals: [250] } ) .toBe(true) @@ -539,7 +547,11 @@ export function storedSessionForMarker(sandbox: CoreSandbox, profile: string, ma } /** The backend's persisted display transcript (REST, the same read the renderer hydrates from). */ -export async function persistedTranscript(page: Page, sessionId: string, profile?: string): Promise { +export async function persistedTranscript( + page: Page, + sessionId: string, + profile?: string +): Promise { const query = profile ? `&profile=${encodeURIComponent(profile)}` : '' const result = await page.evaluate( diff --git a/apps/desktop/e2e/core/interactive-prompts.spec.ts b/apps/desktop/e2e/core/interactive-prompts.spec.ts new file mode 100644 index 0000000000..db5b08cf43 --- /dev/null +++ b/apps/desktop/e2e/core/interactive-prompts.spec.ts @@ -0,0 +1,166 @@ +/** + * C20 core: blocking interactive prompts round-trip through the real chain. + * + * Real Electron + real `hermes serve`, approvals in manual mode; only the LLM + * is faked. For each prompt kind the invariant is end to end, not "a card + * rendered": + * - clarify: exactly one card; the choice the user clicks is exactly what + * the model receives in the tool result on its next request; the turn + * completes; transcript oracle holds. + * - approval (run once): exactly one approval card; the gated command has + * NOT run before the click and HAS run after it (observable side effect); + * the turn completes; transcript oracle holds. + * - approval (deny): the command never runs; the turn still completes (no + * wedged busy state); transcript oracle holds. + */ + +import * as fs from 'node:fs' +import * as path from 'node:path' + +import { expect, type Page, test } from '@playwright/test' + +import { + coreAppEnv, + createCoreSandbox, + currentSessionId, + launchCoreApp, + recordWebSockets, + send, + waitForInteractive, + writeProviderHome +} from './harness' +import { assertTranscriptOracle, installDuplicateSampler, type OracleTarget } from './oracle' +import { type RecordedCompletion, startScriptedProvider } from './provider' + +const nonce = Math.random() + .toString(36) + .slice(2, 8) + .replace(/[^a-z0-9]/g, 'x') + .padEnd(4, 'q') +const U = (n: number) => `U${n}-${nonce}` +const A = (n: number) => `A${n}-${nonce}` +const AI = (n: number) => `A${n}i-${nonce}` + +function viewport(page: Page) { + return page.locator('[data-slot="aui_thread-viewport"]').filter({ visible: true }).first() +} + +/** Text of the tool results the model received in `completion`'s request. */ +function toolResults(completion: RecordedCompletion | undefined): string { + const messages: any[] = completion?.body?.messages ?? [] + + return messages + .filter(m => m?.role === 'tool') + .map(m => (typeof m.content === 'string' ? m.content : JSON.stringify(m.content))) + .join('\n') +} + +test('clarify and approval prompts round-trip exactly once', async () => { + const provider = await startScriptedProvider() + const sandbox = createCoreSandbox('prompts') + writeProviderHome(sandbox.hermesHome, provider.url, '', 'manual') + const { app, page } = await launchCoreApp(coreAppEnv(sandbox)) + const ws = recordWebSockets(page) + const session: OracleTarget = { sessionId: '', expectUserMarkers: [] } + + const finished = (marker: string, step = 0) => + expect + .poll(() => provider.completions.some(c => c.marker === marker && c.step === step && c.finished), { + timeout: 120_000, + message: `provider finished ${marker} step ${step}` + }) + .toBe(true) + + const completion = (marker: string, step: number) => + provider.completions.find(c => c.marker === marker && c.step === step) + + try { + await waitForInteractive(app, page) + await installDuplicateSampler(page) + + await test.step('clarify: the clicked choice is exactly what the model receives', async () => { + const question = `Q1-${nonce} which drink` + const pick = `mate-${nonce}` + const other = `chai-${nonce}` + provider.script(U(1), [ + { + text: [`${AI(1)} `, 'asking'], + toolCalls: [{ name: 'clarify', args: { questions: [{ question, choices: [other, pick] }] } }] + }, + { text: [`${A(1)} `, 'noted ', 'your ', 'choice'] } + ]) + await send(page, `${U(1)} ask me`, 'Enter', ws) + // The open (unanswered) clarify form: single-question or batch shape. + const openForms = page.locator('form[data-clarify-choices], form[data-clarify-batch]').filter({ visible: true }) + await expect(openForms).toHaveCount(1, { timeout: 120_000 }) + await expect(viewport(page).getByText(question)).toHaveCount(1) + await openForms.getByRole('button', { name: new RegExp(pick) }).click() + const submit = openForms.locator('button[type="submit"]') + await expect(submit).toBeEnabled() + await submit.click() + await finished(U(1), 1) + const received = toolResults(completion(U(1), 1)) + expect(received, 'model received the clicked choice').toContain(pick) + expect(received, 'model did not receive the other choice as the answer').not.toMatch( + new RegExp(`answer[^\\n]*${other}`, 'i') + ) + await expect( + page.locator('form[data-clarify-choices], form[data-clarify-batch]').filter({ visible: true }) + ).toHaveCount(0) + await expect(page.locator('[data-clarify-settled]').filter({ visible: true })).toHaveCount(1) + await expect.poll(() => currentSessionId(page)).not.toBe('') + session.sessionId = await currentSessionId(page) + session.expectUserMarkers.push(U(1)) + await assertTranscriptOracle(page, ws, provider, session, 'clarify round trip') + await expect(viewport(page).getByText(question)).toHaveCount(1) + }) + + await test.step('approval (run once): the command runs only after the click', async () => { + const victim = path.join(sandbox.root, `victim-run-${nonce}`) + fs.mkdirSync(victim) + provider.script(U(2), [ + { + text: [`${AI(2)} `, 'needs ', 'approval'], + toolCalls: [{ name: 'terminal', args: { command: `rm -rf ${victim}` } }] + }, + { text: [`${A(2)} `, 'deleted ', 'it'] } + ]) + await send(page, `${U(2)} delete run dir`, 'Enter', ws) + const run = page.locator('[data-approval-run]').filter({ visible: true }) + await expect(run).toHaveCount(1, { timeout: 120_000 }) + expect(fs.existsSync(victim), 'gated command must not run before approval').toBe(true) + await run.click() + await finished(U(2), 1) + expect(fs.existsSync(victim), 'approved command ran').toBe(false) + await expect(page.locator('[data-approval-run]').filter({ visible: true })).toHaveCount(0) + session.expectUserMarkers.push(U(2)) + await assertTranscriptOracle(page, ws, provider, session, 'approval run once') + }) + + await test.step('approval (deny): the command never runs, the turn still completes', async () => { + const victim = path.join(sandbox.root, `victim-deny-${nonce}`) + fs.mkdirSync(victim) + provider.script(U(3), [ + { + text: [`${AI(3)} `, 'needs ', 'approval'], + toolCalls: [{ name: 'terminal', args: { command: `rm -rf ${victim}` } }] + }, + { text: [`${A(3)} `, 'left ', 'it ', 'alone'] } + ]) + await send(page, `${U(3)} delete deny dir`, 'Enter', ws) + const deny = page.locator('[data-approval-deny]').filter({ visible: true }) + await expect(deny).toHaveCount(1, { timeout: 120_000 }) + await deny.click() + await finished(U(3), 1) + expect(fs.existsSync(victim), 'denied command never ran').toBe(true) + expect(toolResults(completion(U(3), 1)), 'the denial reached the model as the tool result').not.toBe('') + await expect(page.locator('[data-approval-deny]').filter({ visible: true })).toHaveCount(0) + session.expectUserMarkers.push(U(3)) + await assertTranscriptOracle(page, ws, provider, session, 'approval deny') + }) + } finally { + await app.close().catch(() => undefined) + await provider.close() + sandbox.cleanup() + } +}) From 12b770c35752a6e6b2d0cc782ba8c0d0a10f5aca Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:04:17 -0700 Subject: [PATCH 027/112] test(desktop-core): README lists the C20 and switch-back specs --- apps/desktop/e2e/core/README.md | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/apps/desktop/e2e/core/README.md b/apps/desktop/e2e/core/README.md index 7dd68d731a..d3062a8677 100644 --- a/apps/desktop/e2e/core/README.md +++ b/apps/desktop/e2e/core/README.md @@ -1,6 +1,6 @@ # Desktop core suite (required CI lane) -A small, deterministic Electron suite that guards two issue classes end to end: +A small, deterministic Electron suite that guards three issue classes end to end: - **C2 transcript integrity** — `transcript-integrity.spec.ts`: one real app + one real `hermes serve`, only the LLM faked (`provider.ts`, scripted per turn @@ -17,6 +17,14 @@ A small, deterministic Electron suite that guards two issue classes end to end: - backend stream integrity: each turn's concatenated `message.delta` / `reasoning.delta` equals what the provider streamed, and `message.complete` equals the final completion. + - one live socket per backend process (#120006). + `switch-back-race.spec.ts` forces both orders of "reply completes" vs "the + switch-back REST hydrate resolves" with gates (no sleeps) under the same + oracle. +- **C20 interactive prompts** — `interactive-prompts.spec.ts`: clarify (one + card; the clicked choice is exactly what the model receives), approval + Run once (the command runs only after the click) and Deny (never runs; the + turn still completes), approvals in manual mode. - **C5 boot / process lifecycle** — `boot-lifecycle.spec.ts`: interactive composer + first turn, exactly one backend; `kill -9` backend → exactly one supervised respawn and a working turn; quit mid-turn with a running tool From 3d8dbad45eed0095639a66e67e4803c12b55b50a Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:21:31 -0700 Subject: [PATCH 028/112] test(desktop-core): wait for message.complete before judging the DOM; explain stuck boots Under heavy load message.complete can trail the persisted row, so the wire check is now a deadline poll run first (duplicate frames never heal, so the poll cannot mask them) and the DOM oracle then covers whatever the renderer does on completion. A boot that never becomes interactive now fails with the blocking element chain, route, body text and the main-process log tail. --- apps/desktop/e2e/core/README.md | 6 +- apps/desktop/e2e/core/harness.ts | 95 +++++++++++++++++++++++--------- apps/desktop/e2e/core/oracle.ts | 15 ++++- 3 files changed, 85 insertions(+), 31 deletions(-) diff --git a/apps/desktop/e2e/core/README.md b/apps/desktop/e2e/core/README.md index d3062a8677..8b811fbeb5 100644 --- a/apps/desktop/e2e/core/README.md +++ b/apps/desktop/e2e/core/README.md @@ -18,9 +18,9 @@ A small, deterministic Electron suite that guards three issue classes end to end `reasoning.delta` equals what the provider streamed, and `message.complete` equals the final completion. - one live socket per backend process (#120006). - `switch-back-race.spec.ts` forces both orders of "reply completes" vs "the - switch-back REST hydrate resolves" with gates (no sleeps) under the same - oracle. + `switch-back-race.spec.ts` forces both orders of "reply completes" vs "the + switch-back REST hydrate resolves" with gates (no sleeps) under the same + oracle. - **C20 interactive prompts** — `interactive-prompts.spec.ts`: clarify (one card; the clicked choice is exactly what the model receives), approval Run once (the command runs only after the click) and Deny (never runs; the diff --git a/apps/desktop/e2e/core/harness.ts b/apps/desktop/e2e/core/harness.ts index 4037bfbb12..a1e24a34a3 100644 --- a/apps/desktop/e2e/core/harness.ts +++ b/apps/desktop/e2e/core/harness.ts @@ -154,11 +154,29 @@ export async function launchCoreApp(env: Record): Promise<{ app: cwd: DESKTOP_ROOT }) + // Keep the main process's stdout/stderr (backend supervisor lines included) + // so a boot that never becomes interactive fails with its own story. + const lines: string[] = [] + const collect = (chunk: Buffer) => { + lines.push(...chunk.toString('utf8').split('\n').filter(Boolean)) + lines.splice(0, Math.max(0, lines.length - 200)) + } + app.process().stdout?.on('data', collect) + app.process().stderr?.on('data', collect) + APP_LOGS.set(app, lines) + const page = await app.firstWindow() return { app, page } } +const APP_LOGS = new WeakMap() + +/** Last main-process output lines of `app` (for failure messages). */ +export function appLogTail(app: ElectronApplication, n = 60): string { + return (APP_LOGS.get(app) ?? []).slice(-n).join('\n') +} + // ─── Process census ───────────────────────────────────────────────────── export interface ProcInfo { @@ -427,35 +445,62 @@ export function composer(page: Page) { /** Composer mounted, no full-viewport overlay above it, window visible. */ export async function waitForInteractive(app: ElectronApplication, page: Page, timeout = 180_000): Promise { - await expect(composer(page)).toBeVisible({ timeout }) - await page.waitForFunction( - () => { - const el = document.elementFromPoint(window.innerWidth / 2, window.innerHeight / 2) - let node: Element | null = el + try { + await expect(composer(page)).toBeVisible({ timeout }) + await page.waitForFunction( + () => { + const el = document.elementFromPoint(window.innerWidth / 2, window.innerHeight / 2) + let node: Element | null = el - if (!el) { - return false - } - - while (node) { - const cs = window.getComputedStyle(node) - - if (cs.position === 'fixed') { - const r = node.getBoundingClientRect() - - if (r.left <= 0 && r.top <= 0 && r.right >= window.innerWidth && r.bottom >= window.innerHeight) { - return false - } + if (!el) { + return false } - node = node.parentElement - } + while (node) { + const cs = window.getComputedStyle(node) + + if (cs.position === 'fixed') { + const r = node.getBoundingClientRect() + + if (r.left <= 0 && r.top <= 0 && r.right >= window.innerWidth && r.bottom >= window.innerHeight) { + return false + } + } + + node = node.parentElement + } + + return true + }, + undefined, + { timeout, polling: 250 } + ) + } catch (error) { + const blocker = await page + .evaluate(() => { + let node: Element | null = document.elementFromPoint(window.innerWidth / 2, window.innerHeight / 2) + const chain: string[] = [] + + while (node && chain.length < 12) { + const slot = node.getAttribute('data-slot') ?? '' + const role = node.getAttribute('role') ?? '' + chain.push( + `${node.tagName.toLowerCase()}${slot ? `[data-slot=${slot}]` : ''}${role ? `[role=${role}]` : ''}${window.getComputedStyle(node).position === 'fixed' ? '{fixed}' : ''}` + ) + node = node.parentElement + } + + return { chain, route: location.hash, text: document.body.innerText.replace(/\s+/g, ' ').slice(0, 800) } + }) + .catch(e => ({ chain: [], route: '', text: `evaluate failed: ${String(e)}` })) + + throw new Error( + `app never became interactive: ${(error as Error).message.split('\n')[0]}\n` + + `route: ${blocker.route}\ncenter element chain: ${blocker.chain.join(' < ')}\nbody text: ${blocker.text}\n` + + `main-process log tail:\n${appLogTail(app)}` + ) + } - return true - }, - undefined, - { timeout, polling: 250 } - ) await expect .poll( () => diff --git a/apps/desktop/e2e/core/oracle.ts b/apps/desktop/e2e/core/oracle.ts index 09c5acc859..357d7c32a9 100644 --- a/apps/desktop/e2e/core/oracle.ts +++ b/apps/desktop/e2e/core/oracle.ts @@ -439,6 +439,18 @@ export async function assertTranscriptOracle( view: { text: '', userBubbles: [] } } + // Wire first: wait until every expected turn's message.complete is on the + // wire (it can trail the persisted row under load), so the DOM check below + // also covers whatever the renderer does on completion. Duplicate/garbled + // frames never heal, so polling cannot mask them. + await expect + .poll(() => wireViolations(ws, provider, new Set(target.expectUserMarkers), new Set(target.lossyWire ?? [])), { + timeout: 60_000, + intervals: [250, 500, 1000], + message: `wire integrity [${label}]` + }) + .toEqual([]) + await expect .poll( async () => { @@ -463,7 +475,4 @@ export async function assertTranscriptOracle( expect(transient.violations, `transient duplicate render during [${label}] (${transient.samples} samples)`).toEqual( [] ) - - const wire = wireViolations(ws, provider, new Set(target.expectUserMarkers), new Set(target.lossyWire ?? [])) - expect(wire, `wire integrity [${label}]`).toEqual([]) } From 20300f0d475d7ecb63c4828449a31ecc50ed1efc Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 05:06:36 -0700 Subject: [PATCH 029/112] test(desktop-core): a pre-exec child of the backend is not a second backend The boot census flagged "backend pids seen during boot: 2" in about 1 of 20 boots. Diagnosis (tight /proc watcher beside 20 repeats): every `hermes serve` forks ~40 probes per boot (git, ps, ldconfig/gcc for find_library, pip, docker version), and between fork and exec each child still shows the backend's argv and HERMES_HOME; ~9 per boot were caught in that window at 2 ms sampling, and the 100 ms census caught one every few boots (the reproduced red: two extra pids with ppid = the backend, one sample each). No double spawn by the supervisor was ever observed. backendProcesses() now drops a serve process whose parent is itself a serve process. A real second spawn has Electron main as parent, and a pre-exec child orphaned by a dead backend is reparented and still counted. The census also records ppid, argv and lifetime per pid so any future extra pid explains itself in the failure message. --- apps/desktop/e2e/core/boot-lifecycle.spec.ts | 25 ++++++++++++++++---- apps/desktop/e2e/core/harness.ts | 20 ++++++++++++++-- 2 files changed, 38 insertions(+), 7 deletions(-) diff --git a/apps/desktop/e2e/core/boot-lifecycle.spec.ts b/apps/desktop/e2e/core/boot-lifecycle.spec.ts index f6c641c61d..5b1c23dcb7 100644 --- a/apps/desktop/e2e/core/boot-lifecycle.spec.ts +++ b/apps/desktop/e2e/core/boot-lifecycle.spec.ts @@ -42,6 +42,7 @@ const nonce = Math.random() .slice(2, 8) .replace(/[^a-z0-9]/g, 'x') .padEnd(4, 'q') + const U = (n: number) => `U${n}-${nonce}` const A = (n: number) => `A${n}-${nonce}` const TOOL_TAG = `core-orphan-${nonce}` @@ -79,15 +80,29 @@ test('boot handshake, supervised respawn, and zero orphans on quit', async () => const ws = recordWebSockets(page) let closed = false - // Background census: every backend pid ever observed. - const seenBackends = new Set() + // Background census: every backend pid ever observed, with its parent, + // command line and lifetime so an unexpected extra pid explains itself. + const seen = new Map() + const t0 = Date.now() const census = setInterval(() => { + const now = Date.now() - t0 + for (const proc of backendProcesses(sandbox)) { - seenBackends.add(proc.pid) + const entry = seen.get(proc.pid) + + if (entry) { + entry.last = now + entry.samples++ + } else { + seen.set(proc.pid, { ppid: proc.ppid, cmdline: proc.cmdline.slice(0, 200), first: now, last: now, samples: 1 }) + } } }, 100) + const describeSeen = () => + JSON.stringify([...seen].map(([pid, e]) => ({ pid, ...e, electronPid: app.process().pid }))) + const finished = (marker: string, step = 0) => expect .poll(() => provider.completions.some(c => c.marker === marker && c.step === step && c.finished), { @@ -110,7 +125,7 @@ test('boot handshake, supervised respawn, and zero orphans on quit', async () => session.sessionId = await currentSessionId(page) session.expectUserMarkers.push(U(1)) await assertTranscriptOracle(page, ws, provider, session, 'boot first turn') - expect(seenBackends.size, `backend pids seen during boot: ${[...seenBackends]}`).toBe(1) + expect(seen.size, `backend pids seen during boot: ${describeSeen()}`).toBe(1) }) await test.step('kill -9 backend: exactly one supervised respawn, app serves again', async () => { @@ -139,7 +154,7 @@ test('boot handshake, supervised respawn, and zero orphans on quit', async () => expect(alive.length, `live backends after recovery: ${JSON.stringify(alive)}`).toBe(1) // Initial + exactly one replacement, ever — a crash loop or a racing // second spawn would add pids here even if they died again. - expect(seenBackends.size, `backend pids ever seen: ${[...seenBackends]}`).toBe(2) + expect(seen.size, `backend pids ever seen: ${describeSeen()}`).toBe(2) }) await test.step('quit mid-turn with a running tool child: zero processes remain', async () => { diff --git a/apps/desktop/e2e/core/harness.ts b/apps/desktop/e2e/core/harness.ts index a1e24a34a3..9dfe9c3f1e 100644 --- a/apps/desktop/e2e/core/harness.ts +++ b/apps/desktop/e2e/core/harness.ts @@ -157,10 +157,12 @@ export async function launchCoreApp(env: Record): Promise<{ app: // Keep the main process's stdout/stderr (backend supervisor lines included) // so a boot that never becomes interactive fails with its own story. const lines: string[] = [] + const collect = (chunk: Buffer) => { lines.push(...chunk.toString('utf8').split('\n').filter(Boolean)) lines.splice(0, Math.max(0, lines.length - 200)) } + app.process().stdout?.on('data', collect) app.process().stderr?.on('data', collect) APP_LOGS.set(app, lines) @@ -228,11 +230,25 @@ export function sandboxProcesses(sandbox: CoreSandbox): ProcInfo[] { return out } -/** The `hermes serve` backend(s) spawned for this sandbox. */ +/** + * The `hermes serve` backend(s) spawned for this sandbox. + * + * A child of the backend still shows the backend's argv and environ between + * fork and exec, and the backend forks ~40 probes per boot (git, ps, + * ldconfig/gcc, pip): a 100 ms sampler catches one in that window every few + * boots. Such a child is not a second backend, so a serve process whose parent + * is itself a serve process is excluded. A real second spawn has the + * supervisor (Electron main) as its parent, and a pre-exec child orphaned by a + * dead backend is reparented away and still counted. + */ export function backendProcesses(sandbox: CoreSandbox): ProcInfo[] { - return sandboxProcesses(sandbox).filter( + const serve = sandboxProcesses(sandbox).filter( proc => / serve( |$)/.test(proc.cmdline) && !/electron/i.test(proc.cmdline.split(' ')[0]) ) + + const pids = new Set(serve.map(proc => proc.pid)) + + return serve.filter(proc => !pids.has(proc.ppid)) } // ─── WebSocket recorder ───────────────────────────────────────────────── From bd736f3b4265cb84840a9b28f8c0c1c9a4f7c794 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 05:06:37 -0700 Subject: [PATCH 030/112] test(desktop-core): onboarding (custom endpoint) then first chat (C2) Fresh home, no provider: the real first-run onboarding, custom endpoint pointed at the scripted provider, then the first chat, a second turn and a reload, each under the transcript oracle (exactly once, DOM == persisted, streamed == provider at every sampled frame). Also asserts the persisted assignment is the entered endpoint, the model is called with the endpoint's advertised model, and one live socket remains after onboarding: the path where #120005 (every word painted twice) shipped. Red when onboarding persists the URL as typed instead of the probed /v1 base (#65488 class). --- apps/desktop/e2e/core/README.md | 4 + .../e2e/core/onboarding-first-chat.spec.ts | 142 ++++++++++++++++++ 2 files changed, 146 insertions(+) create mode 100644 apps/desktop/e2e/core/onboarding-first-chat.spec.ts diff --git a/apps/desktop/e2e/core/README.md b/apps/desktop/e2e/core/README.md index 8b811fbeb5..e078faf7cd 100644 --- a/apps/desktop/e2e/core/README.md +++ b/apps/desktop/e2e/core/README.md @@ -21,6 +21,10 @@ A small, deterministic Electron suite that guards three issue classes end to end `switch-back-race.spec.ts` forces both orders of "reply completes" vs "the switch-back REST hydrate resolves" with gates (no sleeps) under the same oracle. + `onboarding-first-chat.spec.ts` starts from a fresh home with no provider: + the real onboarding (custom endpoint → the fake provider's URL), then the + first chat, a second turn and a reload under the same oracle, plus + persisted config == entered endpoint and one live socket afterwards. - **C20 interactive prompts** — `interactive-prompts.spec.ts`: clarify (one card; the clicked choice is exactly what the model receives), approval Run once (the command runs only after the click) and Deny (never runs; the diff --git a/apps/desktop/e2e/core/onboarding-first-chat.spec.ts b/apps/desktop/e2e/core/onboarding-first-chat.spec.ts new file mode 100644 index 0000000000..6715bf4c0c --- /dev/null +++ b/apps/desktop/e2e/core/onboarding-first-chat.spec.ts @@ -0,0 +1,142 @@ +/** + * C2 core: onboarding completion, then the first chat. + * + * Fresh sandbox, NO provider configured: the real first-run onboarding shows. + * The user picks the local / custom endpoint path and enters the fake + * provider's URL; the real backend probes it (/v1/models), persists the + * assignment, and the overlay closes. Then the first chat must obey the same + * transcript oracle as everything else — this is the path where #120005 + * (every streamed word painted twice) first shipped. + * + * Invariants: the onboarding result is what the backend persisted and what + * the model is actually called with (config base_url == entered URL; the + * first completion carries the endpoint's advertised model); one live socket + * to the backend after onboarding; every message exactly once, DOM == + * persisted, streamed == provider, at every sampled frame. + */ + +import * as fs from 'node:fs' +import * as path from 'node:path' + +import { expect, test } from '@playwright/test' + +import { + coreAppEnv, + createCoreSandbox, + currentSessionId, + launchCoreApp, + recordWebSockets, + send, + waitForInteractive +} from './harness' +import { assertTranscriptOracle, installDuplicateSampler } from './oracle' +import { startScriptedProvider } from './provider' + +const nonce = Math.random() + .toString(36) + .slice(2, 8) + .replace(/[^a-z0-9]/g, 'x') + .padEnd(4, 'q') + +const U = (n: number) => `U${n}-${nonce}` +const A = (n: number) => `A${n}-${nonce}` + +test('onboarding (custom endpoint) then first chat renders exactly once', async () => { + const provider = await startScriptedProvider() + const sandbox = createCoreSandbox('onboard') + const { app, page } = await launchCoreApp(coreAppEnv(sandbox)) + const ws = recordWebSockets(page) + + try { + await installDuplicateSampler(page) + + await test.step('first-run onboarding: custom endpoint', async () => { + // The onboarding surface masks the whole app until it finishes. + const openKeyForm = page + .locator('[data-glass-opaque]') + .filter({ visible: true }) + .getByRole('button', { name: /api key/i }) + + await expect(openKeyForm).toHaveCount(1, { timeout: 180_000 }) + await openKeyForm.click() + await page + .locator('[data-glass-opaque]') + .filter({ visible: true }) + .getByRole('button', { name: /custom endpoint/i }) + .click() + + // The endpoint URL is the form's only plain-text input (API keys are password inputs). + const form = page + .locator('[data-glass-opaque]') + .filter({ visible: true }) + .filter({ has: page.locator('input[type="text"]') }) + + await expect(form).toHaveCount(1) + await form.locator('input[type="text"]').fill(provider.url) + await form.getByRole('button', { name: /connect/i }).click() + // The overlay unmounts only after the backend probed the endpoint, saved + // the assignment and confirmed the runtime is ready; waitForInteractive + // fails while any full-viewport fixed layer still covers the composer. + await waitForInteractive(app, page) + }) + + await test.step('the persisted assignment is what the user entered', async () => { + const config = fs.readFileSync(path.join(sandbox.hermesHome, 'config.yaml'), 'utf8') + expect(config).toMatch(/provider:\s*['"]?custom/) + expect(config).toContain(`127.0.0.1:${new URL(provider.url).port}`) + }) + + await test.step('first chat after onboarding renders exactly once', async () => { + provider.script(U(1), [ + { reasoning: ['R1-', nonce, ' thinking'], text: [`${A(1)} `, 'first ', 'reply ', 'after ', 'setup'] } + ]) + await send(page, `${U(1)} hello`, 'Enter', ws) + await expect + .poll(() => provider.completions.some(c => c.marker === U(1) && c.finished), { + timeout: 120_000, + message: 'the first chat after onboarding reached the configured endpoint' + }) + .toBe(true) + const first = provider.completions.find(c => c.marker === U(1)) + // The fake endpoint advertises exactly one model at /v1/models. + expect(first?.body?.model, 'the model is called with the endpoint-advertised model').toBe('mock-model') + await expect.poll(() => currentSessionId(page)).not.toBe('') + const sessionId = await currentSessionId(page) + await assertTranscriptOracle( + page, + ws, + provider, + { sessionId, expectUserMarkers: [U(1)] }, + 'first chat after onboarding' + ) + await expect + .poll(() => ws.sockets.filter(s => !s.closed).length, { timeout: 30_000, message: 'live backend sockets' }) + .toBe(1) + }) + + await test.step('second turn and a reload keep it exactly once', async () => { + provider.script(U(2), [{ text: [`${A(2)} `, 'second ', 'reply'] }]) + await send(page, `${U(2)} again`, 'Enter', ws) + await expect + .poll(() => provider.completions.some(c => c.marker === U(2) && c.finished), { timeout: 120_000 }) + .toBe(true) + const sessionId = await currentSessionId(page) + await assertTranscriptOracle(page, ws, provider, { sessionId, expectUserMarkers: [U(1), U(2)] }, 'second turn') + await page.reload() + await waitForInteractive(app, page) + await installDuplicateSampler(page) + await expect.poll(() => currentSessionId(page)).toBe(sessionId) + await assertTranscriptOracle( + page, + ws, + provider, + { sessionId, expectUserMarkers: [U(1), U(2)] }, + 'reload after onboarding' + ) + }) + } finally { + await app.close().catch(() => undefined) + await provider.close() + sandbox.cleanup() + } +}) From c7cdc91fe8cbed97e17f0cd06581f59289f10f04 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 05:06:37 -0700 Subject: [PATCH 031/112] style(desktop-core): eslint padding in the core specs --- apps/desktop/e2e/core/interactive-prompts.spec.ts | 1 + apps/desktop/e2e/core/switch-back-race.spec.ts | 2 ++ apps/desktop/e2e/core/transcript-integrity.spec.ts | 1 + 3 files changed, 4 insertions(+) diff --git a/apps/desktop/e2e/core/interactive-prompts.spec.ts b/apps/desktop/e2e/core/interactive-prompts.spec.ts index db5b08cf43..18fb9b6e40 100644 --- a/apps/desktop/e2e/core/interactive-prompts.spec.ts +++ b/apps/desktop/e2e/core/interactive-prompts.spec.ts @@ -37,6 +37,7 @@ const nonce = Math.random() .slice(2, 8) .replace(/[^a-z0-9]/g, 'x') .padEnd(4, 'q') + const U = (n: number) => `U${n}-${nonce}` const A = (n: number) => `A${n}-${nonce}` const AI = (n: number) => `A${n}i-${nonce}` diff --git a/apps/desktop/e2e/core/switch-back-race.spec.ts b/apps/desktop/e2e/core/switch-back-race.spec.ts index 234b93ce60..6130edb737 100644 --- a/apps/desktop/e2e/core/switch-back-race.spec.ts +++ b/apps/desktop/e2e/core/switch-back-race.spec.ts @@ -35,6 +35,7 @@ const nonce = Math.random() .slice(2, 8) .replace(/[^a-z0-9]/g, 'x') .padEnd(4, 'q') + const U = (n: number) => `U${n}-${nonce}` const A = (n: number) => `A${n}-${nonce}` @@ -91,6 +92,7 @@ for (const order of ['complete-before-hydrate', 'hydrate-before-complete'] as co if (!original) { return false } + ipcMain.removeHandler('hermes:api') ipcMain.handle('hermes:api', async (event: any, request: any) => { const p = String(request?.path ?? '') diff --git a/apps/desktop/e2e/core/transcript-integrity.spec.ts b/apps/desktop/e2e/core/transcript-integrity.spec.ts index 882508232e..02325c2497 100644 --- a/apps/desktop/e2e/core/transcript-integrity.spec.ts +++ b/apps/desktop/e2e/core/transcript-integrity.spec.ts @@ -43,6 +43,7 @@ const nonce = Math.random() .slice(2, 8) .replace(/[^a-z0-9]/g, 'x') .padEnd(4, 'q') + const U = (n: number) => `U${n}-${nonce}` const A = (n: number) => `A${n}-${nonce}` const AI = (n: number) => `A${n}i-${nonce}` From 3ed7faa88d27b3523e8381f9b91360b93a3e1143 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 05:39:52 -0700 Subject: [PATCH 032/112] test(desktop-core): keep the external tirith scanner out of the sandbox With no tirith on PATH (a CI runner) the backend installs it from GitHub releases synchronously on the first terminal command, which puts a network download inside a required lane. With one on PATH (a dev box) it fetched a 12 MB threat DB into the sandbox HOME that was still being written after quit and cleanup (the stability loop left a recreated sandbox dir behind). The approval prompts the suite asserts come from Hermes's own dangerous- command detector, so the sandbox config turns tirith off, the same way the suite fakes gh. --- apps/desktop/e2e/core/harness.ts | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/apps/desktop/e2e/core/harness.ts b/apps/desktop/e2e/core/harness.ts index 9dfe9c3f1e..eab9f29ff6 100644 --- a/apps/desktop/e2e/core/harness.ts +++ b/apps/desktop/e2e/core/harness.ts @@ -77,6 +77,13 @@ export function createCoreSandbox(label: string): CoreSandbox { } } +/** + * Sandbox config: only the scripted provider. The external tirith scanner is + * off: with none on PATH the backend downloads it from GitHub on the first + * terminal command (network in a required lane), and with one on PATH it + * fetched a 12 MB threat DB that was still being written after quit. The + * approval prompts under test come from Hermes's own detector. + */ export function providerConfigYaml(providerUrl: string, extra = '', approvals: 'manual' | 'off' = 'off'): string { return `model: default: mock-model @@ -93,6 +100,8 @@ providers: auxiliary: title_generation: enabled: false +security: + tirith_enabled: false approvals: mode: "${approvals}" ${extra}` From f14ee90e528ec4709f15451429f9c762be307ebb Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 06:01:36 -0700 Subject: [PATCH 033/112] test(desktop-core): no lazy promisor fetch from a partial-clone checkout Loop run 7 at 22e8184 went red on "no sandbox process survives quit #1": a `git fetch origin --filter=blob:none --stdin` (+ its index-pack) was still running 60 s after quit. That is git's lazy promisor fetch: the dev checkout is a blob:none partial clone, and a git read by the backend needed a missing blob, so git went to the network, and the fetch outlived the backend (reported as a finding; same class as the gh probe). CI checkouts are not partial clones, so the spawned app gets GIT_NO_LAZY_FETCH=1: dev runs stay offline and deterministic like CI. --- apps/desktop/e2e/core/harness.ts | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/apps/desktop/e2e/core/harness.ts b/apps/desktop/e2e/core/harness.ts index eab9f29ff6..1288681d8d 100644 --- a/apps/desktop/e2e/core/harness.ts +++ b/apps/desktop/e2e/core/harness.ts @@ -147,6 +147,11 @@ export function coreAppEnv(sandbox: CoreSandbox, extra: Record = HERMES_DESKTOP_APP_NAME: `HermesCoreE2E-${path.basename(sandbox.root)}`, HERMES_DESKTOP_SKIP_QUIT_CONFIRM: '1', HERMES_DESKTOP_CDP_PORT: 'off', + // A partial-clone (blob:none) dev checkout turns some backend git read into + // a lazy `git fetch origin` over the network, which outlived quit by >60 s + // (reported as a finding). CI checkouts are not partial; keep dev runs + // offline and deterministic the same way. + GIT_NO_LAZY_FETCH: '1', ...extra } } From 68d5ac7c115f85c0d3f5487833693817eb63aa2d Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 06:50:23 -0700 Subject: [PATCH 034/112] ci(desktop-core): run the core Desktop suite as its own required lane The legacy visual Playwright lane stays disabled; the core suite gets its own reusable workflow (retries 0, one worker, failure-only artifacts) gated on the same python_prod/frontend classification and counted by All required checks pass. The legacy config ignores e2e/core so the specs never run twice. --- .github/workflows/ci.yaml | 9 +++ .github/workflows/e2e-desktop-core.yml | 99 ++++++++++++++++++++++++++ apps/desktop/playwright.config.ts | 5 +- 3 files changed, 111 insertions(+), 2 deletions(-) create mode 100644 .github/workflows/e2e-desktop-core.yml diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index 973f4b0917..7b38efea20 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -136,6 +136,14 @@ jobs: if: false uses: ./.github/workflows/e2e-desktop.yml + e2e-desktop-core: + name: Desktop core E2E + needs: detect + # The deterministic core suite (apps/desktop/e2e/core): transcript + # integrity, boot/respawn/orphans, clarify+approval. Required; retries 0. + if: ${{ needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true' }} + uses: ./.github/workflows/e2e-desktop-core.yml + docs-site: name: Docs Site needs: detect @@ -234,6 +242,7 @@ jobs: - installer-tests - rust-tests - e2e-desktop + - e2e-desktop-core - docs-site - history-check - contributor-check diff --git a/.github/workflows/e2e-desktop-core.yml b/.github/workflows/e2e-desktop-core.yml new file mode 100644 index 0000000000..861be2dd68 --- /dev/null +++ b/.github/workflows/e2e-desktop-core.yml @@ -0,0 +1,99 @@ +name: E2E Desktop core + +# The core Desktop suite (apps/desktop/e2e/core): a small, deterministic lane +# that drives a real Electron build against a real `hermes serve` and a +# scripted loopback model. It guards the failures users must never see: +# duplicated / reordered / lost transcript text, a backend that never becomes +# ready or leaves orphans, and broken clarify/approval round trips. +# +# Unlike the legacy visual lane (e2e-desktop.yml, disabled for flakiness) this +# one runs with retries: 0 and one worker; every wait is event-driven, and the +# suite was held to 10/10 consecutive green full runs on a loaded host before +# it was wired in. + +on: + workflow_call: + +permissions: + contents: read + +concurrency: + group: e2e-desktop-core-${{ github.ref }} + cancel-in-progress: true + +jobs: + core: + name: Desktop core E2E (Linux) + runs-on: ubuntu-latest-32-core + timeout-minutes: 30 + steps: + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + + - name: Install system dependencies for Electron + run: | + sudo apt-get update -qq + sudo apt-get install -y -qq \ + xvfb \ + libgtk-3-0 libnotify4 libnss3 libxss1 libxtst6 \ + xdg-utils libatspi2.0-0 libdrm2 libgbm1 libasound2t64 + + - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 + with: + node-version: 26 + cache: npm + + - name: grab npm 12 + run: npm i -g npm@12 + + # Full npm ci (not --ignore-scripts): electron's postinstall downloads + # the binary we launch. + - uses: ./.github/actions/retry + with: + command: npm ci + + - name: Install uv + uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0 + with: + version: '0.9.28' + enable-cache: true + cache-dependency-glob: | + pyproject.toml + uv.lock + + - name: Set up Python 3.11 + uses: ./.github/actions/retry + with: + command: uv python install 3.11 + + - name: Install Python dependencies + uses: ./.github/actions/retry + with: + command: uv sync --locked --python 3.11 --extra all --extra dev + + - name: Build the desktop app + working-directory: apps/desktop + run: npm run build + + - name: Run the core suite under xvfb + working-directory: apps/desktop + run: | + xvfb-run -a --server-args="-screen 0 1280x1024x24" \ + npx playwright test -c e2e/core/playwright.config.ts --reporter=list + env: + CI: 'true' + OPENROUTER_API_KEY: '' + OPENAI_API_KEY: '' + NOUS_API_KEY: '' + ANTHROPIC_API_KEY: '' + + - name: Upload failure artifacts + if: failure() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: desktop-core-e2e-${{ github.sha }} + path: | + apps/desktop/test-results/core + apps/desktop/playwright-report/core + retention-days: 14 + overwrite: true + if-no-files-found: ignore diff --git a/apps/desktop/playwright.config.ts b/apps/desktop/playwright.config.ts index 100c002252..bc3fd9e6b8 100644 --- a/apps/desktop/playwright.config.ts +++ b/apps/desktop/playwright.config.ts @@ -30,8 +30,9 @@ export default defineConfig({ testDir: './e2e', /* ...except `*.unit.test.ts`, which covers the e2e HELPERS (no Electron, no * app) and is owned by the vitest `electron` project. Without this the - * default testMatch would claim those files too and run them twice. */ - testIgnore: '**/*.unit.test.ts', + * default testMatch would claim those files too and run them twice. + * e2e/core/ has its own config (retries 0, one worker) and CI job. */ + testIgnore: ['**/*.unit.test.ts', 'core/**'], /* The desktop app can take a while to bootstrap on cold CI runners — 90 s * per test gives us headroom without masking real hangs. */ timeout: 90_000, From 1d574487ebb79cb5b3a14280962e18b4c7c8f084 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:32:38 -0700 Subject: [PATCH 035/112] test(desktop-core): 180 s per-test cap, keep traces on a cancelled lane, honest onboarding header A stalled stream burned the 600 s per-test timeout until the 30-min job timeout cancelled the lane, and the failure()-only upload then saved no traces. The CLI --reporter flag also dropped the config's html report. The onboarding spec is a smoke test; #120005 needs a non-default profile. --- .github/workflows/e2e-desktop-core.yml | 4 ++-- apps/desktop/e2e/core/onboarding-first-chat.spec.ts | 5 +++-- apps/desktop/e2e/core/playwright.config.ts | 7 ++++--- 3 files changed, 9 insertions(+), 7 deletions(-) diff --git a/.github/workflows/e2e-desktop-core.yml b/.github/workflows/e2e-desktop-core.yml index 861be2dd68..3eb37accae 100644 --- a/.github/workflows/e2e-desktop-core.yml +++ b/.github/workflows/e2e-desktop-core.yml @@ -78,7 +78,7 @@ jobs: working-directory: apps/desktop run: | xvfb-run -a --server-args="-screen 0 1280x1024x24" \ - npx playwright test -c e2e/core/playwright.config.ts --reporter=list + npx playwright test -c e2e/core/playwright.config.ts env: CI: 'true' OPENROUTER_API_KEY: '' @@ -87,7 +87,7 @@ jobs: ANTHROPIC_API_KEY: '' - name: Upload failure artifacts - if: failure() + if: failure() || cancelled() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: desktop-core-e2e-${{ github.sha }} diff --git a/apps/desktop/e2e/core/onboarding-first-chat.spec.ts b/apps/desktop/e2e/core/onboarding-first-chat.spec.ts index 6715bf4c0c..1ee4a8aeb6 100644 --- a/apps/desktop/e2e/core/onboarding-first-chat.spec.ts +++ b/apps/desktop/e2e/core/onboarding-first-chat.spec.ts @@ -5,8 +5,9 @@ * The user picks the local / custom endpoint path and enters the fake * provider's URL; the real backend probes it (/v1/models), persists the * assignment, and the overlay closes. Then the first chat must obey the same - * transcript oracle as everything else — this is the path where #120005 - * (every streamed word painted twice) first shipped. + * transcript oracle as everything else. A smoke test for onboarding followed + * by a first chat; the #120005 double-paint needs a non-default profile and is + * guarded by transcript-integrity.spec.ts. * * Invariants: the onboarding result is what the backend persisted and what * the model is actually called with (config base_url == entered URL; the diff --git a/apps/desktop/e2e/core/playwright.config.ts b/apps/desktop/e2e/core/playwright.config.ts index 3a28ac90b8..f8a3d846b9 100644 --- a/apps/desktop/e2e/core/playwright.config.ts +++ b/apps/desktop/e2e/core/playwright.config.ts @@ -11,13 +11,14 @@ import { defineConfig } from '@playwright/test' * - no visual baselines / always-on screenshots; artifacts only on failure. * - one worker: every spec owns a real Electron + `hermes serve`; running * them concurrently on a loaded runner is the timing margin we refuse. - * - generous per-test timeout; every wait inside is event-driven with its - * own deadline, so a long timeout never slows a green run. + * - 180 s per test (green runs take 6-36 s): a stalled stream fails the one + * test fast instead of the 30-min job timeout cancelling the whole lane + * before it reports. */ export default defineConfig({ testDir: '.', testMatch: '*.spec.ts', - timeout: 600_000, + timeout: 180_000, expect: { timeout: 60_000 }, retries: 0, workers: 1, From f02031a53116b055d83b2464f74fd5e480a6bdf4 Mon Sep 17 00:00:00 2001 From: robbyczgw-cla Date: Wed, 23 Sep 2026 09:58:21 +0000 Subject: [PATCH 036/112] plugin-catalog: pin web-search-plus to v4.3.0 --- plugin-catalog/web-search-plus.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/plugin-catalog/web-search-plus.yaml b/plugin-catalog/web-search-plus.yaml index 83633d5935..eaf7cd521f 100644 --- a/plugin-catalog/web-search-plus.yaml +++ b/plugin-catalog/web-search-plus.yaml @@ -1,12 +1,12 @@ name: web-search-plus repo: https://github.com/robbyczgw-cla/hermes-web-search-plus -sha: 90e9c3acbeafa18c256fd68ac3298e6e663fc981 +sha: 03457489d9862d66fef8af3d222b2558f38e67c4 description: Multi-provider web search, URL extraction, quality reports, and opt-in research mode maintainer: robbyczgw-cla tier: community category: web docs_url: https://github.com/robbyczgw-cla/hermes-web-search-plus#readme -version: "4.2.1" +version: "4.3.0" capabilities: provides_tools: - web_extract_plus From 00e17daa01be9f99a84de80cfe689ea34dfc6f4d Mon Sep 17 00:00:00 2001 From: robbyczgw-cla Date: Wed, 23 Sep 2026 21:07:33 +0000 Subject: [PATCH 037/112] plugin-catalog: bump web-search-plus to 4.3.1 --- plugin-catalog/web-search-plus.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/plugin-catalog/web-search-plus.yaml b/plugin-catalog/web-search-plus.yaml index eaf7cd521f..d33fbee316 100644 --- a/plugin-catalog/web-search-plus.yaml +++ b/plugin-catalog/web-search-plus.yaml @@ -1,12 +1,12 @@ name: web-search-plus repo: https://github.com/robbyczgw-cla/hermes-web-search-plus -sha: 03457489d9862d66fef8af3d222b2558f38e67c4 +sha: 6751abc222c9bbbeb88473e5293b535701d95f82 description: Multi-provider web search, URL extraction, quality reports, and opt-in research mode maintainer: robbyczgw-cla tier: community category: web docs_url: https://github.com/robbyczgw-cla/hermes-web-search-plus#readme -version: "4.3.0" +version: "4.3.1" capabilities: provides_tools: - web_extract_plus From 811a1c01fc0d193e8af650f686649827803d6b9b Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:41:49 -0700 Subject: [PATCH 038/112] chore(plugin-catalog): web-search-plus disclosure line (config.yaml read-only, local subprocesses) --- plugin-catalog/web-search-plus.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/plugin-catalog/web-search-plus.yaml b/plugin-catalog/web-search-plus.yaml index d33fbee316..5159651514 100644 --- a/plugin-catalog/web-search-plus.yaml +++ b/plugin-catalog/web-search-plus.yaml @@ -1,7 +1,7 @@ name: web-search-plus repo: https://github.com/robbyczgw-cla/hermes-web-search-plus sha: 6751abc222c9bbbeb88473e5293b535701d95f82 -description: Multi-provider web search, URL extraction, quality reports, and opt-in research mode +description: "Multi-provider web search, URL extraction, quality reports, and opt-in research mode. Disclosure — reads the operator's HERMES_HOME/config.yaml (read-only) for its own Desktop settings block (country, language, max_results, auto_routing, searxng_url); may spawn its own search.py and an operator-configured local donsetch CLI as subprocesses (argv, no shell)." maintainer: robbyczgw-cla tier: community category: web From e30f8c63442cee37c2df9af3cff7af113fd4ae11 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 07:29:47 -0700 Subject: [PATCH 039/112] fix(update): a `hermes update` killed mid-pull no longer bricks every entry point Git moves the checkout file by file and HEAD last, so an update killed while "Pulling updates" (Ctrl-C, closed terminal, power loss, OOM) left HEAD on the old commit with a prefix of the files already new. That mixed tree failed at import in every entry point, `hermes update` included (`cannot import name 'load_yaml_file_readonly' from 'utils'`), and only a manual `git reset --hard` could bring the install back. The updater now brackets the fast-forward (and the divergence reconcile) with `.git/hermes-update-pull` naming its pid, the pre-pull commit and the target. `hermes_cli.main` calls `_early_recovery.restore_interrupted_pull()` right after `hermes_bootstrap`, before any other checkout import: when the marker's owner is gone and HEAD is still the pre-pull commit, every path that differs between the two commits goes back to HEAD (the tree the venv was built for), the dead git's index.lock is cleared, and the command re-execs itself (modules already imported, main.py included, may be the half-written ones), so a rerun `hermes update` repairs the install and then updates normally. Paths the update does not change keep local edits; the updater's autostash stays in `git stash list`. Our own pid is never treated as a live owner (containers hand a retry the killed updater's pid). --- hermes_cli/_early_recovery.py | 108 ++++++++++++++++++ hermes_cli/main.py | 16 ++- hermes_cli/update_cmd.py | 9 ++ .../test_update_interrupted_pull.py | 104 +++++++++++++++++ 4 files changed, 232 insertions(+), 5 deletions(-) create mode 100644 tests/hermes_cli/test_update_interrupted_pull.py diff --git a/hermes_cli/_early_recovery.py b/hermes_cli/_early_recovery.py index e049fb7253..813e911fe8 100644 --- a/hermes_cli/_early_recovery.py +++ b/hermes_cli/_early_recovery.py @@ -8,6 +8,7 @@ the known-fragile core packages, using the pins from pyproject.toml). from __future__ import annotations +import contextlib import importlib import os import shutil @@ -210,6 +211,113 @@ def _marker_owner_is_live(marker: Path) -> bool: return False +# ``hermes update`` writes this into ``.git/`` right before git moves the checkout and removes it once +# git is done. Git rewrites the tree file by file and moves HEAD last, so an update killed in between +# leaves HEAD on the old commit with a prefix of the files already new; that mixed tree fails at import +# in every entry point, ``hermes update`` included. A marker whose owner is gone means exactly that. +INTERRUPTED_PULL_MARKER = "hermes-update-pull" +# A fast-forward takes seconds; past this a "live" owner pid is a recycled one. +_INTERRUPTED_PULL_MAX_AGE_SECONDS = 10 * 60 + + +def interrupted_pull_marker(root: Path) -> Path: + return root / ".git" / INTERRUPTED_PULL_MARKER + + +def restore_interrupted_pull(project_root: Path | None = None) -> bool: + """Put back the files a killed ``hermes update`` had half-moved to the new commit. Never raises. + + Returns True when files were restored: modules this process already imported may be the + half-written ones, so the caller must relaunch (``relaunch_after_restore``). + + Fast path (no marker) is one ``stat``. Acts only when the marker's owner is gone and HEAD is still + the pre-pull commit: every path that differs between it and the pull target returns to HEAD (the + commit the venv was built for), so the install is whole again and ``hermes update`` redoes the + update from the start. Paths the update does not change keep any local edits; the updater's + autostash (if any) stays in ``git stash list``. + """ + try: + root = _project_root() if project_root is None else project_root + marker = interrupted_pull_marker(root) + if not marker.is_file() or _pytest_owns_live_checkout(root): + return False + fields = dict(line.partition("=")[::2] for line in marker.read_text(encoding="utf-8").splitlines()) + try: + owner = int(fields.get("pid", "")) + except ValueError: + owner = -1 + # Our own pid is never the owner: this runs at startup, and containers hand a retry the + # killed updater's pid. + if (owner != os.getpid() and _pid_is_running(owner) + and time.time() - marker.stat().st_mtime < _INTERRUPTED_PULL_MAX_AGE_SECONDS): + return False + + def git(*args: str, stdin: str | None = None) -> subprocess.CompletedProcess: + return subprocess.run(["git", "--literal-pathspecs", "-C", str(root), *args], input=stdin, + capture_output=True, text=True, timeout=120, + stdin=None if stdin is not None else subprocess.DEVNULL) + + pre, target = fields.get("pre", "").strip(), fields.get("target", "").strip() + if not pre or not target or git("rev-parse", "HEAD").stdout.strip() != pre: + marker.unlink() # git finished (HEAD moved) or the marker is unusable + return False + diff = git("diff", "--name-status", "-z", "--no-renames", pre, target) + if diff.returncode != 0: + return False + parts = diff.stdout.split("\0") + added = [p for s, p in zip(parts[::2], parts[1::2]) if s == "A"] + kept = [p for s, p in zip(parts[::2], parts[1::2]) if s and s != "A"] + # The dead git's index lock would refuse every command below. + (root / ".git" / "index.lock").unlink(missing_ok=True) + status = git("status", "--porcelain=v1", "-z", "--untracked-files=all", "--no-renames") + dirty = {entry[3:] for entry in status.stdout.split("\0") if len(entry) > 3} + if status.returncode != 0 or dirty.isdisjoint(added + kept): + if status.returncode == 0: + marker.unlink() # the killed git never reached the tree: nothing to put back + return False + print("⚠ A previous `hermes update` was killed while git was writing the new code — " + f"restoring the checkout to {pre[:10]}...", file=sys.stderr) + ok = True + if kept: + ok = git("restore", "--source=HEAD", "--staged", "--worktree", "--pathspec-from-file=-", + "--pathspec-file-nul", stdin="\0".join(kept)).returncode == 0 + if added and ok: + ok = git("rm", "-q", "--cached", "--ignore-unmatch", "--pathspec-from-file=-", + "--pathspec-file-nul", stdin="\0".join(added)).returncode == 0 + for rel in added: + path = root / rel + path.unlink(missing_ok=True) + with contextlib.suppress(OSError): + os.removedirs(path.parent) # stops at the first non-empty dir + if ok: + marker.unlink() + print(" ✓ Checkout restored; `hermes update` updates it again.", file=sys.stderr) + if fields.get("stash", "").strip(): + print(f" Your local changes are still in the update's stash ({fields['stash'].strip()}).", + file=sys.stderr) + return True + print(f" ✗ Could not restore it automatically. Recover with: git -C {root} reset --hard {pre}", + file=sys.stderr) + except Exception: + pass # Never block launch — the import that follows surfaces the real error. + return False + + +def relaunch_after_restore() -> None: + """Re-run this command from the restored tree; never returns. + + Everything imported so far (this package, ``hermes_bootstrap``, ``hermes_cli.main`` itself) may + be the killed git's new files, and they would run against the restored old tree. + """ + argv = [sys.executable, *sys.orig_argv[1:]] + sys.stdout.flush() + sys.stderr.flush() + if sys.platform == "win32": + # os.execv on Windows spawns and exits, detaching the console's wait on us. + sys.exit(subprocess.call(argv)) # windows-footgun: ok — interactive child keeps our console + os.execv(sys.executable, argv) + + def _pinned_specs(packages: list[str], project_root: Path) -> list[str]: """Map bare package names to their pinned specs from pyproject.toml. diff --git a/hermes_cli/main.py b/hermes_cli/main.py index be58df9a90..525203cc94 100644 --- a/hermes_cli/main.py +++ b/hermes_cli/main.py @@ -17,6 +17,16 @@ try: except ModuleNotFoundError: pass +# A `hermes update` killed while git was writing the new tree leaves a mix of old and new files that +# fails at the next import, whichever it is — put the old tree back before importing anything else +# from the checkout, then rerun the command (this module may itself be one of the new files). +# ``_early_recovery`` is stdlib-only and imported unguarded on purpose: same package +# dir, so if IT can't import nothing in hermes_cli can. +from hermes_cli import _early_recovery as _early_recovery_mod + +if _early_recovery_mod.restore_interrupted_pull(): + _early_recovery_mod.relaunch_after_restore() + # Windows: neutralize CPython's ``platform._syscmd_ver`` before anything else # imports — it shells out ``cmd /c ver`` and flashes a console when this # process is windowless (pythonw gateway, kanban workers). No-op on POSIX. @@ -44,13 +54,9 @@ _startup_fast.normalize_hermes_home_env() # the hermes_cli.config/env_loader imports further down would then crash before # main() reaches _recover_from_interrupted_install(). ``_early_recovery`` is # stdlib-only (safe on a corrupted venv) and repairs just enough to finish this -# import; the marker lifecycle stays with the full recovery path. Its own -# import is unguarded on purpose: same package dir, so if IT can't import -# nothing in hermes_cli can. +# import; the marker lifecycle stays with the full recovery path. # It is also the canonical home of the probe/repair tables reused by the full recovery path below. See # #57828. -from hermes_cli import _early_recovery as _early_recovery_mod - try: _early_recovery_mod.recover_if_needed() except Exception: diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index 6a1f42ddc6..01c4f0a273 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -19,6 +19,7 @@ from pathlib import Path from hermes_cli.config import get_hermes_home # noqa: F401 (re-exported; patched via update_cmd) from hermes_cli import update_handoff as _update_handoff from hermes_cli.update_cmd_common import _best_effort +from hermes_cli._early_recovery import interrupted_pull_marker from hermes_constants import get_default_hermes_root, project_venv_dir, venv_python_path # Re-exports: every split-module name stays reachable (and monkeypatchable) as update_cmd.. @@ -866,11 +867,19 @@ def _pull_updates( # critical-path file (PR #28452 incident: orphan merge-conflict markers in hermes_cli/config.py bricked # every user who ran ``hermes update`` for the 7 minutes between the bad commit and the fix landing). pre_pull_sha = _capture_head_sha(git_cmd, _m().PROJECT_ROOT) + # Git moves the tree file by file and HEAD last: if this process dies in between, the next launch + # of any entry point finds this marker and puts the old tree back (_early_recovery). + pull_marker = interrupted_pull_marker(_m().PROJECT_ROOT) + with _best_effort('Could not write the interrupted-pull marker: %s'): + pull_marker.write_text( + f"pid={os.getpid()}\npre={pre_pull_sha}\ntarget=origin/{branch}\nstash={auto_stash_ref or ''}\n", + encoding="utf-8") try: # merge --ff-only the already-fetched ref instead of `git pull`, which would do a # SECOND network fetch; identical in effect given the fresh tracking ref. if _git_run(git_cmd, ["merge", "--ff-only", f"origin/{branch}"]).returncode != 0: _reconcile_diverged_checkout(git_cmd, branch, pre_pull_sha) + pull_marker.unlink(missing_ok=True) # git is done: the tree is whole again _rollback_if_pulled_syntax_error(git_cmd, pre_pull_sha) update_succeeded = True finally: diff --git a/tests/hermes_cli/test_update_interrupted_pull.py b/tests/hermes_cli/test_update_interrupted_pull.py new file mode 100644 index 0000000000..50b4485650 --- /dev/null +++ b/tests/hermes_cli/test_update_interrupted_pull.py @@ -0,0 +1,104 @@ +"""A `hermes update` killed while git writes the new tree must leave a recoverable install. + +Git rewrites the checkout file by file and moves HEAD last, so a kill in between leaves HEAD on the +old commit with some files already new — a mix that fails at import in every entry point. The +updater brackets the move with a marker; the next launch (``_early_recovery``, before any other +checkout import) puts the old tree back so ``hermes update`` can simply run again. +""" + +from __future__ import annotations + +import os +import subprocess +from pathlib import Path + +import pytest + +from hermes_cli import _early_recovery as er +from hermes_cli import update_cmd + + +def _git(root: Path, *args: str) -> str: + return subprocess.run(["git", "-C", str(root), *args], check=True, capture_output=True, + text=True).stdout.strip() + + +@pytest.fixture +def checkout(tmp_path, monkeypatch): + """An install at commit A whose fetched ``origin/main`` is B (modifies, deletes, adds a package).""" + origin = tmp_path / "origin" + origin.mkdir() + _git(origin, "init", "-q", "-b", "main") + _git(origin, "config", "user.email", "t@example.invalid") + _git(origin, "config", "user.name", "t") + for name, body in {"utils.py": "OLD = 1\n", "gone.py": "x = 1\n", "notes.md": "user file\n"}.items(): + (origin / name).write_text(body) + _git(origin, "add", "-A") + _git(origin, "commit", "-qm", "A") + (origin / "utils.py").write_text("NEW = 1\n") + (origin / "gone.py").unlink() + (origin / "newpkg").mkdir() + (origin / "newpkg" / "__init__.py").write_text("from utils import NEW\n") + _git(origin, "add", "-A") + _git(origin, "commit", "-qm", "B") + root = tmp_path / "install" + _git(tmp_path, "clone", "-q", str(origin), str(root)) + _git(root, "reset", "-q", "--hard", "HEAD~1") + monkeypatch.setattr("hermes_cli.main.PROJECT_ROOT", root) + return root, _git(root, "rev-parse", "HEAD"), _git(root, "rev-parse", "origin/main") + + +def _pull(root: Path) -> None: + update_cmd._pull_updates(["git"], "main", None, prompt_for_restore=False, gw_input_fn=None, + discard_local_changes=False, keep_stash=False) + + +def _kill_mid_pull(root: Path, monkeypatch) -> None: + """The fast-forward writes one new file, holds index.lock, and the process dies.""" + real = update_cmd._git_run + + def dying_git_run(git_cmd, args, *a, **kw): + if args[:1] == ["merge"]: + (root / "newpkg").mkdir() + (root / "newpkg" / "__init__.py").write_text("from utils import NEW\n") + (root / ".git" / "index.lock").touch() + raise KeyboardInterrupt # SIGKILL: nothing after this line of the updater runs + return real(git_cmd, args, *a, **kw) + + monkeypatch.setattr(update_cmd, "_git_run", dying_git_run) + with pytest.raises(KeyboardInterrupt): + _pull(root) + monkeypatch.setattr(update_cmd, "_git_run", real) + + +def test_killed_pull_is_restored_on_next_launch_and_update_reruns(checkout, monkeypatch): + root, a, b = checkout + (root / "notes.md").write_text("edited while bricked\n") # a path the update never touches + _kill_mid_pull(root, monkeypatch) + assert _git(root, "rev-parse", "HEAD") == a # the torn state: HEAD old, newpkg already new + marker = er.interrupted_pull_marker(root) + # A retry in a container gets the killed updater's pid: our own pid is never a live owner. + assert f"pid={os.getpid()}" in marker.read_text() + + assert er.restore_interrupted_pull(root) is True, "restored files mean the caller must relaunch" + + assert _git(root, "rev-parse", "HEAD") == a + assert _git(root, "status", "--porcelain", "--untracked-files=all") == "M notes.md" + assert not (root / "newpkg").exists() and not (root / ".git" / "index.lock").exists() + assert not marker.exists() + (root / "notes.md").write_text("user file\n") + _pull(root) # `hermes update` again: a normal fast-forward + assert _git(root, "rev-parse", "HEAD") == b and not marker.exists() + + +def test_pull_marker_of_a_live_updater_is_left_alone(checkout, monkeypatch): + root, a, _b = checkout + _kill_mid_pull(root, monkeypatch) + marker = er.interrupted_pull_marker(root) + # Another `hermes` launched while an update is mid-pull must not race its git. + marker.write_text(marker.read_text().replace(f"pid={os.getpid()}", f"pid={os.getppid()}")) + + assert er.restore_interrupted_pull(root) is False + + assert marker.exists() and (root / "newpkg" / "__init__.py").exists() + assert (root / ".git" / "index.lock").exists() From f035bf872ec5e50cbab8bc5903a586106a6dc0d9 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:04:22 +0000 Subject: [PATCH 040/112] fix(update): interrupted-pull restore never touches the user's own edits Review blocker on this PR: the pull marker was removed only on the success path. When `_reconcile_diverged_checkout` exited (merge conflict on a custom branch, failed reset) the marker stayed with a dead owner, the updater told the user to `git stash apply`, and the next `hermes` launch ran `git restore` over every dirty file upstream also changed, wiping the re-applied work (and restoring files mid-merge, leaving a dangling MERGE_HEAD). - The marker is removed on every exit of the git phase except KeyboardInterrupt, which reaches git too and can leave a torn tree. - The marker records the resolved target commit, not `origin/`, so a later `git fetch` cannot widen the restore. - Startup restores only paths whose content is exactly the target blob (per `git hash-object`, so autocrlf/filters match) or, for deletions, that are gone; anything else is the user's and stays. - Startup does nothing while a merge, rebase, cherry-pick or revert is in progress. - The marker lives in the real git dir, so linked-worktree installs are covered too; the bare `except Exception: pass` is narrowed and reports. - New test pins the main.py wiring: the restore/relaunch runs before any other hermes_cli module is imported. --- hermes_cli/_early_recovery.py | 100 +++++++++++++----- hermes_cli/update_cmd.py | 20 ++-- .../test_update_interrupted_pull.py | 89 +++++++++++++--- 3 files changed, 160 insertions(+), 49 deletions(-) diff --git a/hermes_cli/_early_recovery.py b/hermes_cli/_early_recovery.py index 813e911fe8..16f0fe1841 100644 --- a/hermes_cli/_early_recovery.py +++ b/hermes_cli/_early_recovery.py @@ -211,36 +211,80 @@ def _marker_owner_is_live(marker: Path) -> bool: return False -# ``hermes update`` writes this into ``.git/`` right before git moves the checkout and removes it once -# git is done. Git rewrites the tree file by file and moves HEAD last, so an update killed in between -# leaves HEAD on the old commit with a prefix of the files already new; that mixed tree fails at import -# in every entry point, ``hermes update`` included. A marker whose owner is gone means exactly that. +# ``hermes update`` writes this into the git dir right before git moves the checkout and removes it +# once git has exited (a kill is the only exit that keeps it). Git rewrites the tree file by file and +# moves HEAD last, so an update killed in between leaves HEAD on the old commit with a prefix of the +# files already new; that mixed tree fails at import in every entry point, ``hermes update`` included. INTERRUPTED_PULL_MARKER = "hermes-update-pull" # A fast-forward takes seconds; past this a "live" owner pid is a recycled one. _INTERRUPTED_PULL_MAX_AGE_SECONDS = 10 * 60 +# The user (or a killed updater) is mid-operation: its own state files own the tree. +_GIT_OPERATION_IN_PROGRESS = ("MERGE_HEAD", "CHERRY_PICK_HEAD", "REVERT_HEAD", "rebase-merge", "rebase-apply") +_REGULAR_FILE_MODES = ("100644", "100755") + + +def _git_dir(root: Path) -> Path: + """``root``'s git dir: ``.git`` itself, or where a linked worktree's ``.git`` file points.""" + dot_git = root / ".git" + if dot_git.is_file(): + text = dot_git.read_text(encoding="utf-8").strip() + if text.startswith("gitdir:"): + return root / text[len("gitdir:"):].strip() + return dot_git def interrupted_pull_marker(root: Path) -> Path: - return root / ".git" / INTERRUPTED_PULL_MARKER + return _git_dir(root) / INTERRUPTED_PULL_MARKER + + +def _paths_git_wrote(git, root: Path, pre: str, target: str) -> tuple[list[str], list[str]] | None: + """Paths the killed git already moved to ``target``: (restore from HEAD, delete as added). + + Only a path whose content is exactly ``target``'s blob (or, for a deletion, that is gone) counts; + anything else is the user's own edit — e.g. a re-applied stash — and is left alone. + """ + diff = git("diff", "--raw", "-z", "--no-renames", "--no-abbrev", pre, target) + if diff.returncode != 0: + return None + parts = diff.stdout.split("\0") + entries = [] # (status, path, target blob) + for meta, path in zip(parts[::2], parts[1::2]): + old_mode, new_mode, _old_blob, new_blob, status = meta.lstrip(":").split() + if status == "D" and old_mode in _REGULAR_FILE_MODES: + entries.append((status, path, None)) + elif status != "D" and new_mode in _REGULAR_FILE_MODES: + entries.append((status, path, new_blob)) + present = [path for _s, path, blob in entries if blob and (root / path).is_file()] + hashed = git("hash-object", "--stdin-paths", stdin="\n".join(present) + "\n") if present else None + if hashed is not None and hashed.returncode != 0: + return None + worktree_blob = dict(zip(present, hashed.stdout.split() if hashed else ())) + restore, added = [], [] + for status, path, blob in entries: + written = (not (root / path).exists()) if blob is None else worktree_blob.get(path) == blob + if written: + (added if status == "A" else restore).append(path) + return restore, added def restore_interrupted_pull(project_root: Path | None = None) -> bool: - """Put back the files a killed ``hermes update`` had half-moved to the new commit. Never raises. + """Put back the files a killed ``hermes update`` had half-moved to the new commit. Returns True when files were restored: modules this process already imported may be the half-written ones, so the caller must relaunch (``relaunch_after_restore``). - Fast path (no marker) is one ``stat``. Acts only when the marker's owner is gone and HEAD is still - the pre-pull commit: every path that differs between it and the pull target returns to HEAD (the - commit the venv was built for), so the install is whole again and ``hermes update`` redoes the - update from the start. Paths the update does not change keep any local edits; the updater's - autostash (if any) stays in ``git stash list``. + Fast path (no marker) is one or two ``stat`` calls. Acts only when the marker's owner is gone, + HEAD is still the pre-pull commit and no merge/rebase is in progress; then every path whose + content is the pull target's returns to HEAD (the commit the venv was built for), so the install is + whole again and ``hermes update`` redoes the update from the start. Local edits are never touched; + the updater's autostash (if any) stays in ``git stash list``. """ try: root = _project_root() if project_root is None else project_root marker = interrupted_pull_marker(root) if not marker.is_file() or _pytest_owns_live_checkout(root): return False + git_dir = marker.parent fields = dict(line.partition("=")[::2] for line in marker.read_text(encoding="utf-8").splitlines()) try: owner = int(fields.get("pid", "")) @@ -251,36 +295,33 @@ def restore_interrupted_pull(project_root: Path | None = None) -> bool: if (owner != os.getpid() and _pid_is_running(owner) and time.time() - marker.stat().st_mtime < _INTERRUPTED_PULL_MAX_AGE_SECONDS): return False + if any((git_dir / name).exists() for name in _GIT_OPERATION_IN_PROGRESS): + return False def git(*args: str, stdin: str | None = None) -> subprocess.CompletedProcess: return subprocess.run(["git", "--literal-pathspecs", "-C", str(root), *args], input=stdin, - capture_output=True, text=True, timeout=120, - stdin=None if stdin is not None else subprocess.DEVNULL) + capture_output=True, text=True, encoding="utf-8", errors="replace", + timeout=120, stdin=None if stdin is not None else subprocess.DEVNULL) pre, target = fields.get("pre", "").strip(), fields.get("target", "").strip() if not pre or not target or git("rev-parse", "HEAD").stdout.strip() != pre: marker.unlink() # git finished (HEAD moved) or the marker is unusable return False - diff = git("diff", "--name-status", "-z", "--no-renames", pre, target) - if diff.returncode != 0: + written = _paths_git_wrote(git, root, pre, target) + if written is None: return False - parts = diff.stdout.split("\0") - added = [p for s, p in zip(parts[::2], parts[1::2]) if s == "A"] - kept = [p for s, p in zip(parts[::2], parts[1::2]) if s and s != "A"] - # The dead git's index lock would refuse every command below. - (root / ".git" / "index.lock").unlink(missing_ok=True) - status = git("status", "--porcelain=v1", "-z", "--untracked-files=all", "--no-renames") - dirty = {entry[3:] for entry in status.stdout.split("\0") if len(entry) > 3} - if status.returncode != 0 or dirty.isdisjoint(added + kept): - if status.returncode == 0: - marker.unlink() # the killed git never reached the tree: nothing to put back + restore, added = written + if not restore and not added: + marker.unlink() # the killed git never reached the tree: nothing to put back return False print("⚠ A previous `hermes update` was killed while git was writing the new code — " f"restoring the checkout to {pre[:10]}...", file=sys.stderr) + # The dead git's index lock would refuse every command below. + (git_dir / "index.lock").unlink(missing_ok=True) ok = True - if kept: + if restore: ok = git("restore", "--source=HEAD", "--staged", "--worktree", "--pathspec-from-file=-", - "--pathspec-file-nul", stdin="\0".join(kept)).returncode == 0 + "--pathspec-file-nul", stdin="\0".join(restore)).returncode == 0 if added and ok: ok = git("rm", "-q", "--cached", "--ignore-unmatch", "--pathspec-from-file=-", "--pathspec-file-nul", stdin="\0".join(added)).returncode == 0 @@ -298,8 +339,9 @@ def restore_interrupted_pull(project_root: Path | None = None) -> bool: return True print(f" ✗ Could not restore it automatically. Recover with: git -C {root} reset --hard {pre}", file=sys.stderr) - except Exception: - pass # Never block launch — the import that follows surfaces the real error. + except (OSError, subprocess.SubprocessError, ValueError) as exc: + # Never block launch: the import that follows surfaces any real breakage. + print(f"⚠ Could not check for an interrupted `hermes update`: {exc}", file=sys.stderr) return False diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index 01c4f0a273..702742bd18 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -868,17 +868,25 @@ def _pull_updates( # every user who ran ``hermes update`` for the 7 minutes between the bad commit and the fix landing). pre_pull_sha = _capture_head_sha(git_cmd, _m().PROJECT_ROOT) # Git moves the tree file by file and HEAD last: if this process dies in between, the next launch - # of any entry point finds this marker and puts the old tree back (_early_recovery). + # of any entry point finds this marker and puts the old tree back (_early_recovery). The target is + # the resolved commit, so a later `git fetch` cannot widen what that restore considers. pull_marker = interrupted_pull_marker(_m().PROJECT_ROOT) + target_sha = (_git_run(git_cmd, ["rev-parse", f"origin/{branch}^{{commit}}"]).stdout or "").strip() with _best_effort('Could not write the interrupted-pull marker: %s'): pull_marker.write_text( - f"pid={os.getpid()}\npre={pre_pull_sha}\ntarget=origin/{branch}\nstash={auto_stash_ref or ''}\n", + f"pid={os.getpid()}\npre={pre_pull_sha}\ntarget={target_sha}\nstash={auto_stash_ref or ''}\n", encoding="utf-8") try: - # merge --ff-only the already-fetched ref instead of `git pull`, which would do a - # SECOND network fetch; identical in effect given the fresh tracking ref. - if _git_run(git_cmd, ["merge", "--ff-only", f"origin/{branch}"]).returncode != 0: - _reconcile_diverged_checkout(git_cmd, branch, pre_pull_sha) + try: + # merge --ff-only the already-fetched ref instead of `git pull`, which would do a + # SECOND network fetch; identical in effect given the fresh tracking ref. + if _git_run(git_cmd, ["merge", "--ff-only", f"origin/{branch}"]).returncode != 0: + _reconcile_diverged_checkout(git_cmd, branch, pre_pull_sha) + except KeyboardInterrupt: + raise # Ctrl-C reached git too (same process group): the tree may be torn, keep the marker + except BaseException: + pull_marker.unlink(missing_ok=True) # git exited on its own (sys.exit on conflict/reset failure) + raise pull_marker.unlink(missing_ok=True) # git is done: the tree is whole again _rollback_if_pulled_syntax_error(git_cmd, pre_pull_sha) update_succeeded = True diff --git a/tests/hermes_cli/test_update_interrupted_pull.py b/tests/hermes_cli/test_update_interrupted_pull.py index 50b4485650..da89822326 100644 --- a/tests/hermes_cli/test_update_interrupted_pull.py +++ b/tests/hermes_cli/test_update_interrupted_pull.py @@ -10,6 +10,7 @@ from __future__ import annotations import os import subprocess +import sys from pathlib import Path import pytest @@ -20,7 +21,7 @@ from hermes_cli import update_cmd def _git(root: Path, *args: str) -> str: return subprocess.run(["git", "-C", str(root), *args], check=True, capture_output=True, - text=True).stdout.strip() + text=True, encoding="utf-8").stdout.strip() @pytest.fixture @@ -31,14 +32,15 @@ def checkout(tmp_path, monkeypatch): _git(origin, "init", "-q", "-b", "main") _git(origin, "config", "user.email", "t@example.invalid") _git(origin, "config", "user.name", "t") - for name, body in {"utils.py": "OLD = 1\n", "gone.py": "x = 1\n", "notes.md": "user file\n"}.items(): - (origin / name).write_text(body) + for name, body in {"utils.py": "OLD = 1\n", "other.py": "a = 1\n", "gone.py": "x = 1\n"}.items(): + (origin / name).write_text(body, encoding="utf-8", newline="") _git(origin, "add", "-A") _git(origin, "commit", "-qm", "A") - (origin / "utils.py").write_text("NEW = 1\n") + (origin / "utils.py").write_text("NEW = 1\n", encoding="utf-8", newline="") + (origin / "other.py").write_text("a = 2\n", encoding="utf-8", newline="") (origin / "gone.py").unlink() (origin / "newpkg").mkdir() - (origin / "newpkg" / "__init__.py").write_text("from utils import NEW\n") + (origin / "newpkg" / "__init__.py").write_text("from utils import NEW\n", encoding="utf-8", newline="") _git(origin, "add", "-A") _git(origin, "commit", "-qm", "B") root = tmp_path / "install" @@ -59,8 +61,9 @@ def _kill_mid_pull(root: Path, monkeypatch) -> None: def dying_git_run(git_cmd, args, *a, **kw): if args[:1] == ["merge"]: + (root / "utils.py").write_text("NEW = 1\n", encoding="utf-8", newline="") (root / "newpkg").mkdir() - (root / "newpkg" / "__init__.py").write_text("from utils import NEW\n") + (root / "newpkg" / "__init__.py").write_text("from utils import NEW\n", encoding="utf-8", newline="") (root / ".git" / "index.lock").touch() raise KeyboardInterrupt # SIGKILL: nothing after this line of the updater runs return real(git_cmd, args, *a, **kw) @@ -73,32 +76,90 @@ def _kill_mid_pull(root: Path, monkeypatch) -> None: def test_killed_pull_is_restored_on_next_launch_and_update_reruns(checkout, monkeypatch): root, a, b = checkout - (root / "notes.md").write_text("edited while bricked\n") # a path the update never touches _kill_mid_pull(root, monkeypatch) - assert _git(root, "rev-parse", "HEAD") == a # the torn state: HEAD old, newpkg already new + assert _git(root, "rev-parse", "HEAD") == a # the torn state: HEAD old, utils + newpkg already new marker = er.interrupted_pull_marker(root) # A retry in a container gets the killed updater's pid: our own pid is never a live owner. - assert f"pid={os.getpid()}" in marker.read_text() + recorded = marker.read_text(encoding="utf-8") + assert f"pid={os.getpid()}" in recorded and f"target={b}" in recorded # the commit, not the ref name + # The user re-applies their stash to a file the update also changes (git had not written it yet). + (root / "other.py").write_text("a = 1 # my edit\n", encoding="utf-8", newline="") assert er.restore_interrupted_pull(root) is True, "restored files mean the caller must relaunch" assert _git(root, "rev-parse", "HEAD") == a - assert _git(root, "status", "--porcelain", "--untracked-files=all") == "M notes.md" + assert _git(root, "status", "--porcelain", "--untracked-files=all") == "M other.py" + assert (root / "other.py").read_text(encoding="utf-8") == "a = 1 # my edit\n", "the user's edit survives" assert not (root / "newpkg").exists() and not (root / ".git" / "index.lock").exists() assert not marker.exists() - (root / "notes.md").write_text("user file\n") + (root / "other.py").write_text("a = 1\n", encoding="utf-8", newline="") _pull(root) # `hermes update` again: a normal fast-forward assert _git(root, "rev-parse", "HEAD") == b and not marker.exists() +def test_failed_update_leaves_no_marker_and_restore_never_touches_user_work(checkout): + """sys.exit on a merge conflict is not a kill: nothing may be restored behind the user's back.""" + root, a, b = checkout + _git(root, "config", "user.email", "t@example.invalid") + _git(root, "config", "user.name", "t") + _git(root, "checkout", "-q", "-b", "mywork") + (root / "other.py").write_text("a = 'mine'\n", encoding="utf-8", newline="") + _git(root, "commit", "-qam", "local work that conflicts upstream") + with pytest.raises(SystemExit): + _pull(root) + marker = er.interrupted_pull_marker(root) + assert not marker.exists() + + # Even a leftover marker (an older updater, or a kill mid-reconcile) stays out of the user's way: + # following the printed advice leaves a merge in progress, and edits git never wrote are theirs. + stale = f"pid=0\npre={_git(root, 'rev-parse', 'HEAD')}\ntarget={b}\nstash=\n" + marker.write_text(stale, encoding="utf-8", newline="") + merge = subprocess.run(["git", "-C", str(root), "merge", "origin/main"], + capture_output=True, text=True, encoding="utf-8") + assert (root / ".git" / "MERGE_HEAD").exists(), merge.stdout + merge.stderr + (root / "utils.py").write_text("OLD = 1 # resolved by hand\n", encoding="utf-8", newline="") + before = _git(root, "status", "--porcelain", "--untracked-files=all") + assert er.restore_interrupted_pull(root) is False + assert _git(root, "status", "--porcelain", "--untracked-files=all") == before + _git(root, "reset", "-q", "--hard") # the user gives up on the merge + (root / "utils.py").write_text("OLD = 1 # my stash, re-applied\n", encoding="utf-8", newline="") + assert er.restore_interrupted_pull(root) is False + assert (root / "utils.py").read_text(encoding="utf-8") == "OLD = 1 # my stash, re-applied\n" + assert not marker.exists(), "git wrote nothing: the marker is spent" + + def test_pull_marker_of_a_live_updater_is_left_alone(checkout, monkeypatch): root, a, _b = checkout _kill_mid_pull(root, monkeypatch) marker = er.interrupted_pull_marker(root) # Another `hermes` launched while an update is mid-pull must not race its git. - marker.write_text(marker.read_text().replace(f"pid={os.getpid()}", f"pid={os.getppid()}")) - - assert er.restore_interrupted_pull(root) is False + updater = subprocess.Popen([sys.executable, "-c", "import time; time.sleep(60)"]) + try: + live = marker.read_text(encoding="utf-8").replace(f"pid={os.getpid()}", f"pid={updater.pid}") + marker.write_text(live, encoding="utf-8", newline="") + assert er.restore_interrupted_pull(root) is False + finally: + updater.kill() + updater.wait() assert marker.exists() and (root / "newpkg" / "__init__.py").exists() assert (root / ".git" / "index.lock").exists() + + +def test_main_restores_before_importing_anything_else_from_the_checkout(): + """``hermes_cli.main`` itself may be one of the half-written files: the restore and relaunch run + before any other checkout module is imported.""" + probe = ( + "import sys\n" + "from hermes_cli import _early_recovery as er\n" + "er.restore_interrupted_pull = lambda: True\n" + "def relaunch():\n" + " print(','.join(sorted(m for m in sys.modules if m.split('.')[0] == 'hermes_cli'))); sys.exit(42)\n" + "er.relaunch_after_restore = relaunch\n" + "import hermes_cli.main\n" + ) + root = Path(er.__file__).resolve().parent.parent + result = subprocess.run([sys.executable, "-c", probe], cwd=root, env={**os.environ, "PYTHONPATH": str(root)}, + capture_output=True, text=True, encoding="utf-8", timeout=60) + assert result.returncode == 42, result.stderr + assert result.stdout.strip() == "hermes_cli,hermes_cli._early_recovery,hermes_cli.main" From 6efb2296c96f3fc1ce02fa3f6066cb6d1cbd9c41 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:28:51 +0000 Subject: [PATCH 041/112] test(update): live A/B harness for the interrupted-pull repair Disposable origin/install, real autostash + _pull_updates, real SIGKILLs and the real entry point; covers the user-edit cases from the review (kill before git wrote, failed update + manual merge, torn tree with a user edit) and a linked-worktree install. One VERDICT line. --- evals/update_pipeline/interrupted_pull_ab.sh | 147 +++++++++++++++++++ 1 file changed, 147 insertions(+) create mode 100755 evals/update_pipeline/interrupted_pull_ab.sh diff --git a/evals/update_pipeline/interrupted_pull_ab.sh b/evals/update_pipeline/interrupted_pull_ab.sh new file mode 100755 index 0000000000..87c6b937af --- /dev/null +++ b/evals/update_pipeline/interrupted_pull_ab.sh @@ -0,0 +1,147 @@ +#!/usr/bin/env bash +# Live A/B for the interrupted-pull repair (hermes_cli/_early_recovery.py::restore_interrupted_pull). +# +# evals/update_pipeline/interrupted_pull_ab.sh