From f2241e866ff4e5346aa43af6a1bbc40bc32b9853 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:05:27 -0700 Subject: [PATCH 001/104] test: scripted recording loopback LLM provider for core E2E suites One real OpenAI-compatible server (JSON + SSE) that records every request and scripts tool calls, reasoning, long streams and provider faults, so E2E tests drive the real client stack instead of mocking the agent loop. --- tests/fakes/fake_llm_provider.py | 381 +++++++++++++++++++++++++++++++ 1 file changed, 381 insertions(+) create mode 100644 tests/fakes/fake_llm_provider.py diff --git a/tests/fakes/fake_llm_provider.py b/tests/fakes/fake_llm_provider.py new file mode 100644 index 0000000000..0550c6f9f3 --- /dev/null +++ b/tests/fakes/fake_llm_provider.py @@ -0,0 +1,381 @@ +"""Scripted, recording loopback LLM provider for end-to-end tests. + +One real HTTP server on 127.0.0.1 that speaks the OpenAI Chat Completions +wire format (JSON and SSE streaming). Every request body is recorded so a test +can assert on exactly what Hermes sent (history integrity, prompt-cache prefix +stability, routing/credential isolation), and every response is scripted so a +test can drive tool calls, reasoning, long streams and provider faults through +the real client stack instead of mocking the agent loop. + +Main-turn requests (those carrying ``tools``) consume the script in order; +requests without ``tools`` are auxiliary calls (title generation, compression +summaries, judges) and are answered by ``aux`` so they never eat a scripted +turn. When the script is exhausted, main turns answer ``default_text``. + +Usage:: + + with FakeLLMServer([ToolCall("terminal", {"command": "echo hi"}), Text("done")]) as srv: + write_hermes_home(home, srv.base_url) + ...run hermes... + assert srv.main_requests()[1]["messages"][-1]["role"] == "tool" + +Run standalone for manual probes: ``python -m tests.fakes.fake_llm_provider 8765``. +""" + +from __future__ import annotations + +import json +import threading +import time +from dataclasses import dataclass, field +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path +from typing import Any, Callable, Union + +MODEL_ID = "fake-model" + + +# Scripted responses --------------------------------------------------------- + + +@dataclass +class Text: + """A plain assistant answer, optionally with reasoning, streamed in chunks.""" + + text: str + reasoning: str | None = None + chunk_chars: int = 8 + delay_per_chunk: float = 0.0 + prompt_tokens: int = 100 + completion_tokens: int = 20 + cached_tokens: int = 0 + + +@dataclass +class ToolCall: + """One assistant turn issuing one or more tool calls. + + ``calls`` may be a single ``(name, args)`` or pass ``name``/``args`` directly; + ``parallel`` adds more calls to the same assistant message. + """ + + name: str + args: dict[str, Any] = field(default_factory=dict) + parallel: list[tuple[str, dict[str, Any]]] = field(default_factory=list) + text: str | None = None + + +@dataclass +class Error: + """An HTTP error response (429/500/400 ...).""" + + status: int = 500 + message: str = "scripted failure" + retry_after: float | None = None + + +@dataclass +class Hang: + """Accept the request and never answer within ``seconds``.""" + + seconds: float = 3600.0 + + +@dataclass +class DropMidStream: + """Stream ``text[:after_chars]`` then close the socket without a finish chunk.""" + + text: str = "partial answer that never finishes" + after_chars: int = 12 + + +Response = Union[Text, ToolCall, Error, Hang, DropMidStream] +Responder = Callable[[dict[str, Any]], Response] + + +# Server --------------------------------------------------------------------- + + +class FakeLLMServer: + """Threaded loopback provider. Use as a context manager.""" + + def __init__( + self, + script: list[Response] | Responder | None = None, + *, + default_text: str = "ok", + aux: Responder | None = None, + api_key: str | None = None, + ) -> None: + self._script: list[Response] = list(script) if isinstance(script, list) else [] + self._responder: Responder | None = script if callable(script) else None + self.default_text = default_text + self._aux = aux or (lambda _req: Text("Fake summary of the earlier conversation.")) + self.expected_api_key = api_key + self.requests: list[dict[str, Any]] = [] + self._lock = threading.Lock() + self._stop = threading.Event() + self._server: ThreadingHTTPServer | None = None + self._thread: threading.Thread | None = None + self._tool_seq = 0 + + # lifecycle + def __enter__(self) -> "FakeLLMServer": + self.start() + return self + + def __exit__(self, *_exc: object) -> None: + self.stop() + + def start(self) -> None: + server = ThreadingHTTPServer(("127.0.0.1", 0), _handler_for(self)) + server.daemon_threads = True + self._server = server + self._thread = threading.Thread(target=server.serve_forever, name="fake-llm", daemon=True) + self._thread.start() + + def stop(self) -> None: + self._stop.set() + if self._server is not None: + self._server.shutdown() + self._server.server_close() + + @property + def port(self) -> int: + assert self._server is not None, "server not started" + return self._server.server_address[1] + + @property + def base_url(self) -> str: + return f"http://127.0.0.1:{self.port}/v1" + + # scripting + def push(self, *responses: Response) -> None: + with self._lock: + self._script.extend(responses) + + def _next_main(self, record: dict[str, Any]) -> Response: + if self._responder is not None: + return self._responder(record) + with self._lock: + if self._script: + return self._script.pop(0) + return Text(self.default_text) + + # inspection + def main_requests(self) -> list[dict[str, Any]]: + return [r["body"] for r in self.requests if r["kind"] == "main"] + + def aux_requests(self) -> list[dict[str, Any]]: + return [r["body"] for r in self.requests if r["kind"] == "aux"] + + def wait_for_requests(self, n: int, timeout: float = 30.0, kind: str = "main") -> None: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + if sum(1 for r in self.requests if r["kind"] == kind) >= n: + return + time.sleep(0.02) + raise AssertionError(f"expected {n} {kind} requests, saw {len(self.requests)} total") + + def next_tool_call_id(self) -> str: + with self._lock: + self._tool_seq += 1 + return f"call_fake_{self._tool_seq}" + + +def _handler_for(server: FakeLLMServer) -> type[BaseHTTPRequestHandler]: + class Handler(BaseHTTPRequestHandler): + protocol_version = "HTTP/1.1" + + def log_message(self, *_a: object) -> None: + pass + + def _send_json(self, status: int, payload: dict[str, Any], headers: dict[str, str] | None = None) -> None: + body = json.dumps(payload).encode() + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + for k, v in (headers or {}).items(): + self.send_header(k, v) + self.end_headers() + self.wfile.write(body) + + def do_GET(self) -> None: # noqa: N802 + if self.path.rstrip("/").endswith("/models"): + self._send_json(200, {"object": "list", "data": [ + {"id": MODEL_ID, "object": "model", "context_length": 128000}, + ]}) + return + self._send_json(404, {"error": {"message": "not found"}}) + + def do_POST(self) -> None: # noqa: N802 + raw = self.rfile.read(int(self.headers.get("Content-Length", 0) or 0)) + try: + body = json.loads(raw or b"{}") + except json.JSONDecodeError: + self._send_json(400, {"error": {"message": "invalid json"}}) + return + auth = self.headers.get("Authorization", "") + kind = "main" if body.get("tools") else "aux" + record = { + "path": self.path, + "kind": kind, + "auth": auth, + "headers": {k.lower(): v for k, v in self.headers.items()}, + "body": body, + "t": time.time(), + } + with server._lock: + server.requests.append(record) + if server.expected_api_key is not None and auth != f"Bearer {server.expected_api_key}": + self._send_json(401, {"error": {"message": "invalid api key", "type": "authentication_error"}}) + return + if not self.path.rstrip("/").endswith("/chat/completions"): + self._send_json(404, {"error": {"message": f"unsupported path {self.path}"}}) + return + resp = server._next_main(record) if kind == "main" else server._aux(record) + record["response"] = type(resp).__name__ + self._respond(resp, bool(body.get("stream"))) + + # response rendering + def _respond(self, resp: Response, stream: bool) -> None: + if isinstance(resp, Error): + headers = {"Retry-After": str(resp.retry_after)} if resp.retry_after is not None else {} + self._send_json(resp.status, {"error": {"message": resp.message, "type": "server_error"}}, headers) + return + if isinstance(resp, Hang): + server._stop.wait(resp.seconds) + return + if isinstance(resp, DropMidStream): + self._start_sse() + self._sse(_chunk({"role": "assistant", "content": ""})) + self._sse(_chunk({"content": resp.text[: resp.after_chars]})) + self.wfile.flush() + self.close_connection = True + return + message, finish, usage = _message_for(resp, server) + if not stream: + self._send_json(200, { + "id": "chatcmpl-fake", "object": "chat.completion", "created": int(time.time()), + "model": MODEL_ID, + "choices": [{"index": 0, "message": message, "finish_reason": finish}], + "usage": usage, + }) + return + self._start_sse() + self._sse(_chunk({"role": "assistant", "content": ""})) + if isinstance(resp, Text): + if resp.reasoning: + for piece in _pieces(resp.reasoning, resp.chunk_chars): + self._sse(_chunk({"reasoning_content": piece})) + for piece in _pieces(resp.text, resp.chunk_chars): + if resp.delay_per_chunk: + time.sleep(resp.delay_per_chunk) + self._sse(_chunk({"content": piece})) + else: + if message.get("content"): + self._sse(_chunk({"content": message["content"]})) + for i, tc in enumerate(message["tool_calls"]): + self._sse(_chunk({"tool_calls": [{ + "index": i, "id": tc["id"], "type": "function", + "function": {"name": tc["function"]["name"], "arguments": ""}, + }]})) + self._sse(_chunk({"tool_calls": [{ + "index": i, "function": {"arguments": tc["function"]["arguments"]}, + }]})) + last = _chunk({}, finish) + last["usage"] = usage + self._sse(last) + self.wfile.write(b"data: [DONE]\n\n") + self.wfile.flush() + self.close_connection = True + + def _start_sse(self) -> None: + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.send_header("Cache-Control", "no-cache") + self.send_header("Connection", "close") + self.end_headers() + + def _sse(self, payload: dict[str, Any]) -> None: + self.wfile.write(f"data: {json.dumps(payload)}\n\n".encode()) + self.wfile.flush() + + return Handler + + +def _pieces(text: str, size: int) -> list[str]: + size = max(1, size) + return [text[i : i + size] for i in range(0, len(text), size)] or [""] + + +def _chunk(delta: dict[str, Any], finish: str | None = None) -> dict[str, Any]: + return { + "id": "chatcmpl-fake", "object": "chat.completion.chunk", "created": int(time.time()), + "model": MODEL_ID, "choices": [{"index": 0, "delta": delta, "finish_reason": finish}], + } + + +def _message_for(resp: Text | ToolCall, server: FakeLLMServer) -> tuple[dict[str, Any], str, dict[str, Any]]: + if isinstance(resp, Text): + message: dict[str, Any] = {"role": "assistant", "content": resp.text} + if resp.reasoning: + message["reasoning_content"] = resp.reasoning + usage = { + "prompt_tokens": resp.prompt_tokens, + "completion_tokens": resp.completion_tokens, + "total_tokens": resp.prompt_tokens + resp.completion_tokens, + "prompt_tokens_details": {"cached_tokens": resp.cached_tokens}, + } + return message, "stop", usage + calls = [(resp.name, resp.args), *resp.parallel] + tool_calls = [ + {"id": server.next_tool_call_id(), "type": "function", + "function": {"name": name, "arguments": json.dumps(args)}} + for name, args in calls + ] + message = {"role": "assistant", "content": resp.text, "tool_calls": tool_calls} + usage = {"prompt_tokens": 100, "completion_tokens": 10, "total_tokens": 110} + return message, "tool_calls", usage + + +# HERMES_HOME wiring --------------------------------------------------------- + + +def write_hermes_home( + home: Path, + base_url: str, + *, + api_key: str = "sk-fake-e2e", + extra_config: str = "", +) -> Path: + """Write a minimal config.yaml + .env routing the main model to ``base_url``. + + Auxiliary tasks use the same endpoint (``auto`` resolves to the main + provider), retries are capped so fault tests finish quickly, and no real + provider credential is ever present. + """ + home.mkdir(parents=True, exist_ok=True) + (home / "config.yaml").write_text( + "model:\n" + " provider: custom\n" + f" base_url: {base_url}\n" + f" default: {MODEL_ID}\n" + " context_length: 128000\n" + "agent:\n" + " api_max_retries: 1\n" + + extra_config, + encoding="utf-8", + ) + (home / ".env").write_text(f"OPENAI_API_KEY={api_key}\n", encoding="utf-8") + return home + + +if __name__ == "__main__": # pragma: no cover - manual probe entry point + import sys + + port = int(sys.argv[1]) if len(sys.argv) > 1 else 0 + srv = FakeLLMServer() + srv._server = ThreadingHTTPServer(("127.0.0.1", port), _handler_for(srv)) + print(f"fake provider on {srv.base_url}", flush=True) + srv._server.serve_forever() From 462afbe58da888c70435a6c78327bfd8628ab3ac Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:21:45 -0700 Subject: [PATCH 002/104] test: fake provider gains StallMidStream, Raw bodies and verbatim tool-call arguments Chaos lane (C8 agent-turn liveness) needs an upstream that opens the SSE stream and then goes silent without closing, arbitrary malformed bodies, and truncated/malformed tool-call argument JSON. All additions are new response types or accept a str where a dict was required, so existing scripts are unchanged. --- tests/fakes/fake_llm_provider.py | 45 ++++++++++++++++++++++++++++---- 1 file changed, 40 insertions(+), 5 deletions(-) diff --git a/tests/fakes/fake_llm_provider.py b/tests/fakes/fake_llm_provider.py index 0550c6f9f3..b99f56e18e 100644 --- a/tests/fakes/fake_llm_provider.py +++ b/tests/fakes/fake_llm_provider.py @@ -56,12 +56,13 @@ class ToolCall: """One assistant turn issuing one or more tool calls. ``calls`` may be a single ``(name, args)`` or pass ``name``/``args`` directly; - ``parallel`` adds more calls to the same assistant message. + ``parallel`` adds more calls to the same assistant message. A ``str`` args is + sent verbatim as the ``arguments`` string (e.g. malformed/truncated JSON). """ name: str - args: dict[str, Any] = field(default_factory=dict) - parallel: list[tuple[str, dict[str, Any]]] = field(default_factory=list) + args: dict[str, Any] | str = field(default_factory=dict) + parallel: list[tuple[str, dict[str, Any] | str]] = field(default_factory=list) text: str | None = None @@ -89,7 +90,26 @@ class DropMidStream: after_chars: int = 12 -Response = Union[Text, ToolCall, Error, Hang, DropMidStream] +@dataclass +class StallMidStream: + """Open the SSE stream, send ``text[:after_chars]``, then go silent for ``seconds`` + without closing (a wedged upstream that keeps the socket open).""" + + text: str = "partial answer that stalls" + after_chars: int = 8 + seconds: float = 3600.0 + + +@dataclass +class Raw: + """Send an arbitrary body verbatim (malformed JSON, HTML error pages, ...).""" + + body: str = "this is not json" + status: int = 200 + content_type: str = "application/json" + + +Response = Union[Text, ToolCall, Error, Hang, DropMidStream, StallMidStream, Raw] Responder = Callable[[dict[str, Any]], Response] @@ -246,6 +266,21 @@ def _handler_for(server: FakeLLMServer) -> type[BaseHTTPRequestHandler]: if isinstance(resp, Hang): server._stop.wait(resp.seconds) return + if isinstance(resp, Raw): + body = resp.body.encode() + self.send_response(resp.status) + self.send_header("Content-Type", resp.content_type) + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + return + if isinstance(resp, StallMidStream): + self._start_sse() + self._sse(_chunk({"role": "assistant", "content": ""})) + self._sse(_chunk({"content": resp.text[: resp.after_chars]})) + server._stop.wait(resp.seconds) + self.close_connection = True + return if isinstance(resp, DropMidStream): self._start_sse() self._sse(_chunk({"role": "assistant", "content": ""})) @@ -331,7 +366,7 @@ def _message_for(resp: Text | ToolCall, server: FakeLLMServer) -> tuple[dict[str calls = [(resp.name, resp.args), *resp.parallel] tool_calls = [ {"id": server.next_tool_call_id(), "type": "function", - "function": {"name": name, "arguments": json.dumps(args)}} + "function": {"name": name, "arguments": args if isinstance(args, str) else json.dumps(args)}} for name, args in calls ] message = {"role": "assistant", "content": resp.text, "tool_calls": tool_calls} From 52dfbd2eed3cb0c05b691da16c15caff5d830ebc Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:44:17 -0700 Subject: [PATCH 003/104] ci: run tests/e2e per-file in parallel with a 30-minute budget The core E2E suites spawn real processes (serve, gateway, tui_gateway, MCP servers, concurrent SQLite writers). One subprocess per file keeps them isolated and parallel instead of one sequential pytest. --- .github/workflows/tests.yml | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 0f2a221f67..9e82ba34c0 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -152,7 +152,7 @@ jobs: e2e: runs-on: ubuntu-latest - timeout-minutes: 15 + timeout-minutes: 30 steps: - name: Checkout code uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 @@ -212,9 +212,12 @@ jobs: run: uv cache prune --ci - name: Run e2e tests + # One subprocess per file, in parallel: the core suites spawn real + # processes (serve, gateway, tui_gateway, MCP servers, SQLite writers) + # and must not share interpreter state. run: | source .venv/bin/activate - python -m pytest tests/e2e/ -v --tb=short + scripts/run_tests.sh --include-integration tests/e2e env: OPENROUTER_API_KEY: "" OPENAI_API_KEY: "" From adc35881ed5325c8e7b89a69dd12b5eaab7540e3 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:42:32 -0700 Subject: [PATCH 004/104] test: multi-process SQLite torture chamber for state.db integrity (C1) Class C1: state.db corruption, WAL generations unlinked under a live holder, lost/duplicated rows, repair destroying data, fd leaks. Every role is a real OS process on one WAL state.db through the production SessionDB: gateway- and TUI-like writers with intent/ack journals, a dashboard reader living across episodes, short-lived openers (SessionDB, bare sqlite3, real `hermes sessions list/stats`), FTS rebuild/optimize and repair_state_db_schema. Nine seeded episodes inject kill -9 mid-write, SIGTERM graceful close, POSIX lock cancellation by a stray in-process open/close, chmod flips, concurrent and killed FTS rebuilds, repair against a live and an offline store, FTS corruption and a whole-fleet SIGKILL, then assert the same invariants: integrity_check ok and still WAL, no child holding a (deleted) db/-wal/-shm (/proc fd monitor), every acked append stored exactly once, counts only grow, repair never lowers them, FTS == canonical rows and search finds acked rows once, no role errors, bounded fds for the long-lived reader and churner. Red-proof: disabling the OFD WAL lock guard (75e155ab09b, whose revert no longer applies cleanly) makes the lock_cancellation episode kill the gateway writer with SIGBUS / fail sibling openers, 3/3 runs. --- tests/e2e/core/__init__.py | 0 tests/e2e/core/sqlite/__init__.py | 0 tests/e2e/core/sqlite/_helpers.py | 377 +++++++++++++++++ tests/e2e/core/sqlite/_roles.py | 315 +++++++++++++++ tests/e2e/core/sqlite/test_torture_chamber.py | 380 ++++++++++++++++++ 5 files changed, 1072 insertions(+) create mode 100644 tests/e2e/core/__init__.py create mode 100644 tests/e2e/core/sqlite/__init__.py create mode 100644 tests/e2e/core/sqlite/_helpers.py create mode 100644 tests/e2e/core/sqlite/_roles.py create mode 100644 tests/e2e/core/sqlite/test_torture_chamber.py diff --git a/tests/e2e/core/__init__.py b/tests/e2e/core/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/tests/e2e/core/sqlite/__init__.py b/tests/e2e/core/sqlite/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/tests/e2e/core/sqlite/_helpers.py b/tests/e2e/core/sqlite/_helpers.py new file mode 100644 index 0000000000..649d7d09c1 --- /dev/null +++ b/tests/e2e/core/sqlite/_helpers.py @@ -0,0 +1,377 @@ +"""Harness for the multi-process SQLite torture chamber (issue class C1: state.db integrity). + +One real ``state.db`` in WAL mode under ``tmp_path``; every role is a separate OS process running +``_roles.py`` against it (see that module for the journal/report protocol). This module owns: + +* process lifecycle (spawn / ready / stop / SIGTERM / SIGKILL, PIDs recorded, nothing else touched); +* a background ``/proc//fd`` monitor over OUR children that records any ``(deleted)`` ``-wal``/``-shm`` + (or main-file) descriptor — the kernel-level signature of a WAL generation unlinked under a live holder; +* the invariant checks, done in the test process with short-lived bare ``sqlite3`` connections only (the + test process never imports ``hermes_state``, so it never becomes a foreign holder of the file). + +Reuses the conformance harness's deadline polling / SIGKILL reaping (``tests/conformance/persistence``). +""" + +from __future__ import annotations + +import json +import os +import random +import signal +import sqlite3 +import subprocess +import sys +import threading +import time +import zlib +from pathlib import Path + +from tests.conformance.persistence._harness import REPO_ROOT, kill9_and_reap, wait_for + +ROLES = Path(__file__).with_name("_roles.py") +DEFAULT_SEED = 20260923 +SEED_ENV = "HERMES_SQLITE_TORTURE_SEED" + + +def base_seed() -> int: + return int(os.environ.get(SEED_ENV) or DEFAULT_SEED) + + +def episode_seed(label: str) -> int: + return base_seed() + zlib.crc32(label.encode()) + + +def child_env(home: Path, hermes_home: Path) -> dict: + """Probe hygiene: private HOME/HERMES_HOME, no provider credentials, repo importable.""" + env = {k: v for k, v in os.environ.items() if not k.endswith("_API_KEY")} + env.update({ + "HOME": str(home), + "HERMES_HOME": str(hermes_home), + "PYTHONUNBUFFERED": "1", + # The child's HERMES_HOME *is* this chamber's private home, which the live-DB guard reads as + # "the real Hermes root"; HOME/HERMES_HOME above already keep it off the production install. + "HERMES_STATE_DB_GUARD_BYPASS": "1", + "PYTHONPATH": os.pathsep.join(p for p in (str(REPO_ROOT), os.environ.get("PYTHONPATH", "")) if p), + }) + return env + + +class Chamber: + """A private HERMES_HOME + state.db plus the processes playing roles against it.""" + + def __init__(self, root: Path, *, journal_mode: str = "wal"): + self.root = root + self.home = root / "home" + self.hermes_home = self.home / ".hermes" + self.hermes_home.mkdir(parents=True, exist_ok=True) + (self.hermes_home / "config.yaml").write_text( + f"database:\n journal_mode: {journal_mode}\n", encoding="utf-8") + self.db = self.hermes_home / "state.db" + self.work = root / "work" + self.work.mkdir(exist_ok=True) + self.env = child_env(self.home, self.hermes_home) + self.procs: dict[str, subprocess.Popen] = {} + self.writer_runs: list[str] = [] + self.reader_seq = 0 + self.reader_name: str | None = None + self.deleted_hits: list[tuple[str, int, str]] = [] + self.fd_samples: dict[str, list[int]] = {} + self._lock = threading.Lock() + self._monitor_stop = threading.Event() + self._monitor = threading.Thread(target=self._scan_loop, name="deleted-fd-monitor", daemon=True) + self._monitor.start() + + # -- lifecycle --------------------------------------------------------------------------------- + def spawn(self, role: str, name: str, *, env: dict | None = None, **args) -> subprocess.Popen: + assert name not in self.procs, f"duplicate role name {name}" + payload = {"workdir": str(self.work), "name": name, "db": str(self.db), + "stop": str(self.work / f"{name}.stop"), **args} + stderr = open(self.work / f"{name}.stderr", "wb") # noqa: SIM115 - closed in reap() + proc = subprocess.Popen( + [sys.executable, str(ROLES), role, json.dumps(payload)], + cwd=str(REPO_ROOT), env={**self.env, **(env or {})}, stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, stderr=stderr, + ) + proc._stderr_file = stderr # type: ignore[attr-defined] + with self._lock: + self.procs[name] = proc + if role == "writer": + self.writer_runs.append(name) + return proc + + def spawn_cli(self, name: str, *argv: str) -> subprocess.Popen: + """A real `hermes …` CLI subprocess against this HERMES_HOME.""" + stderr = open(self.work / f"{name}.stderr", "wb") # noqa: SIM115 - closed in reap() + proc = subprocess.Popen( + [sys.executable, "-m", "hermes_cli.main", *argv], cwd=str(REPO_ROOT), env=self.env, + stdin=subprocess.DEVNULL, stdout=open(self.work / f"{name}.stdout", "wb"), stderr=stderr, # noqa: SIM115 + ) + proc._stderr_file = stderr # type: ignore[attr-defined] + with self._lock: + self.procs[name] = proc + return proc + + def stderr(self, name: str) -> str: + path = self.work / f"{name}.stderr" + return path.read_text(encoding="utf-8", errors="replace")[-3000:] if path.exists() else "" + + def wait_event(self, name: str, event: str, *, deadline: float = 60.0) -> dict: + found: list[dict] = [] + + def _has() -> bool: + found[:] = [e for e in self.events(name) if e.get("event") in (event, "error")] + return bool(found) + + self._wait(name, _has, f"{name}:{event}", deadline) + assert found[0].get("event") == event, f"{name} reported {found[0]}\n{self.stderr(name)}" + return found[0] + + def wait_acks(self, name: str, n: int, *, deadline: float = 60.0) -> None: + self._wait(name, lambda: len(self.journal(name)[1]) >= n, f"{n} acks from {name}", deadline) + + def _wait(self, name: str, predicate, what: str, deadline: float) -> None: + try: + wait_for(predicate, deadline=deadline, what=what, child=self.procs.get(name)) + except AssertionError as exc: + tail = [e for e in self.events(name) if e.get("event") == "error"][-1:] + raise AssertionError(f"{exc}\n{name} stderr:\n{self.stderr(name)}\n{tail}") from None + + def request_stop(self, name: str) -> None: + (self.work / f"{name}.stop").touch() + + def reap(self, name: str, *, deadline: float = 60.0) -> int: + proc = self.procs[name] + try: + rc = proc.wait(timeout=deadline) + except subprocess.TimeoutExpired: + kill9_and_reap(proc) + raise AssertionError(f"{name} did not exit within {deadline}s\n{self.stderr(name)}") from None + finally: + proc._stderr_file.close() # type: ignore[attr-defined] + return rc + + def stop(self, name: str, *, deadline: float = 60.0) -> int: + self.request_stop(name) + return self.reap(name, deadline=deadline) + + def sigterm(self, name: str) -> None: + self.procs[name].send_signal(signal.SIGTERM) + + def kill9(self, name: str) -> None: + proc = self.procs[name] + kill9_and_reap(proc) + proc._stderr_file.close() # type: ignore[attr-defined] + + def live(self) -> list[tuple[str, subprocess.Popen]]: + with self._lock: + return [(n, p) for n, p in self.procs.items() if p.poll() is None] + + def shutdown(self) -> None: + for name, proc in self.live(): + self.request_stop(name) + end = time.monotonic() + 20 + for _name, proc in list(self.procs.items()): + try: + proc.wait(timeout=max(0.1, end - time.monotonic())) + except subprocess.TimeoutExpired: + kill9_and_reap(proc) + f = getattr(proc, "_stderr_file", None) + if f is not None and not f.closed: + f.close() + self._monitor_stop.set() + self._monitor.join(timeout=5) + + # -- kernel truth: (deleted) sidecars held by our children -------------------------------------- + def _scan_loop(self) -> None: + targets = {str(self.db), f"{self.db}-wal", f"{self.db}-shm"} + while not self._monitor_stop.is_set(): + for name, proc in self.live(): + fd_dir = f"/proc/{proc.pid}/fd" + try: + fds = os.listdir(fd_dir) + except OSError: + continue + for fd in fds: + try: + link = os.readlink(f"{fd_dir}/{fd}") + except OSError: + continue + if link.endswith(" (deleted)") and link[: -len(" (deleted)")] in targets: + with self._lock: + self.deleted_hits.append((name, proc.pid, link)) + self._monitor_stop.wait(0.02) + + def deleted_hits_snapshot(self) -> list[tuple[str, int, str]]: + with self._lock: + return sorted(set(self.deleted_hits)) + + # -- reports ------------------------------------------------------------------------------------- + def events(self, name: str) -> list[dict]: + path = self.work / f"{name}.report" + if not path.exists(): + return [] + out = [] + for line in path.read_text(encoding="utf-8").splitlines(): + if line.strip(): + try: + out.append(json.loads(line)) + except json.JSONDecodeError: + pass # torn final line of a SIGKILLed child + return out + + def errors(self) -> list[tuple[str, dict]]: + return [(name, e) for name in list(self.procs) for e in self.events(name) if e.get("event") == "error"] + + def journal(self, name: str) -> tuple[dict[str, str], set[str]]: + """``(intents: tok -> sid, acked tokens)`` for one writer run.""" + path = self.work / f"{name}.journal" + intents: dict[str, str] = {} + acked: set[str] = set() + if not path.exists(): + return intents, acked + for line in path.read_text(encoding="utf-8").splitlines(): + parts = line.split() + if len(parts) >= 3 and parts[0] == "I": + intents[parts[1]] = parts[2] + elif len(parts) == 2 and parts[0] == "A": + acked.add(parts[1]) + return intents, acked + + +# -- invariants (bare sqlite3, short-lived connections) -------------------------------------------------- + + +def _connect(db: Path, *, ro: bool = True) -> sqlite3.Connection: + uri = f"file:{db}?mode=ro" if ro else f"file:{db}" + return sqlite3.connect(uri, uri=True, timeout=30.0) + + +def integrity_rows(db: Path) -> list[str]: + conn = _connect(db) + try: + return [r[0] for r in conn.execute("PRAGMA integrity_check").fetchall()] + except sqlite3.DatabaseError as exc: + return [f"integrity_check raised: {exc!r}"] + finally: + conn.close() + + +def journal_mode(db: Path) -> str: + conn = _connect(db) + try: + return str(conn.execute("PRAGMA journal_mode").fetchone()[0]).lower() + finally: + conn.close() + + +def counts(db: Path) -> dict[str, int]: + """Per-session canonical row counts (plus ``__total__``).""" + conn = _connect(db) + try: + rows = conn.execute("SELECT session_id, count(*) FROM messages GROUP BY session_id").fetchall() + finally: + conn.close() + out = {sid: n for sid, n in rows} + out["__total__"] = sum(out.values()) + return out + + +def token_counts(db: Path) -> dict[str, int]: + """How many canonical rows carry each writer token (first word of the content).""" + conn = _connect(db) + try: + rows = conn.execute("SELECT content FROM messages WHERE content LIKE 'TK%'").fetchall() + finally: + conn.close() + out: dict[str, int] = {} + for (content,) in rows: + tok = content.split(" ", 1)[0] + out[tok] = out.get(tok, 0) + 1 + return out + + +def fts_problems(db: Path, sample_tokens: list[str]) -> list[str]: + """FTS must mirror canonical rows exactly: docsize == source rows for each external-content index, + FTS5's own 'integrity-check' against the content, and a sample of acked tokens found exactly once.""" + problems: list[str] = [] + conn = _connect(db, ro=False) + try: + tables = {r[0] for r in conn.execute("SELECT name FROM sqlite_master").fetchall()} + meta = dict(conn.execute("SELECT key, value FROM state_meta").fetchall()) if "state_meta" in tables else {} + pending = {k: v for k, v in meta.items() if k.startswith("fts_rebuild") or k in ("fts_stale",)} + if pending: + problems.append(f"FTS recovery still pending in state_meta: {pending}") + canon = conn.execute("SELECT count(*) FROM messages").fetchone()[0] + pairs = [("messages_fts", "messages_fts_src"), ("messages_fts_trigram", "messages_fts_trigram_src")] + for fts, src in pairs: + if fts not in tables: + problems.append(f"{fts} missing") + continue + indexed = conn.execute(f"SELECT count(*) FROM {fts}_docsize").fetchone()[0] + expected = conn.execute(f"SELECT count(*) FROM {src}").fetchone()[0] + if indexed != expected: + problems.append(f"{fts}: {indexed} indexed rows vs {expected} source rows (canonical {canon})") + try: + conn.execute(f"INSERT INTO {fts}({fts}, rank) VALUES('integrity-check', 1)") + except sqlite3.DatabaseError as exc: + problems.append(f"{fts} integrity-check: {exc}") + for tok in sample_tokens: + hits = conn.execute("SELECT count(*) FROM messages_fts WHERE messages_fts MATCH ?", + (f'"{tok}"',)).fetchone()[0] + if hits != 1: + problems.append(f"session search finds acked {tok} {hits}x (expected 1)") + conn.rollback() + finally: + conn.close() + return problems + + +def exactly_once_problems(chamber: Chamber) -> list[str]: + """Every acknowledged append is stored exactly once; an unacknowledged in-flight append at most once; + no stored writer row that no writer ever intended.""" + stored = token_counts(chamber.db) + problems: list[str] = [] + intended: set[str] = set() + for run in chamber.writer_runs: + intents, acked = chamber.journal(run) + intended.update(intents) + for tok in acked: + if stored.get(tok, 0) != 1: + problems.append(f"{run}: acked {tok} stored {stored.get(tok, 0)}x") + for tok in set(intents) - acked: + if stored.get(tok, 0) > 1: + problems.append(f"{run}: in-flight {tok} stored {stored[tok]}x") + phantom = sorted(set(stored) - intended) + if phantom: + problems.append(f"{len(phantom)} stored rows no writer intended, e.g. {phantom[:3]}") + return problems + + +def compress_journal(chamber: Chamber, run: str) -> list[list[str]]: + """``C`` lines of an agent run: the turn bases each ``/compress here N`` promised to keep verbatim.""" + path = chamber.work / f"{run}.journal" + if not path.exists(): + return [] + return [line.split()[1:] for line in path.read_text(encoding="utf-8").splitlines() if line.startswith("C ")] + + +def token_flags(db: Path, token: str) -> list[tuple[int, int]]: + """``(active, compacted)`` of every canonical row whose content carries ``token``.""" + conn = _connect(db) + try: + rows = conn.execute("SELECT active, compacted FROM messages WHERE content LIKE ?", + (f"%{token}%",)).fetchall() + finally: + conn.close() + return sorted((int(a), int(c)) for a, c in rows) + + +def acked_tokens(chamber: Chamber, runs: list[str]) -> list[str]: + out: list[str] = [] + for run in runs: + out.extend(sorted(chamber.journal(run)[1])) + return out + + +def sample(rng: random.Random, items: list[str], k: int) -> list[str]: + return rng.sample(items, min(k, len(items))) diff --git a/tests/e2e/core/sqlite/_roles.py b/tests/e2e/core/sqlite/_roles.py new file mode 100644 index 0000000000..f9ddc59d1a --- /dev/null +++ b/tests/e2e/core/sqlite/_roles.py @@ -0,0 +1,315 @@ +"""Child-process roles for the SQLite torture chamber. + +Run as ``python _roles.py ``. Every role is a real OS process that opens the shared +``state.db`` through the production ``SessionDB`` (or, for the non-hermes opener, a bare ``sqlite3`` +connection) and reports through append-only files, so a ``kill -9`` loses nothing it already reported: + +* ``.journal`` — writers: ``I `` before an append, ``A `` once it returned. +* ``.report`` — JSON lines: ``ready`` / ``stats`` / ``error`` / ``closed`` events. + +Tokens are single FTS words (``TK`` + alnum) so the test can look every acknowledged append up by content, +through ``messages`` and through the FTS indexes. +""" + +from __future__ import annotations + +import json +import os +import signal +import sqlite3 +import sys +import time +import traceback +from pathlib import Path + +REPO = Path(__file__).resolve().parents[4] +sys.path.insert(0, str(REPO)) + +TOOL_FILLER = "y" * 3000 # longer than the FTS tool-content prefix, so the projection is exercised +_stop_requested = False + + +def _on_sigterm(_signum, _frame): + # Gateway/TUI shape: SIGTERM is a graceful shutdown that ends in SessionDB.close() (checkpoint on close). + global _stop_requested + _stop_requested = True + + +class Out: + """O_APPEND line writers: every line reaches the page cache before the next action, so it survives + a SIGKILL of this process (only a kernel crash could lose it).""" + + def __init__(self, workdir: Path, name: str): + self.journal_fd = os.open(workdir / f"{name}.journal", os.O_WRONLY | os.O_CREAT | os.O_APPEND, 0o600) + self.report_fd = os.open(workdir / f"{name}.report", os.O_WRONLY | os.O_CREAT | os.O_APPEND, 0o600) + + def journal(self, line: str) -> None: + os.write(self.journal_fd, (line + "\n").encode()) + + def report(self, **event) -> None: + event.setdefault("pid", os.getpid()) + event.setdefault("t", time.time()) + os.write(self.report_fd, (json.dumps(event) + "\n").encode()) + + +def _fd_count() -> int: + return len(os.listdir("/proc/self/fd")) if os.path.isdir("/proc/self/fd") else -1 + + +def _stray_close(db_path: Path) -> None: + """What a plugin, a tool read of ~/.hermes or a header probe does: open + close the live files in THIS + process. POSIX drops every lock this process holds on the inode (sqlite.org/howtocorrupt.html §2.2).""" + for suffix in ("", "-shm", "-wal"): + try: + os.close(os.open(str(db_path) + suffix, os.O_RDONLY)) + except OSError: + pass + + +def _stopping(stop_file: Path) -> bool: + return _stop_requested or stop_file.exists() + + +def role_writer(a: dict, out: Out) -> int: + """Long-lived writer (gateway- or TUI-like): appends until told to stop, journaling intent + ack.""" + from hermes_state import SessionDB + + db_path, stop_file = Path(a["db"]), Path(a["stop"]) + tag = a["tag"] + db = SessionDB(db_path=db_path) + for sid in a["sessions"]: + if db.get_session(sid) is None: + db.create_session(sid, a.get("source", "cli")) + out.report(event="ready", wal=bool(getattr(db, "_wal_active", False)), fds=_fd_count()) + i = 0 + max_appends = int(a.get("max_appends", 10**9)) + try: + while not _stopping(stop_file) and i < max_appends: + sid = a["sessions"][i % len(a["sessions"])] + tok = f"TK{tag}x{os.getpid()}x{i}" + if a.get("batch_every") and i % a["batch_every"] == 0: + msgs = [ + {"role": "user", "content": f"{tok}u please run it"}, + {"role": "tool", "content": f"{tok}t {TOOL_FILLER}", "tool_name": "terminal", + "tool_call_id": f"c{i}"}, + ] + out.journal(f"I {tok}u {sid} 1") + out.journal(f"I {tok}t {sid} 1") + db.append_messages_batch(sid, msgs) + out.journal(f"A {tok}u") + out.journal(f"A {tok}t") + else: + role = ("user", "assistant")[i % 2] + out.journal(f"I {tok} {sid} 1") + db.append_message(sid, role=role, content=f"{tok} turn {i} of {tag}") + out.journal(f"A {tok}") + i += 1 + if a.get("stray_every") and i % a["stray_every"] == 0: + _stray_close(db_path) + if a.get("pace"): + time.sleep(a["pace"]) + except BaseException as exc: # a refused/failed append is exactly what the suite exists to catch + out.report(event="error", error=repr(exc), tb=traceback.format_exc()[-3000:], appends=i) + return 2 + out.report(event="stats", appends=i, fds=_fd_count()) + try: + db.close() + except BaseException as exc: + out.report(event="error", error=f"close: {exc!r}", tb=traceback.format_exc()[-3000:]) + return 2 + out.report(event="closed", appends=i) + return 0 + + +def role_reader(a: dict, out: Out) -> int: + """Dashboard-like reader: opens a writable SessionDB at startup and polls until stopped. Reports any + per-session count that went DOWN (no compaction runs in this chamber) and its fd count per pass.""" + from hermes_state import SessionDB + + db_path, stop_file = Path(a["db"]), Path(a["stop"]) + db = SessionDB(db_path=db_path) + out.report(event="ready", fds=_fd_count()) + seen: dict[str, int] = {} + passes = 0 + try: + while not _stopping(stop_file): + for row in db.list_sessions_rich(limit=200): + sid = row["id"] + n = db.message_count(sid) + if n < seen.get(sid, 0): + out.report(event="error", error=f"count went down for {sid}: {seen[sid]} -> {n}") + seen[sid] = max(n, seen.get(sid, 0)) + passes += 1 + if passes % 5 == 0: + out.report(event="stats", passes=passes, fds=_fd_count(), total=sum(seen.values())) + time.sleep(a.get("pace", 0.05)) + except BaseException as exc: + out.report(event="error", error=repr(exc), tb=traceback.format_exc()[-3000:]) + return 2 + out.report(event="stats", passes=passes, fds=_fd_count(), total=sum(seen.values())) + db.close() + out.report(event="closed") + return 0 + + +def role_churn(a: dict, out: Out) -> int: + """`hermes sessions list` / doctor / cron-guard shape: open, read, close — ``iterations`` times in one + process, alternating the production SessionDB with a bare sqlite3 opener (sqlite3 shell, backup tool). + The fd count must not grow with the number of cycles.""" + from hermes_state import SessionDB + + db_path = Path(a["db"]) + start_fds = _fd_count() + fds_after_warmup = None + try: + for i in range(int(a["iterations"])): + if i % 2 == 0: + db = SessionDB(db_path=db_path) + db.message_count() + db.close() + else: + conn = sqlite3.connect(str(db_path), timeout=30.0) + conn.execute("SELECT count(*) FROM messages").fetchone() + conn.close() + if i == 3: + fds_after_warmup = _fd_count() + except BaseException as exc: + out.report(event="error", error=repr(exc), tb=traceback.format_exc()[-3000:]) + return 2 + out.report(event="stats", start_fds=start_fds, warm_fds=fds_after_warmup, end_fds=_fd_count(), + iterations=a["iterations"]) + return 0 + + +def role_opener(a: dict, out: Out) -> int: + """One short-lived process: open, count, close, exit (the last-close checkpoint path).""" + db_path = Path(a["db"]) + try: + if a.get("raw"): + conn = sqlite3.connect(str(db_path), timeout=30.0) + n = conn.execute("SELECT count(*) FROM messages").fetchone()[0] + conn.close() + else: + from hermes_state import SessionDB + db = SessionDB(db_path=db_path) + n = db.message_count() + db.close() + except BaseException as exc: + # may_fail: the chmod episode opens a read-only file; a clean refusal is correct, damage is not. + if a.get("may_fail"): + out.report(event="refused", error=repr(exc)) + return 0 + out.report(event="error", error=repr(exc), tb=traceback.format_exc()[-3000:]) + return 2 + out.report(event="stats", count=n) + return 0 + + +def _raw_count(db_path: Path) -> int: + conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True, timeout=30.0) + try: + return conn.execute("SELECT count(*) FROM messages").fetchone()[0] + finally: + conn.close() + + +def role_fts(a: dict, out: Out) -> int: + """Maintenance pass: full FTS rebuild + optimize through SessionDB (cross-process admission).""" + from hermes_state import SessionDB + + db = SessionDB(db_path=Path(a["db"])) + out.report(event="ready") + try: + rebuilt = db.rebuild_fts() + optimized = db.optimize_fts() + except BaseException as exc: + out.report(event="error", error=repr(exc), tb=traceback.format_exc()[-3000:]) + return 2 + db.close() + out.report(event="stats", rebuilt=rebuilt, optimized=optimized) + return 0 + + +def role_repair(a: dict, out: Out) -> int: + """`repair_state_db_schema` as the CLI/doctor/startup recovery calls it.""" + from hermes_state_repair import repair_state_db_schema + + db_path = Path(a["db"]) + try: + before = _raw_count(db_path) + result = repair_state_db_schema(db_path, backup=bool(a.get("backup", True))) + after = _raw_count(db_path) + except BaseException as exc: + out.report(event="error", error=repr(exc), tb=traceback.format_exc()[-3000:]) + return 2 + out.report(event="stats", before=before, after=after, + result={k: (str(v) if v is not None else None) for k, v in result.items()}) + return 0 + + +def role_agent(a: dict, out: Out) -> int: + """A real AIAgent turn loop on the shared state.db, the LLM behind the loopback fake provider. + + Every turn is ``user(Q) -> tool call (terminal echo T) -> answer(A)``; the journal + records ``I `` before ``run_conversation`` and ``A `` once it returned (the turn is durable). + With ``compress_at`` the agent runs ``/compress here `` after that turn, exactly as the CLI/TUI + slash command does, and journals ``C ``. ``resume`` reloads history from state.db first + (a fresh process resuming the session).""" + from agent.conversation_compression_manual import compress_now, parse_compress_args + from hermes_state import SessionDB + from run_agent import AIAgent + + stop_file = Path(a["stop"]) + db = SessionDB(db_path=Path(a["db"])) + agent = AIAgent(base_url=a["base_url"], api_key="sk-fake-e2e", model="fake-model", quiet_mode=True, + session_db=db, session_id=a["session_id"], skip_context_files=True, skip_memory=True) + history = db.get_messages_as_conversation(a["session_id"]) if a.get("resume") else None + out.report(event="ready", micro=bool(getattr(agent.context_compressor, "_micro_compact_enabled", False)), + resumed=len(history or [])) + bases: list[str] = [] + try: + for i in range(int(a["turns"])): + if _stopping(stop_file): + break + base = f"TA{a['tag']}{os.getpid()}N{i}" # TA: agent turns; TK: plain writer rows + out.journal(f"I {base} {a['session_id']} 3") + result = agent.run_conversation(f"{base}Q question {i} " + "filler " * 120, + conversation_history=history) + history = result["messages"] + if not any(base + "A" in str(m.get("content") or "") for m in history): + raise AssertionError(f"turn {i} ended without its answer: {result.get('final_response')!r}") + out.journal(f"A {base}") + bases.append(base) + if a.get("compress_at") is not None and i == int(a["compress_at"]): + keep = int(a.get("keep", 2)) + res = compress_now(agent, history, parse_compress_args(f"here {keep}"), + skip_without_window=True) + if res.status != "compressed": + raise AssertionError(f"/compress here {keep} did not compress: {res.status}") + history = res.after_messages + out.journal("C " + " ".join(bases[-keep:])) + except BaseException as exc: + out.report(event="error", error=repr(exc), tb=traceback.format_exc()[-4000:]) + return 2 + out.report(event="stats", turns=len(bases)) + db.close() + out.report(event="closed") + return 0 + + +ROLES = { + "agent": role_agent, + "writer": role_writer, "reader": role_reader, "churn": role_churn, "opener": role_opener, + "fts": role_fts, "repair": role_repair, +} + + +def main() -> int: + role, args = sys.argv[1], json.loads(sys.argv[2]) + signal.signal(signal.SIGTERM, _on_sigterm) + out = Out(Path(args["workdir"]), args["name"]) + return ROLES[role](args, out) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/e2e/core/sqlite/test_torture_chamber.py b/tests/e2e/core/sqlite/test_torture_chamber.py new file mode 100644 index 0000000000..f27d92fbc3 --- /dev/null +++ b/tests/e2e/core/sqlite/test_torture_chamber.py @@ -0,0 +1,380 @@ +"""SQLite torture chamber: state.db integrity under real multi-process load (issue class C1). + +One WAL ``state.db``; every role is its own OS process running the production ``SessionDB``: +a gateway-like writer, a TUI-like writer, a dashboard-like reader that opens at startup and lives across +episodes, short-lived openers (production ``SessionDB`` and bare ``sqlite3``, plus the real +``hermes sessions list`` / ``sessions stats`` CLI), FTS rebuild/optimize maintenance, and +``repair_state_db_schema``. Each episode injects one fault class — ``kill -9`` mid-write, SIGTERM graceful +close, POSIX lock cancellation by a stray in-process open/close, chmod flips, concurrent FTS rebuilds, +repair against a live and an offline store, FTS corruption, a whole-fleet SIGKILL — and then asserts the +SAME invariants: + +* ``PRAGMA integrity_check`` is ``ok`` and the store is still in WAL mode; +* no child ever held a ``(deleted)`` ``state.db``/``-wal``/``-shm`` descriptor (``/proc//fd`` scan); +* every acknowledged append is stored exactly once (per-writer intent/ack journals), an in-flight append at + most once, and no row exists that no writer intended; +* canonical row counts only grow (no compaction runs here), and repair never lowers them; +* FTS mirrors the canonical rows (docsize == source rows, FTS5 ``integrity-check``) and session search + finds acked messages exactly once; +* no role hit an error; the long-lived reader's and the open/close churner's fd counts stay bounded. + +Randomness (ack thresholds, kill points) is seeded per episode; the seed is in every failure message and +``HERMES_SQLITE_TORTURE_SEED`` replays a run. +""" + +from __future__ import annotations + +import os +import random +import sqlite3 +import sys +import time + +import pytest + +from tests.conformance.persistence._harness import wait_for +from tests.e2e.core.sqlite._helpers import ( + SEED_ENV, + Chamber, + acked_tokens, + base_seed, + counts, + episode_seed, + exactly_once_problems, + fts_problems, + integrity_rows, + journal_mode, + sample, +) + +pytestmark = [ + pytest.mark.skipif(not sys.platform.startswith("linux"), reason="/proc fd scan and POSIX lock semantics"), +] + +READER_FD_SLACK = 6 +CHURN_FD_SLACK = 2 + + +def _sqlite_wal_capable() -> bool: + # Mirror of the requires_wal gate: SQLite 3.7.0-3.51.2 (minus backports) has the WAL-reset bug, and + # Hermes deliberately falls back to DELETE there, so a WAL chamber is not deployable on that runtime. + v = sqlite3.sqlite_version_info + return v >= (3, 51, 3) or v in ((3, 50, 7), (3, 44, 6)) + + +@pytest.fixture(scope="module") +def chamber(tmp_path_factory): + if not _sqlite_wal_capable(): + pytest.skip(f"linked SQLite {sqlite3.sqlite_version} runs Hermes in DELETE mode; chamber needs WAL") + ch = Chamber(tmp_path_factory.mktemp("chamber")) + _ensure_reader(ch) + yield ch + ch.shutdown() + + +def _ensure_reader(ch: Chamber) -> None: + if ch.reader_name and ch.procs[ch.reader_name].poll() is None: + return + ch.reader_seq += 1 + ch.reader_name = f"reader{ch.reader_seq}" + ch.spawn("reader", ch.reader_name, pace=0.02) + ch.wait_event(ch.reader_name, "ready") + + +def _stop_reader(ch: Chamber) -> None: + if ch.reader_name and ch.procs[ch.reader_name].poll() is None: + assert ch.stop(ch.reader_name) == 0, ch.stderr(ch.reader_name) + + +def _writers(ch: Chamber, ep: str, **kw) -> tuple[str, str]: + gw, tui = f"{ep}-gw", f"{ep}-tui" + ch.spawn("writer", gw, tag="g", sessions=["gw-a", "gw-b", "gw-c"], source="telegram", batch_every=7, **kw) + ch.spawn("writer", tui, tag="t", sessions=["tui-a"], source="tui", batch_every=5, **kw) + ch.wait_event(gw, "ready") + ch.wait_event(tui, "ready") + return gw, tui + + +def _background(ch: Chamber, ep: str, *, openers: int = 4, churn: int = 40) -> list[str]: + names = [] + if churn: + ch.spawn("churn", f"{ep}-churn", iterations=churn) + names.append(f"{ep}-churn") + for i in range(openers): + name = f"{ep}-open{i}" + ch.spawn("opener", name, raw=bool(i % 2)) + names.append(name) + return names + + +def _reap_all(ch: Chamber, names: list[str]) -> None: + for name in names: + proc = ch.procs[name] + if proc.returncode is None: + rc = ch.reap(name) + errors = [e.get("error") for e in ch.events(name) if e.get("event") == "error"] + assert rc == 0, f"{name} exited {rc}: {errors}\n{ch.stderr(name)}" + + +def _stop_all(ch: Chamber, names: list[str]) -> None: + for name in names: + if ch.procs[name].poll() is None: + ch.request_stop(name) + _reap_all(ch, names) + + +# -- episodes -------------------------------------------------------------------------------------------- + + +def ep_baseline(ch, ep, rng): + gw, tui = _writers(ch, ep) + bg = _background(ch, ep) + ch.spawn_cli(f"{ep}-cli-list", "sessions", "list") + ch.spawn_cli(f"{ep}-cli-stats", "sessions", "stats") + ch.wait_acks(gw, rng.randint(60, 120)) + ch.wait_acks(tui, rng.randint(40, 80)) + _reap_all(ch, bg + [f"{ep}-cli-list", f"{ep}-cli-stats"]) + _stop_all(ch, [gw, tui]) + return [gw, tui] + + +def ep_kill9_writer_midwrite(ch, ep, rng): + gw, tui = _writers(ch, ep) + ch.spawn("churn", f"{ep}-churn", iterations=400) + ch.wait_acks(gw, rng.randint(20, 60)) + ch.kill9(gw) # gateway dies mid-append: the in-flight row is either whole or absent + ch.kill9(f"{ep}-churn") # and a CLI dies holding the file open + gw2 = f"{ep}-gw2" + ch.spawn("writer", gw2, tag="g", sessions=["gw-a", "gw-b"], source="telegram", batch_every=3) + bg = _background(ch, ep, churn=0) + ch.wait_acks(gw2, rng.randint(30, 60)) + ch.wait_acks(tui, 40) + _reap_all(ch, bg) + _stop_all(ch, [gw2, tui]) + return [gw, tui, gw2] + + +def ep_sigterm_graceful_close(ch, ep, rng): + gw, tui = _writers(ch, ep) + bg = _background(ch, ep, openers=6, churn=60) + ch.wait_acks(gw, rng.randint(30, 80)) + ch.sigterm(gw) + ch.sigterm(tui) # both close (checkpoint-on-close) while the churner and openers are mid-cycle + for name in (gw, tui): + assert ch.reap(name) == 0, ch.stderr(name) + assert any(e.get("event") == "closed" for e in ch.events(name)), f"{name} never closed cleanly" + gw2 = f"{ep}-gw2" + ch.spawn("writer", gw2, tag="g", sessions=["gw-a"], source="telegram") + ch.wait_acks(gw2, 30) + _reap_all(ch, bg) + _stop_all(ch, [gw2]) + return [gw, tui, gw2] + + +def ep_lock_cancellation(ch, ep, rng): + """Plugins, tool reads of ~/.hermes and header probes open+close state.db inside the writer process, + which cancels its POSIX locks; a sibling's last close must still not end the writer's WAL generation. + Field topology: the gateway (+ a TUI) are the only long-lived holders — no dashboard reader pinning the + file — so a short-lived CLI's close is the last close SQLite can see.""" + _stop_reader(ch) + gw, tui = _writers(ch, ep, stray_every=rng.randint(2, 4)) + names = [] + target_gw, target_tui = rng.randint(80, 140), rng.randint(60, 100) + wave = 0 + while len(ch.journal(gw)[1]) < target_gw or len(ch.journal(tui)[1]) < target_tui: + batch = _background(ch, f"{ep}-w{wave}", openers=4, churn=6) + _reap_all(ch, batch) + names += batch + wave += 1 + assert wave < 200, "writers made no progress" + for w in (gw, tui): + rc = ch.procs[w].poll() + assert rc is None, (f"{w} died (rc={rc}) after {len(ch.journal(w)[1])} acks: {ch.events(w)[-1:]}\n" + f"deleted sidecars held: {ch.deleted_hits_snapshot()}\n{ch.stderr(w)}") + _stop_all(ch, [gw, tui]) + _ensure_reader(ch) + return [gw, tui] + + +def ep_chmod_flip(ch, ep, rng): + gw, tui = _writers(ch, ep) + files = [ch.db, ch.db.with_name(ch.db.name + "-wal"), ch.db.with_name(ch.db.name + "-shm")] + try: + for i in range(rng.randint(6, 10)): + for f in files: + if f.exists(): + os.chmod(f, 0o644) # a permissive mode the next SessionDB open tightens + ch.spawn("opener", f"{ep}-tighten{i}") + _reap_all(ch, [f"{ep}-tighten{i}"]) + os.chmod(ch.db, 0o400) # read-only main file: new openers may refuse, must not damage + ch.spawn("opener", f"{ep}-ro{i}", may_fail=True) + ch.spawn("opener", f"{ep}-rorw{i}", may_fail=True, raw=True) + _reap_all(ch, [f"{ep}-ro{i}", f"{ep}-rorw{i}"]) + os.chmod(ch.db, 0o600) + finally: + for f in files: + if f.exists(): + os.chmod(f, 0o600) + n_gw = len(ch.journal(gw)[1]) + ch.wait_acks(gw, n_gw + 20) # still writing after every permission flip + _stop_all(ch, [gw, tui]) + return [gw, tui] + + +def ep_fts_rebuild_concurrent(ch, ep, rng): + gw, tui = _writers(ch, ep) + ch.wait_acks(gw, 20) + ch.spawn("fts", f"{ep}-fts-a") + ch.spawn("fts", f"{ep}-fts-b") + ch.spawn("fts", f"{ep}-fts-killed") + ch.wait_event(f"{ep}-fts-killed", "ready") + time.sleep(rng.uniform(0.0, 0.05)) # fault-injection jitter, not synchronization + ch.kill9(f"{ep}-fts-killed") # maintenance killed mid-rebuild + _reap_all(ch, [f"{ep}-fts-a", f"{ep}-fts-b"]) + rebuilt = [e for n in ("a", "b") for e in ch.events(f"{ep}-fts-{n}") if e.get("event") == "stats"] + assert len(rebuilt) == 2, rebuilt + ch.wait_acks(gw, len(ch.journal(gw)[1]) + 20) + _stop_all(ch, [gw, tui]) + return [gw, tui] + + +def _repair(ch: Chamber, name: str) -> dict: + ch.spawn("repair", name) + _reap_all(ch, [name]) + stats = [e for e in ch.events(name) if e.get("event") == "stats"] + assert stats, ch.events(name) + s = stats[0] + assert s["after"] >= s["before"], f"repair lowered canonical rows: {s}" + return s + + +def ep_repair_live_then_offline(ch, ep, rng): + gw, tui = _writers(ch, ep) + ch.wait_acks(gw, rng.randint(20, 50)) + _repair(ch, f"{ep}-repair-live") # must not cost an acked row (checked by the episode invariants) + ch.wait_acks(gw, len(ch.journal(gw)[1]) + 20) + _stop_all(ch, [gw, tui]) + _stop_reader(ch) + _repair(ch, f"{ep}-repair-offline") + _ensure_reader(ch) + return [gw, tui] + + +def ep_fts_corruption_fail_open(ch, ep, rng): + """A corrupt FTS index must not cost a canonical write; repair + the next open restore search.""" + _stop_reader(ch) + conn = sqlite3.connect(str(ch.db), timeout=30.0) + try: + conn.execute("UPDATE messages_fts_data SET block = X'DEADBEEFDEADBEEFDEADBEEFDEADBEEF'") + conn.commit() + finally: + conn.close() + gw, tui = _writers(ch, ep) + ch.wait_acks(gw, rng.randint(20, 50)) + ch.wait_acks(tui, 20) + _repair(ch, f"{ep}-repair-live") # a repair racing live writers must not overwrite their acked rows + _stop_all(ch, [gw, tui]) + _repair(ch, f"{ep}-repair-offline") + ch.spawn("opener", f"{ep}-reopen") # the next SessionDB open finishes the stale-FTS recovery + _reap_all(ch, [f"{ep}-reopen"]) + _ensure_reader(ch) + return [gw, tui] + + +def ep_kill9_everything(ch, ep, rng): + """Update fleet-restart / OOM-killer shape: every process on the file dies at once, mid-write.""" + gw, tui = _writers(ch, ep, stray_every=5) + ch.spawn("churn", f"{ep}-churn", iterations=400) + ch.spawn("fts", f"{ep}-fts") + ch.wait_acks(gw, rng.randint(30, 90)) + for name in (gw, tui, f"{ep}-churn", f"{ep}-fts", ch.reader_name): + if ch.procs[name].poll() is None: + ch.kill9(name) + gw2 = f"{ep}-gw2" + ch.spawn("writer", gw2, tag="g", sessions=["gw-a", "tui-a"], source="telegram", batch_every=4) + _ensure_reader(ch) + ch.wait_acks(gw2, 30) + _stop_all(ch, [gw2]) + return [gw, tui, gw2] + + +EPISODES = { + "baseline": ep_baseline, + "kill9_writer_midwrite": ep_kill9_writer_midwrite, + "sigterm_graceful_close": ep_sigterm_graceful_close, + "lock_cancellation": ep_lock_cancellation, + "chmod_flip": ep_chmod_flip, + "fts_rebuild_concurrent": ep_fts_rebuild_concurrent, + "repair_live_then_offline": ep_repair_live_then_offline, + "fts_corruption_fail_open": ep_fts_corruption_fail_open, + "kill9_everything": ep_kill9_everything, +} + + +def _reader_fd_problems(ch: Chamber) -> list[str]: + fds = [e["fds"] for e in ch.events(ch.reader_name) if e.get("fds") is not None] + if len(fds) < 4: + return [] + warm = max(fds[:3]) # after the first passes the read pool has settled + if max(fds) > warm + READER_FD_SLACK: + return [f"{ch.reader_name} fd count grew {warm} -> {max(fds)} over {len(fds)} samples ({fds[-5:]})"] + return [] + + +def _churn_fd_problems(ch: Chamber, ep: str) -> list[str]: + out = [] + for name in ch.procs: + if not name.startswith(ep) or "churn" not in name: + continue + for e in ch.events(name): + if e.get("event") == "stats" and e.get("warm_fds") is not None: + if e["end_fds"] > e["warm_fds"] + CHURN_FD_SLACK: + out.append(f"{name}: fds {e['warm_fds']} after warmup -> {e['end_fds']} after " + f"{e['iterations']} open/close cycles") + return out + + +@pytest.mark.parametrize("episode", list(EPISODES)) +def test_torture_episode(chamber, episode): + seed = episode_seed(episode) + rng = random.Random(seed) + ctx = f"[episode={episode} seed={seed} base={base_seed()}; replay: {SEED_ENV}={base_seed()}]" + _ensure_reader(chamber) + before = counts(chamber.db) if chamber.db.exists() else {"__total__": 0} + started = time.monotonic() + + runs = EPISODES[episode](chamber, episode, rng) + + # Nothing but the long-lived reader may still be running on the file. + stragglers = [n for n, _p in chamber.live() if n != chamber.reader_name] + assert not stragglers, f"{ctx} roles still running: {stragglers}" + problems: list[str] = [] + problems += [f"{n}: {e.get('error')}\n{e.get('tb', '')}" for n, e in chamber.errors()] + problems += [f"{name} (pid {pid}) held {link}" for name, pid, link in chamber.deleted_hits_snapshot()] + rows = integrity_rows(chamber.db) + if rows != ["ok"]: + problems.append(f"integrity_check: {rows[:5]}") + mode = journal_mode(chamber.db) + if mode != "wal": + problems.append(f"journal_mode is {mode!r} after the episode (was wal)") + problems += exactly_once_problems(chamber) + after = counts(chamber.db) + for sid, n in before.items(): + if after.get(sid, 0) < n: + problems.append(f"canonical rows went down for {sid}: {n} -> {after.get(sid, 0)}") + episode_acks = acked_tokens(chamber, runs) + if not episode_acks: + problems.append("episode acknowledged no appends") + problems += fts_problems(chamber.db, sample(rng, episode_acks, 12)) + problems += _reader_fd_problems(chamber) + problems += _churn_fd_problems(chamber, episode) + assert not problems, f"{ctx} ({time.monotonic() - started:.1f}s)\n" + "\n".join(problems) + + +def test_long_lived_reader_saw_every_episode_grow_only(chamber): + """The dashboard reader lives across episodes: it never saw a session shrink and never errored.""" + wait_for(lambda: any(e.get("event") == "stats" for e in chamber.events(chamber.reader_name)), + what="reader stats") + errors = [(n, e) for n, e in chamber.errors() if n.startswith("reader")] + assert not errors, errors + assert not _reader_fd_problems(chamber) From 485e25ce68af8cc210d30cb9e4cde992e6c93992 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:42:32 -0700 Subject: [PATCH 005/104] test: compaction persistence under multi-process contention (C1) Class C1, "state.db corrupting from compaction": real AIAgent processes run real turns (user -> terminal tool call -> answer) through the loopback fake provider on one shared WAL state.db while a plain writer, a reader and an open/close churner work the same file. A gateway-like agent micro-compacts every turn; a TUI-like agent runs /compress here 2; faults are clean, kill -9 mid-turn and kill -9 timed into the micro-compaction summary/commit window; a fresh process then resumes the session. Invariants: every acked user turn live exactly once, its tool result and answer recoverable (live or compacted history) and never live twice, the /compress-kept exchanges live exactly once, canonical rows never shrink (external sampler + reader), integrity ok, FTS == canonical, resumed request carries every acked user turn exactly once, plain-writer appends exactly once. Red-proof: reverting 91df54184d3 (micro-compaction tool rows) turns every absorbed tool result/answer into active=0,compacted=0 debris; reverting 526d135a96a (/compress here tail) leaves the kept exchanges with no live row. --- .../core/sqlite/test_compaction_contention.py | 288 ++++++++++++++++++ 1 file changed, 288 insertions(+) create mode 100644 tests/e2e/core/sqlite/test_compaction_contention.py diff --git a/tests/e2e/core/sqlite/test_compaction_contention.py b/tests/e2e/core/sqlite/test_compaction_contention.py new file mode 100644 index 0000000000..09fd847ffb --- /dev/null +++ b/tests/e2e/core/sqlite/test_compaction_contention.py @@ -0,0 +1,288 @@ +"""Compaction persistence under multi-process contention (issue class C1, "state.db corrupting from +compaction"). + +Real ``AIAgent`` processes drive real turns (user -> terminal tool call -> answer) through the loopback fake +provider on ONE shared WAL ``state.db``, while other processes write to, read and open/close the same file: + +* a gateway-like agent with rolling micro-compaction on (every turn folds the oldest exchange into a summary + and commits it through ``archive_and_compact``); +* a TUI-like agent that runs ``/compress here 2`` mid-session (in-place manual compaction, kept tail); +* a plain gateway writer on another session, a dashboard reader, and a CLI-style open/close churner. + +Fault matrix: clean, ``kill -9`` of the micro-compacting agent mid-turn, and ``kill -9`` timed into its +compaction (the summary stream / commit window). A fresh process then resumes the session and runs more +turns. Invariants after every episode: + +* no acknowledged turn is lost: its user message is in the live transcript exactly once (user turns are + never absorbed by compaction), its tool result and answer are still recoverable — live (``active=1``) or + as compacted history (``compacted=1``) — and never live twice; +* the exchanges ``/compress here N`` promised to keep are live exactly once; +* canonical row counts only grow (compaction archives, it never deletes) — sampled continuously from outside + and by the reader; ``integrity_check`` ok; FTS mirrors canonical rows; no ``(deleted)`` WAL held; +* the resumed process sends the model every acknowledged user turn exactly once (persisted == sent); +* the plain writer's acked appends are stored exactly once. +""" + +from __future__ import annotations + +import json +import random +import re +import sqlite3 +import sys +import threading +import time +from dataclasses import dataclass + +import pytest + +from tests.e2e.core.sqlite._helpers import ( + SEED_ENV, + Chamber, + base_seed, + compress_journal, + counts, + episode_seed, + exactly_once_problems, + fts_problems, + integrity_rows, + token_flags, +) +from tests.fakes.fake_llm_provider import FakeLLMServer, Text, ToolCall, write_hermes_home + +pytestmark = [ + pytest.mark.skipif(not sys.platform.startswith("linux"), reason="/proc fd scan and POSIX lock semantics"), +] + +Q_TOKEN = re.compile(r"TA[a-z]+\d+N\d+Q") +MICRO_CONFIG = "compression:\n micro_compact: true\n target_ratio: 0.1\n protect_last_n: 4\n" +DB_CONFIG = "database:\n journal_mode: wal\n" +ANSWER_FILLER = "lorem " * 1200 # big enough that the compressible window is non-empty at 64K context + + +def _text(content) -> str: + if isinstance(content, list): + return " ".join(str(p.get("text", "")) for p in content if isinstance(p, dict)) + return str(content or "") + + +def _responder(record: dict): + """Main turns: a user turn gets a terminal tool call, the tool result gets the final answer.""" + messages = record["body"]["messages"] + last_user = next(_text(m.get("content")) for m in reversed(messages) if m.get("role") == "user") + base = Q_TOKEN.findall(last_user)[-1][:-1] + if messages[-1].get("role") == "tool": + return Text(f"{base}A answer. {ANSWER_FILLER}", chunk_chars=2000) + return ToolCall("terminal", {"command": f"echo {base}T"}) + + +class _Aux: + """Summaries for compaction; flags every micro-compaction summary request of the gateway agent.""" + + def __init__(self): + self.summaries = 0 + self._cond = threading.Condition() + + def wait_beyond(self, seen: int, still_running) -> bool: + """Block until a summary request newer than ``seen`` arrives (False if the agent exited first).""" + with self._cond: + while self.summaries <= seen: + if not still_running(): + return False + self._cond.wait(0.05) + return True + + def __call__(self, record: dict): + body = json.dumps(record["body"]) + if re.search(r"TAg\d+N\d+T", body): # the exchange text carries the tool result -> a summary call + with self._cond: + self.summaries += 1 + self._cond.notify_all() + return Text("Rolling summary: the user asked numbered questions and each was answered.", + chunk_chars=6, delay_per_chunk=0.01) + + +@dataclass +class Rig: + ch: Chamber + server: FakeLLMServer + aux: _Aux + tui_env: dict + + +@pytest.fixture(scope="module") +def rig(tmp_path_factory): + v = sqlite3.sqlite_version_info + if not (v >= (3, 51, 3) or v in ((3, 50, 7), (3, 44, 6))): + pytest.skip(f"linked SQLite {sqlite3.sqlite_version} runs Hermes in DELETE mode") + aux = _Aux() + server = FakeLLMServer(_responder, aux=aux) + server.start() + ch = Chamber(tmp_path_factory.mktemp("compaction")) + ctx = " context_length: 128000\n" + write_hermes_home(ch.hermes_home, server.base_url, extra_config=MICRO_CONFIG + DB_CONFIG) + tui_home = ch.root / "tui-home" + write_hermes_home(tui_home / ".hermes", server.base_url, extra_config=DB_CONFIG) + for home in (ch.hermes_home, tui_home / ".hermes"): + cfg = home / "config.yaml" + cfg.write_text(cfg.read_text(encoding="utf-8").replace(ctx, " context_length: 64000\n"), encoding="utf-8") + yield Rig(ch, server, aux, {"HOME": str(tui_home), "HERMES_HOME": str(tui_home / ".hermes")}) + ch.shutdown() + server.stop() + + +class _GrowOnly(threading.Thread): + """Samples the canonical row count from outside every ~50 ms; compaction must never shrink it.""" + + def __init__(self, ch: Chamber): + super().__init__(daemon=True) + self.ch, self.stop, self.violations, self.samples = ch, threading.Event(), [], 0 + + def run(self): + last = 0 + while not self.stop.is_set(): + try: + n = counts(self.ch.db)["__total__"] + except sqlite3.OperationalError: + n = last # busy under a checkpoint: skip the sample, never guess + if n < last: + self.violations.append(f"canonical rows shrank {last} -> {n}") + last = max(last, n) + self.samples += 1 + self.stop.wait(0.05) + + +def _agent(rig: Rig, name, *, sid, tag, turns, env=None, **kw): + rig.ch.spawn("agent", name, env=env, base_url=rig.server.base_url, session_id=sid, tag=tag, turns=turns, **kw) + rig.ch.wait_event(name, "ready", deadline=90) + + +def _turn_problems(ch: Chamber, run: str, *, compacting: bool) -> list[str]: + intents, acked = ch.journal(run) + problems = [] + for base in sorted(acked): + q, t, a = token_flags(ch.db, base + "Q"), token_flags(ch.db, base + "T"), token_flags(ch.db, base + "A") + live_q = sum(1 for f in q if f[0] == 1) + if live_q != 1: + problems.append(f"{run}: acked user turn {base}Q is live {live_q}x (rows {q})") + for label, flags in (("tool result", t), ("answer", a)): + live = sum(1 for f in flags if f[0] == 1) + kept = sum(1 for f in flags if f[0] == 1 or f[1] == 1) + if kept < 1: + problems.append(f"{run}: acked {label} of {base} is unrecoverable: every row is " + f"archived-as-superseded (active=0, compacted=0): {flags}") + if live > 1: + problems.append(f"{run}: acked {label} of {base} is live {live}x: {flags}") + if not compacting and live != 1: + problems.append(f"{run}: {label} of {base} not live without any compaction: {flags}") + for base in sorted(set(intents) - acked): # the turn a SIGKILL interrupted + live_q = sum(1 for f in token_flags(ch.db, base + "Q") if f[0] == 1) + if live_q > 1: + problems.append(f"{run}: in-flight user turn {base}Q is live {live_q}x") + for kept in compress_journal(ch, run): + for base in kept: + for suffix in "QTA": + flags = token_flags(ch.db, base + suffix) + if sum(1 for f in flags if f[0] == 1) != 1: + problems.append(f"{run}: /compress here kept {base}{suffix} but it is live " + f"{sum(1 for f in flags if f[0] == 1)}x (rows {flags})") + return problems + + +def _resume_problems(rig: Rig, sid: str, runs: list[str], first_request: int) -> list[str]: + """The fresh process must send the model every acked user turn of the session exactly once.""" + ch = rig.ch + requests = rig.server.main_requests()[first_request:] + if not requests: + return [f"resume of {sid} sent no request"] + sent = " ".join(_text(m.get("content")) for m in requests[-1]["messages"]) + problems = [] + for run in runs: + for base in sorted(ch.journal(run)[1]): + n = sent.count(base + "Q") + if n != 1: + problems.append(f"resumed request carries acked user turn {base}Q {n}x (persisted != sent)") + return problems + + +def _compacted_rows(ch: Chamber, sid: str) -> int: + conn = sqlite3.connect(f"file:{ch.db}?mode=ro", uri=True, timeout=30.0) + try: + return conn.execute("SELECT count(*) FROM messages WHERE session_id = ? AND compacted = 1", + (sid,)).fetchone()[0] + finally: + conn.close() + + +def _reap_ok(ch: Chamber, name: str, deadline: float = 120.0) -> None: + rc = ch.reap(name, deadline=deadline) + errors = [e.get("error") for e in ch.events(name) if e.get("event") == "error"] + assert rc == 0, f"{name} exited {rc}: {errors}\n{ch.stderr(name)}" + + +@pytest.mark.parametrize("fault", ["clean", "kill9_agent_mid_turn", "kill9_during_micro_compaction"]) +def test_compaction_episode(rig, fault): + ch = rig.ch + seed = episode_seed(f"compaction:{fault}") + rng = random.Random(seed) + ctx = f"[fault={fault} seed={seed} base={base_seed()}; replay: {SEED_ENV}={base_seed()}]" + gw_sid, tui_sid = f"{fault}-gw", f"{fault}-tui" + gw, tui, plain, reader = f"{fault}-agent-gw", f"{fault}-agent-tui", f"{fault}-plain", f"{fault}-reader" + sampler = _GrowOnly(ch) + sampler.start() + try: + _agent(rig, gw, sid=gw_sid, tag="g", turns=5) + _agent(rig, tui, sid=tui_sid, tag="t", turns=5, compress_at=3, keep=2, env=rig.tui_env) + micro = [e for e in ch.events(gw) if e.get("event") == "ready"][0]["micro"] + assert micro, f"{ctx} gateway agent did not load micro_compact from its config" + ch.spawn("writer", plain, tag="p", sessions=[f"{fault}-plain"], source="telegram", pace=0.01) + ch.spawn("reader", reader, pace=0.05) + ch.spawn("churn", f"{fault}-churn", iterations=30) + + if fault == "kill9_agent_mid_turn": + ch.wait_acks(gw, rng.randint(2, 4), deadline=120) + time.sleep(rng.uniform(0.0, 0.8)) # fault-injection jitter inside the next turn, not a sync + ch.kill9(gw) + elif fault == "kill9_during_micro_compaction": + ch.wait_acks(gw, rng.randint(2, 3), deadline=120) + # Turn k's summary precedes ack k, so the next one belongs to a later turn's finalize. + if rig.aux.wait_beyond(rig.aux.summaries, lambda: ch.procs[gw].poll() is None): + time.sleep(rng.uniform(0.0, 0.3)) # inside the summary stream or its archive_and_compact commit + ch.kill9(gw) + else: + _reap_ok(ch, gw) + _reap_ok(ch, tui) + _reap_ok(ch, f"{fault}-churn") + ch.request_stop(plain) + _reap_ok(ch, plain) + + # A fresh process resumes the gateway session and keeps going (micro-compaction still on). + first = len(rig.server.main_requests()) + resumed = f"{fault}-agent-gw-resume" + _agent(rig, resumed, sid=gw_sid, tag="g", turns=1, resume=True) + _reap_ok(ch, resumed) + ch.request_stop(reader) + _reap_ok(ch, reader) + finally: + sampler.stop.set() + sampler.join(timeout=10) + + problems: list[str] = [] + problems += [f"{n}: {e.get('error')}\n{e.get('tb', '')}" for n, e in ch.errors() if n.startswith(fault)] + problems += [f"{n} (pid {pid}) held {link}" for n, pid, link in ch.deleted_hits_snapshot()] + rows = integrity_rows(ch.db) + if rows != ["ok"]: + problems.append(f"integrity_check: {rows[:5]}") + problems += sampler.violations + problems += _turn_problems(ch, gw, compacting=True) + problems += _turn_problems(ch, resumed, compacting=True) + problems += _turn_problems(ch, tui, compacting=True) + problems += _resume_problems(rig, gw_sid, [gw], first) + problems += exactly_once_problems(ch) + problems += fts_problems(ch.db, []) + if not compress_journal(ch, tui): + problems.append("the TUI agent never ran /compress here") + for sid in (gw_sid, tui_sid): # vacuity guard: a compaction commit really landed in this episode + if not _compacted_rows(ch, sid): + problems.append(f"no compacted history rows in {sid}: compaction never committed") + assert not problems, f"{ctx} ({sampler.samples} row-count samples)\n" + "\n".join(problems) From 7b5b7135170804b6f4caa26a66a2171c5b22e7f7 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 02:50:29 -0700 Subject: [PATCH 006/104] test(fakes): optional request-derived usage and Text.finish_reason in the fake provider Backward compatible: FakeLLMServer(prompt_tokens_fn=...) reports usage.prompt_tokens computed from each request body (token-driven logic such as compaction triggers sees a realistic, growing count instead of a constant 100), and Text(finish_reason=...) lets a scripted reply end with e.g. 'length' (truncated summaries). Defaults are unchanged. --- tests/fakes/fake_llm_provider.py | 26 ++++++++++++++++++-------- 1 file changed, 18 insertions(+), 8 deletions(-) diff --git a/tests/fakes/fake_llm_provider.py b/tests/fakes/fake_llm_provider.py index b99f56e18e..4cafadab3b 100644 --- a/tests/fakes/fake_llm_provider.py +++ b/tests/fakes/fake_llm_provider.py @@ -49,6 +49,7 @@ class Text: prompt_tokens: int = 100 completion_tokens: int = 20 cached_tokens: int = 0 + finish_reason: str = "stop" @dataclass @@ -126,12 +127,16 @@ class FakeLLMServer: default_text: str = "ok", aux: Responder | None = None, api_key: str | None = None, + prompt_tokens_fn: Callable[[dict[str, Any]], int] | None = None, ) -> None: self._script: list[Response] = list(script) if isinstance(script, list) else [] self._responder: Responder | None = script if callable(script) else None self.default_text = default_text self._aux = aux or (lambda _req: Text("Fake summary of the earlier conversation.")) self.expected_api_key = api_key + # Optional: derive reported ``usage.prompt_tokens`` from each request body (so token-driven + # logic such as compaction triggers sees a realistic, growing count instead of a constant). + self.prompt_tokens_fn = prompt_tokens_fn self.requests: list[dict[str, Any]] = [] self._lock = threading.Lock() self._stop = threading.Event() @@ -255,10 +260,11 @@ def _handler_for(server: FakeLLMServer) -> type[BaseHTTPRequestHandler]: return resp = server._next_main(record) if kind == "main" else server._aux(record) record["response"] = type(resp).__name__ - self._respond(resp, bool(body.get("stream"))) + prompt_tokens = server.prompt_tokens_fn(body) if server.prompt_tokens_fn else None + self._respond(resp, bool(body.get("stream")), prompt_tokens) # response rendering - def _respond(self, resp: Response, stream: bool) -> None: + def _respond(self, resp: Response, stream: bool, prompt_tokens: int | None = None) -> None: if isinstance(resp, Error): headers = {"Retry-After": str(resp.retry_after)} if resp.retry_after is not None else {} self._send_json(resp.status, {"error": {"message": resp.message, "type": "server_error"}}, headers) @@ -288,7 +294,7 @@ def _handler_for(server: FakeLLMServer) -> type[BaseHTTPRequestHandler]: self.wfile.flush() self.close_connection = True return - message, finish, usage = _message_for(resp, server) + message, finish, usage = _message_for(resp, server, prompt_tokens) if not stream: self._send_json(200, { "id": "chatcmpl-fake", "object": "chat.completion", "created": int(time.time()), @@ -351,18 +357,21 @@ def _chunk(delta: dict[str, Any], finish: str | None = None) -> dict[str, Any]: } -def _message_for(resp: Text | ToolCall, server: FakeLLMServer) -> tuple[dict[str, Any], str, dict[str, Any]]: +def _message_for( + resp: Text | ToolCall, server: FakeLLMServer, prompt_tokens: int | None = None, +) -> tuple[dict[str, Any], str, dict[str, Any]]: if isinstance(resp, Text): message: dict[str, Any] = {"role": "assistant", "content": resp.text} if resp.reasoning: message["reasoning_content"] = resp.reasoning + pt = resp.prompt_tokens if prompt_tokens is None else prompt_tokens usage = { - "prompt_tokens": resp.prompt_tokens, + "prompt_tokens": pt, "completion_tokens": resp.completion_tokens, - "total_tokens": resp.prompt_tokens + resp.completion_tokens, + "total_tokens": pt + resp.completion_tokens, "prompt_tokens_details": {"cached_tokens": resp.cached_tokens}, } - return message, "stop", usage + return message, resp.finish_reason, usage calls = [(resp.name, resp.args), *resp.parallel] tool_calls = [ {"id": server.next_tool_call_id(), "type": "function", @@ -370,7 +379,8 @@ def _message_for(resp: Text | ToolCall, server: FakeLLMServer) -> tuple[dict[str for name, args in calls ] message = {"role": "assistant", "content": resp.text, "tool_calls": tool_calls} - usage = {"prompt_tokens": 100, "completion_tokens": 10, "total_tokens": 110} + pt = 100 if prompt_tokens is None else prompt_tokens + usage = {"prompt_tokens": pt, "completion_tokens": 10, "total_tokens": pt + 10} return message, "tool_calls", usage From b2e139b1573b365947e8d101960404d008171b3c Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:01:15 -0700 Subject: [PATCH 007/104] test: compaction property + liveness E2E on seeded random sessions (C4) Class C4 (context compression correctness/liveness). A real AIAgent + SessionDB on disk drives seeded random sessions (tool-pair heavy incl. parallel calls, huge single turns, image parts, long answers, single tool results larger than the whole threshold, a cron-shaped single job turn with 14 tool rounds, a provider whose real window is smaller than the configured one) against the recording fake provider. After every turn and for every request it asserts: bounded wall time and summarizer calls, all scripted steps consumed or an explicit error, no user message lost from state.db (live or compacted=1), superseded rows always have a live/ archived copy, tool_call/tool_result pairs matched in live rows and in every request, system prompt stable, the in-flight ask (cron job prompt) after every summary (#100818), and persisted == sent: state.db at request time is exactly the request's history prefix, live rows equal the in-memory history, and a fresh agent resumed from state.db continues identically. Exposed the stale current_turn_user_idx bug fixed in the previous commit. --- tests/e2e/core/compaction/__init__.py | 0 tests/e2e/core/compaction/_helpers.py | 648 ++++++++++++++++++ tests/e2e/core/compaction/conftest.py | 57 ++ .../core/compaction/test_compaction_auto.py | 86 +++ 4 files changed, 791 insertions(+) create mode 100644 tests/e2e/core/compaction/__init__.py create mode 100644 tests/e2e/core/compaction/_helpers.py create mode 100644 tests/e2e/core/compaction/conftest.py create mode 100644 tests/e2e/core/compaction/test_compaction_auto.py diff --git a/tests/e2e/core/compaction/__init__.py b/tests/e2e/core/compaction/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/tests/e2e/core/compaction/_helpers.py b/tests/e2e/core/compaction/_helpers.py new file mode 100644 index 0000000000..e6b3326eb3 --- /dev/null +++ b/tests/e2e/core/compaction/_helpers.py @@ -0,0 +1,648 @@ +"""Lane-private harness for the compaction property / liveness suites (class C4). + +Everything real except the LLM: a real ``AIAgent`` with a real ``SessionDB`` on disk, real tools +(``read_file`` on seeded files), real compaction, all talking to the recording loopback provider in +``tests/fakes/fake_llm_provider.py``. The provider is shared per module; each scenario installs its own +main-turn and summarizer responders on the ``Dispatch`` object. + +The invariants live here so every test file asserts the SAME properties: + +* ``assert_wire_request_ok`` — every request the model sees: system first, first user (job/task + anchor) present, tool_call/tool_result pairs matched, no user->user. +* ``assert_persisted_prefix`` — what state.db holds at request time is exactly the history prefix + the request carries; the rest is only the in-flight turn (persisted == sent). +* ``assert_db_recoverable`` — no user message is lost: each is live (``active=1``) or archived + as summarized history (``compacted=1``); a superseded duplicate (``active=0, compacted=0``) always + has an identical live/archived copy. +* ``assert_db_matches_history`` — the live rows equal the in-memory history the host keeps. +""" + +from __future__ import annotations + +import base64 +import json +import random +import re +import sqlite3 +import threading +import time +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Callable, Iterable + +from tests.fakes.fake_llm_provider import MODEL_ID, Error, FakeLLMServer, Hang, Text, ToolCall + +# Every compaction knob is tiny so a handful of turns crosses the trigger. The window must stay at the +# 64K agent floor; ``threshold_tokens`` (an absolute cap) pulls the trigger far below the ratio floor. +CONTEXT_LENGTH = 64_000 +THRESHOLD_TOKENS = 12_000 +# Seconds the "slow" summarizer hangs; the summary timeout is pinned far below it. +SLOW_HANG_SECONDS = 60.0 +SLOW_SUMMARY_TIMEOUT_SECONDS = 3.0 +# Seeded session shape shared by the automatic suites. +SEEDS = (11, 23, 37) +N_TURNS = 10 +# Legitimate summarizer traffic per turn: max_attempts (3) x (primary + one main-model retry) plus the +# stall fallback route. A runaway compaction loop (the 877-call class) blows far past this. +MAX_SUMMARY_CALLS_PER_TURN = 8 +# Upper bound on one turn's wall time. The slowest legitimate turn is a slow-summarizer turn that waits +# one pinned timeout per attempt; 10x headroom over that keeps the bound load-proof. +TURN_WALL_BOUND_SECONDS = 120.0 + +MARKER_RE = re.compile(r"\bU\d+x\d+x[0-9a-f]{6}\b") +GOOD_SUMMARY_TOKEN = "SUMMARY-OK" +# Bodies of the bad summaries; none may ever reach the model or state.db as context. +REFUSAL_TEXT = "I'm sorry, but I can't help with summarizing this conversation." +TRUNCATED_TEXT = "## Goal\nThe user is work" + +_PNG_1PX = base64.b64encode(bytes.fromhex( + "89504e470d0a1a0a0000000d49484452000000010000000108060000001f15c4890000000d49444154" + "78da63f8cfc0f01f0005000201ffa3a8b2b10000000049454e44ae426082")).decode() + + +def estimate_prompt_tokens(body: dict[str, Any]) -> int: + """What the fake provider reports as ``usage.prompt_tokens``: chars/4 of the whole request.""" + return len(json.dumps(body.get("messages", []))) // 4 + len(json.dumps(body.get("tools", []))) // 4 + + +# --------------------------------------------------------------------------------------------------- +# Provider dispatch +# --------------------------------------------------------------------------------------------------- + + +class Dispatch: + """Per-scenario responders behind one module-scoped provider.""" + + def __init__(self) -> None: + self.main: Callable[[dict[str, Any]], Any] = lambda _rec: Text("ok") + self.aux: Callable[[dict[str, Any]], Any] = lambda _rec: Text(good_summary(0)) + + +def start_provider() -> tuple[FakeLLMServer, Dispatch]: + dispatch = Dispatch() + server = FakeLLMServer( + lambda rec: dispatch.main(rec), aux=lambda rec: dispatch.aux(rec), + prompt_tokens_fn=estimate_prompt_tokens) + server.start() + return server, dispatch + + +def good_summary(n: int) -> str: + return ( + f"## Goal\nKeep helping with the seeded task ({GOOD_SUMMARY_TOKEN}-{n}).\n" + "## Progress\n### Done\n- Read several files and answered.\n" + "## Next Steps\n- Continue with the latest request.\n") + + +def summary_responder(mode: str, calls: list) -> Callable[[dict[str, Any]], Any]: + """Summarizer behaviours from the C4 harness spec.""" + + def respond(rec: dict[str, Any]) -> Any: + calls.append(rec) + n = len(calls) + if mode == "good": + return Text(good_summary(n), chunk_chars=64) + if mode == "empty": + return Text("") + if mode == "refusal": + return Text(REFUSAL_TEXT) + if mode == "http500": + return Error(500, "summarizer upstream exploded") + if mode == "truncated": + return Text(TRUNCATED_TEXT, finish_reason="length") + if mode == "slow": + return Hang(SLOW_HANG_SECONDS) + raise AssertionError(f"unknown summarizer mode {mode!r}") + + return respond + + +def compaction_config(*, extra: str = "", slow: bool = False, threshold_tokens: int = THRESHOLD_TOKENS) -> str: + cfg = ( + "compression:\n" + f" threshold_tokens: {threshold_tokens}\n" + " protect_last_n: 4\n" + + extra + + "auxiliary:\n" + " title_generation:\n" + " enabled: false\n" + ) + if slow: + cfg += f" compression:\n timeout: {SLOW_SUMMARY_TIMEOUT_SECONDS}\n" + return cfg + + +def write_home(hermes_home: Path, base_url: str, extra_config: str) -> None: + hermes_home.mkdir(parents=True, exist_ok=True) + (hermes_home / "config.yaml").write_text( + "model:\n" + " provider: custom\n" + f" base_url: {base_url}\n" + f" default: {MODEL_ID}\n" + f" context_length: {CONTEXT_LENGTH}\n" + "agent:\n" + " api_max_retries: 1\n" + + extra_config, + encoding="utf-8", + ) + (hermes_home / ".env").write_text("OPENAI_API_KEY=sk-fake-e2e\n", encoding="utf-8") + + +# --------------------------------------------------------------------------------------------------- +# Seeded transcripts +# --------------------------------------------------------------------------------------------------- + + +@dataclass +class TurnSpec: + marker: str + user: Any # str, or a list of content parts (image turns) + steps: list[Any] + kind: str + + +def _write_file(workdir: Path, name: str, rng: random.Random, chars: int) -> Path: + line_no = 0 + lines = [] + total = 0 + while total < chars: + line = f"{name}:{line_no} " + " ".join(rng.choice(("alpha", "bravo", "delta", "omega", "sigma")) for _ in range(14)) + lines.append(line) + total += len(line) + 1 + line_no += 1 + path = workdir / name + path.write_text("\n".join(lines) + "\n", encoding="utf-8") + return path + + +def generate_transcript(seed: int, n_turns: int, workdir: Path, *, kinds: Iterable[str] | None = None) -> list[TurnSpec]: + """A seeded random session: tool-pair heavy (incl. parallel calls), huge single turns, image parts, + long answers, and turns whose single tool result alone exceeds the compaction threshold.""" + rng = random.Random(seed) + workdir.mkdir(parents=True, exist_ok=True) + weights = {"tools": 5, "parallel": 3, "huge_user": 1, "image": 1, "chat": 2, "tail_bomb": 1} + if kinds is not None: + weights = {k: weights[k] for k in kinds} + names, ws = zip(*weights.items()) + specs: list[TurnSpec] = [] + fileno = 0 + + def new_file(chars: int) -> str: + nonlocal fileno + fileno += 1 + return str(_write_file(workdir, f"s{seed}f{fileno}.txt", rng, chars)) + + for i in range(n_turns): + marker = f"U{seed}x{i}x{rng.getrandbits(24):06x}" + # Turn 0 is always a plain tool turn: its user message is the session's task anchor. + kind = "tools" if i == 0 else rng.choices(names, ws)[0] + steps: list[Any] = [] + user: Any = f"{marker} please read the next file and report" + if kind == "tools": + for _ in range(rng.randint(1, 3)): + steps.append(ToolCall("read_file", {"path": new_file(rng.randint(600, 9000))})) + elif kind == "parallel": + for _ in range(rng.randint(1, 2)): + extra = [("read_file", {"path": new_file(rng.randint(400, 6000))}) for _ in range(rng.randint(1, 3))] + steps.append(ToolCall("read_file", {"path": new_file(rng.randint(400, 6000))}, parallel=extra)) + elif kind == "huge_user": + filler = " ".join(f"w{rng.randint(0, 99999)}" for _ in range(rng.randint(2500, 6000))) + user = f"{marker} here is a large paste to keep in mind: {filler}" + elif kind == "image": + user = [ + {"type": "text", "text": f"{marker} what is in this picture?"}, + {"type": "image_url", "image_url": {"url": f"data:image/png;base64,{_PNG_1PX}"}}, + ] + elif kind == "tail_bomb": + steps.append(ToolCall("read_file", {"path": new_file(int(THRESHOLD_TOKENS * 4 * 1.3))})) + answer_len = rng.randint(4000, 12000) if kind == "chat" else rng.randint(20, 200) + answer = f"answer for {marker}: " + " ".join(f"t{rng.randint(0, 9999)}" for _ in range(answer_len // 6)) + steps.append(Text(answer, chunk_chars=256)) + specs.append(TurnSpec(marker=marker, user=user, steps=steps, kind=kind)) + return specs + + +def job_turn(seed: int, workdir: Path, rounds: int) -> TurnSpec: + """A cron-shaped run: ONE user message (the job prompt) followed by many tool rounds, so compaction + has to fire mid-turn with the job prompt as the only real ask.""" + rng = random.Random(seed) + workdir.mkdir(parents=True, exist_ok=True) + marker = f"U{seed}x0x{rng.getrandbits(24):06x}" + steps: list[Any] = [] + for k in range(rounds): + path = _write_file(workdir, f"job{seed}r{k}.txt", rng, rng.randint(3000, 9000)) + steps.append(ToolCall("read_file", {"path": str(path)})) + steps.append(Text(f"job report for {marker}: done", chunk_chars=64)) + return TurnSpec(marker=marker, user=f"{marker} daily job: read every report file and summarize", steps=steps, kind="job") + + +# --------------------------------------------------------------------------------------------------- +# Normalisation + invariants +# --------------------------------------------------------------------------------------------------- + + +def flatten(content: Any) -> str: + if content is None: + return "" + if isinstance(content, str): + return content + if isinstance(content, list): + out = [] + for part in content: + if isinstance(part, dict): + out.append(str(part.get("text") or "")) + else: + out.append(str(part)) + return "\n".join(out) + return str(content) + + +def _squash(text: str) -> str: + return " ".join(text.split()) + + +def _calls(tool_calls: Any) -> tuple: + if isinstance(tool_calls, str): + try: + tool_calls = json.loads(tool_calls) + except ValueError: + tool_calls = [] + out = [] + for tc in tool_calls or []: + fn = tc.get("function") or {} + args = fn.get("arguments") + try: + args = json.dumps(json.loads(args), sort_keys=True) if isinstance(args, str) else json.dumps(args, sort_keys=True) + except ValueError: + pass + out.append((tc.get("id"), fn.get("name"), args)) + return tuple(out) + + +def norm(msg: dict[str, Any]) -> tuple: + """Identity of a message that must survive persistence and request building unchanged. + + A user message is identified by the seeded markers it carries (image turns are rendered to text + differently for the wire and for storage by design); everything else by its full content.""" + role = msg.get("role") + text = flatten(msg.get("content")) + if role == "user": + marks = tuple(sorted(set(MARKER_RE.findall(text)))) + return ("user", marks if marks else _squash(text)) + if role == "assistant": + return ("assistant", _squash(text), _calls(msg.get("tool_calls"))) + if role == "tool": + # Tool output may be shrunk on the wire by designed lossy steps (old-result pruning, oversized-result + # spill to disk); the pairing, not the bytes, is the invariant here. + return ("tool", msg.get("tool_call_id")) + return (role, _squash(text)) + + +def merge_users(items: list[tuple]) -> list[tuple]: + """Alternation repair merges back-to-back same-role rows (e.g. after an aborted turn) on the wire; the + merged row carries exactly the same user messages, so compare with the same merge applied.""" + out: list[tuple] = [] + for item in items: + if (item[0] == "user" and out and out[-1][0] == "user" + and isinstance(item[1], tuple) and isinstance(out[-1][1], tuple)): + out[-1] = ("user", tuple(sorted(set(out[-1][1]) | set(item[1])))) + continue + # Same repair for assistant->assistant (a split summary row followed by the next assistant). + if item[0] == "assistant" and out and out[-1][0] == "assistant" and not out[-1][2]: + out[-1] = ("assistant", _squash(f"{out[-1][1]} {item[1]}"), item[2]) + continue + out.append(item) + return out + + +def norm_list(msgs: Iterable[dict[str, Any]]) -> list[tuple]: + return merge_users([norm(m) for m in msgs if m.get("role") != "system"]) + + +def db_rows(db_path: Path, session_id: str) -> list[dict[str, Any]]: + conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True, timeout=30) + try: + conn.row_factory = sqlite3.Row + rows = conn.execute( + "SELECT id, role, content, tool_call_id, tool_calls, active, compacted FROM messages " + "WHERE session_id = ? ORDER BY id", (session_id,)).fetchall() + out = [] + for r in rows: + d = dict(r) + content = d["content"] + if isinstance(content, str) and content.startswith("\x00json:"): + d["content"] = flatten(json.loads(content[len("\x00json:"):])) + out.append(d) + return out + finally: + conn.close() + + +def active_rows(rows: list[dict[str, Any]]) -> list[dict[str, Any]]: + return [r for r in rows if r["active"]] + + +def assert_tool_pairs(msgs: list[dict[str, Any]], where: str, *, allow_trailing_pending: bool = False) -> None: + pending: dict[str, int] = {} + seen_calls: set[str] = set() + seen_results: set[str] = set() + for i, m in enumerate(msgs): + role = m.get("role") + if role == "assistant": + assert not pending, f"{where}: tool call(s) {sorted(pending)} unanswered before assistant #{i}" + for call_id, _name, _args in _calls(m.get("tool_calls")): + assert call_id not in seen_calls, f"{where}: tool_call id {call_id} issued twice (#{i})" + seen_calls.add(call_id) + pending[call_id] = i + elif role == "tool": + cid = m.get("tool_call_id") + assert cid in pending, f"{where}: orphan tool result {cid!r} at #{i} (no open assistant tool_call)" + assert cid not in seen_results, f"{where}: tool result {cid} duplicated (#{i})" + seen_results.add(cid) + del pending[cid] + elif role == "user": + assert not pending, f"{where}: tool call(s) {sorted(pending)} unanswered before user #{i}" + if not allow_trailing_pending: + assert not pending, f"{where}: tool call(s) {sorted(pending)} never answered" + + +def assert_wire_request_ok(body: dict[str, Any], turn_marker: str, base_system: str | None, where: str) -> None: + msgs = body["messages"] + assert msgs and msgs[0]["role"] == "system", f"{where}: system prompt missing from the request" + assert all(m["role"] != "system" for m in msgs[1:]), f"{where}: a second system message in the request" + if base_system is not None: + # The system anchor is built once per session; compaction may only append its note. + assert msgs[0]["content"].startswith(base_system), f"{where}: the session system prompt was rewritten" + # The in-flight ask (for a cron run: the job prompt) must reach the model AFTER any summary; a summary + # that swallows it leaves the model told to "respond to the message below" with nothing below (#100818). + text = "\n".join(flatten(m.get("content")) for m in msgs[1:] if m["role"] in ("user", "assistant", "tool")) + ask_at = max((i for i, m in enumerate(msgs) if m["role"] == "user" and turn_marker in flatten(m.get("content"))), + default=-1) + assert ask_at > 0, f"{where}: the in-flight user message {turn_marker} is missing from the request" + if GOOD_SUMMARY_TOKEN in text: + assert text.rfind(turn_marker) > text.rfind(GOOD_SUMMARY_TOKEN), ( + f"{where}: the in-flight user message {turn_marker} only survives inside/before the summary") + assert_tool_pairs(msgs, where) + for a, b in zip(msgs, msgs[1:]): + assert not (a["role"] == "user" and b["role"] == "user"), f"{where}: two consecutive user messages" + assert msgs[-1]["role"] in ("user", "tool"), f"{where}: request ends with {msgs[-1]['role']}" + + +def assert_persisted_prefix(snapshot: list[tuple], body: dict[str, Any], turn_marker: str, first_new_call: int, where: str) -> None: + """The durable rows at request time are exactly the request's history prefix; whatever follows is + only the in-flight turn (its own user message and the tool calls it issued).""" + sent = norm_list(body["messages"]) + expected = list(snapshot) + if not any(it[0] == "user" and isinstance(it[1], tuple) and turn_marker in it[1] for it in expected): + expected = merge_users(expected + [("user", (turn_marker,))]) + assert sent[: len(expected)] == expected, ( + f"{where}: request history diverges from state.db (+ the in-flight user message) at the time it " + f"was sent\n first mismatch {_diff_at(sent, expected)}\n sent={_brief(sent)}\n db ={_brief(expected)}") + for item in sent[len(expected):]: + assert item[0] != "user", f"{where}: unpersisted user message {item[1]} in the request after the in-flight turn" + if item[0] == "assistant": + for call_id, _n, _a in item[2]: + assert int(call_id.rsplit("_", 1)[1]) >= first_new_call, ( + f"{where}: unpersisted tool call {call_id} from an earlier turn in the request") + + +def assert_db_recoverable(rows: list[dict[str, Any]], sent_markers: Iterable[str], where: str) -> None: + live_or_archived = [r for r in rows if r["active"] or r["compacted"]] + marks_kept: set[str] = set() + for r in live_or_archived: + if r["role"] == "user": + marks_kept.update(MARKER_RE.findall(r["content"] or "")) + lost = [m for m in sent_markers if m not in marks_kept] + assert not lost, f"{where}: user message(s) {lost} are neither live nor archived in state.db (data loss)" + # A carried row may be re-inserted with the summary merged into its content, so the live copy only + # has to contain the original bytes (same role, same tool linkage). + keep: dict[tuple, list[str]] = {} + for r in live_or_archived: + keep.setdefault(_row_key(r), []).append(r["content"] or "") + # Tool outputs are exempt: pruning old tool results to stubs is a designed, lossy compaction step. + for r in rows: + if not r["active"] and not r["compacted"] and r["role"] in ("user", "assistant"): + copies = keep.get(_row_key(r), []) + if r["role"] == "user" and MARKER_RE.search(r["content"] or ""): + # Multimodal user rows are stored as a placeholder by the turn flush but as the full parts + # list when carried; identity is the seeded marker. + marks = set(MARKER_RE.findall(r["content"])) + covered = any(marks <= set(MARKER_RE.findall(c)) for c in copies) + else: + covered = any((r["content"] or "") in c for c in copies) + assert covered, ( + f"{where}: row {r['id']} ({r['role']}) was retired as a carried duplicate but no live or " + f"archived copy exists: {str(r['content'])[:80]!r}; same-linkage copies: " + f"{[c[:200] for c in copies]}") + + +def assert_db_matches_history(rows: list[dict[str, Any]], history: list[dict[str, Any]], where: str) -> None: + live = norm_list(_row_msg(r) for r in active_rows(rows)) + mem = norm_list(history) + assert live == mem, ( + f"{where}: live state.db rows differ from the host's in-memory history at #{_first_mismatch(live, mem)}\n" + f" db ={_brief(live)}\n mem={_brief(mem)}") + assert_tool_pairs([_row_msg(r) for r in active_rows(rows)], f"{where} (state.db live rows)") + + +def _row_msg(r: dict[str, Any]) -> dict[str, Any]: + return {"role": r["role"], "content": r["content"], "tool_calls": r["tool_calls"], "tool_call_id": r["tool_call_id"]} + + +def _row_key(r: dict[str, Any]) -> tuple: + return (r["role"], r["tool_call_id"], json.dumps(_calls(r["tool_calls"]))) + + +def _diff_at(a: list, b: list) -> str: + i = _first_mismatch(a, b) + x = a[i] if i < len(a) else None + y = b[i] if i < len(b) else None + return f"#{i}: {str(x)[:160]!r} ... {str(x)[-160:]!r}\n vs {str(y)[:160]!r} ... {str(y)[-160:]!r}" + + +def _first_mismatch(a: list, b: list) -> int: + for i, (x, y) in enumerate(zip(a, b)): + if x != y: + return i + return min(len(a), len(b)) + + +def _brief(items: list[tuple]) -> list[str]: + out = [] + for it in items: + if it[0] == "user": + out.append(f"U{it[1] if isinstance(it[1], tuple) else '(' + it[1][:30] + ')'}") + elif it[0] == "assistant": + out.append(f"A[{','.join(c[0] for c in it[2])}]{it[1][:24]!r}") + else: + out.append(f"{it[0][0].upper()}:{it[1]}") + return out + + +# --------------------------------------------------------------------------------------------------- +# Scenario driver +# --------------------------------------------------------------------------------------------------- + + +@dataclass +class TurnRecord: + spec: TurnSpec + result: dict[str, Any] + elapsed: float + main_requests: list[dict[str, Any]] + summary_calls: int + statuses: list[tuple] + aborted: bool = False + + +@dataclass +class Scenario: + """One session: its own HERMES_HOME, state.db and agent against the shared provider.""" + + server: FakeLLMServer + dispatch: Dispatch + home: Path + mode: str = "good" + session_id: str = "compaction-e2e" + platform: str = "cli" + window: int | None = None # provider-enforced context window (tokens); None = unlimited + history: list[dict[str, Any]] = field(default_factory=list) + sent_markers: list[str] = field(default_factory=list) + summary_calls: list = field(default_factory=list) + statuses: list[tuple] = field(default_factory=list) + turns: list[TurnRecord] = field(default_factory=list) + base_system: str | None = None + + def __post_init__(self) -> None: + self.db_path = self.home / "state.db" + self._queue: list[Any] = [] + self._qlock = threading.Lock() + self._snapshots: list[tuple[dict[str, Any], list[tuple]]] = [] + self.overflow_rejections = 0 + self.db = None + self.agent = None + + # provider side ------------------------------------------------------------------------------ + def install(self) -> None: + self.dispatch.main = self._main + self.dispatch.aux = summary_responder(self.mode, self.summary_calls) + + def _main(self, rec: dict[str, Any]) -> Any: + body = rec["body"] + snap = norm_list(_row_msg(r) for r in active_rows(db_rows(self.db_path, self.session_id))) if self.db_path.exists() else [] + self._snapshots.append((body, snap)) + if self.window is not None and estimate_prompt_tokens(body) > self.window: + self.overflow_rejections += 1 + return Error(400, f"This model's maximum context length is {self.window} tokens. However, your " + f"messages resulted in {estimate_prompt_tokens(body)} tokens. (context_length_exceeded)") + with self._qlock: + if self._queue: + return self._queue.pop(0) + return Text("(unscripted extra request)") + + # agent side --------------------------------------------------------------------------------- + def build_agent(self) -> Any: + from hermes_state import SessionDB + from run_agent import AIAgent + + if self.db is None: + self.db = SessionDB(db_path=self.db_path) + self.agent = AIAgent( + provider="custom", base_url=self.server.base_url, api_key="sk-fake-e2e", model=MODEL_ID, + session_db=self.db, session_id=self.session_id, quiet_mode=True, platform=self.platform, + enabled_toolsets=["file"], skip_context_files=True, skip_memory=True, + status_callback=lambda *a, **_k: self.statuses.append(tuple(a)), + ) + return self.agent + + def resume_from_db(self) -> None: + """Gateway/Desktop shape: a fresh agent whose history is whatever state.db hands back.""" + self.close_agent() + self.build_agent() + self.history = self.db.get_messages_as_conversation(self.session_id) + + def close_agent(self) -> None: + if self.agent is not None: + close = getattr(self.agent, "close", None) + if callable(close): + close() + self.agent = None + + def close(self) -> None: + self.close_agent() + if self.db is not None: + self.db.close() + self.db = None + + def run_turn(self, spec: TurnSpec, *, allow_abort: bool = False) -> TurnRecord: + if self.agent is None: + self.build_agent() + self.install() + with self._qlock: + self._queue = list(spec.steps) + first_req = len(self._snapshots) + summary_before = len(self.summary_calls) + status_before = len(self.statuses) + with self.server._lock: + first_new_call = self.server._tool_seq + 1 + started = time.monotonic() + result = self.agent.run_conversation(spec.user, conversation_history=self.history) + elapsed = time.monotonic() - started + self.sent_markers.append(spec.marker) + where = f"turn {len(self.turns)} ({spec.kind}, mode={self.mode})" + + # (1) liveness: bounded wall time, every scripted step consumed exactly once. + assert elapsed < TURN_WALL_BOUND_SECONDS, f"{where}: turn took {elapsed:.1f}s" + with self._qlock: + unconsumed = len(self._queue) + self._queue = [] + if unconsumed: + # Stopping early is only acceptable as an explicit, user-visible failure. + assert allow_abort, ( + f"{where}: turn ended with {unconsumed} scripted step(s) unconsumed: " + f"{ {k: result.get(k) for k in ('failed', 'error', 'turn_exit_reason', 'final_response')} }") + assert result.get("error") or result.get("failed"), ( + f"{where}: turn stopped early without an explicit error: " + f"{ {k: result.get(k) for k in ('failed', 'error', 'turn_exit_reason', 'final_response')} }") + reqs = self._snapshots[first_req:] + if self.base_system is None and reqs: + self.base_system = reqs[0][0]["messages"][0]["content"] + # (2)+(4) every request: well-formed, anchored, pairs matched, persisted prefix + in-flight turn. + for k, (body, snap) in enumerate(reqs): + w = f"{where} request {k}" + assert_wire_request_ok(body, spec.marker, self.base_system, w) + assert_persisted_prefix(snap, body, spec.marker, first_new_call, w) + rows = db_rows(self.db_path, self.session_id) + assert_db_recoverable(rows, self.sent_markers, where) + self.history = result["messages"] + assert_db_matches_history(rows, self.history, where) + rec = TurnRecord(spec, result, elapsed, [b for b, _ in reqs], len(self.summary_calls) - summary_before, + self.statuses[status_before:], aborted=bool(unconsumed)) + self.turns.append(rec) + return rec + + # derived facts ---------------------------------------------------------------------------------- + def committed_compactions(self) -> int: + return sum(1 for r in db_rows(self.db_path, self.session_id) if r["compacted"]) + + def summary_rows(self) -> list[dict[str, Any]]: + return [r for r in db_rows(self.db_path, self.session_id) if r["active"] and GOOD_SUMMARY_TOKEN in (r["content"] or "")] + + +def drive(sc: Scenario, seed: int, workdir: Path, n_turns: int = N_TURNS, *, resume_every: int = 4, + kinds: Iterable[str] | None = None) -> list[TurnSpec]: + """Run a seeded session through ``sc``; every ``resume_every`` turns the host is replaced by a fresh + agent resumed from state.db (the gateway/Desktop shape). Every turn asserts the shared invariants.""" + specs = generate_transcript(seed, n_turns, workdir, kinds=kinds) + for i, spec in enumerate(specs): + if i and resume_every and i % resume_every == 0: + sc.resume_from_db() + rec = sc.run_turn(spec) + assert rec.summary_calls <= MAX_SUMMARY_CALLS_PER_TURN, ( + f"turn {i}: {rec.summary_calls} summarizer calls in one turn (runaway compaction)") + return specs + + +def warned(sc: Scenario) -> bool: + """The user was told something about context/compaction (status ``warn`` / ``error`` lines).""" + return any(s and s[0] in ("warn", "error") for s in sc.statuses) diff --git a/tests/e2e/core/compaction/conftest.py b/tests/e2e/core/compaction/conftest.py new file mode 100644 index 0000000000..734495b295 --- /dev/null +++ b/tests/e2e/core/compaction/conftest.py @@ -0,0 +1,57 @@ +"""Lane-private fixtures for the compaction suites (one fake provider per module, one session per test).""" + +from __future__ import annotations + +import pytest + +from tests.e2e.core.compaction._helpers import ( + SLOW_SUMMARY_TIMEOUT_SECONDS, + THRESHOLD_TOKENS, + Scenario, + compaction_config, + start_provider, + write_home, +) + + +@pytest.fixture(scope="module") +def provider(): + server, dispatch = start_provider() + try: + yield server, dispatch + finally: + server.stop() + + +@pytest.fixture +def make_scenario(provider, tmp_path, monkeypatch): + """Build a Scenario with its own HERMES_HOME/state.db; closes every agent + DB at teardown.""" + server, dispatch = provider + made: list[Scenario] = [] + + def make(mode: str = "good", *, extra: str = "", platform: str = "cli", window: int | None = None, + threshold_tokens: int = THRESHOLD_TOKENS, name: str = "hermes_home") -> Scenario: + # The hermetic root conftest already isolates HERMES_HOME per test; HOME stays put because the + # state.db live-system guard treats $HOME/.hermes as production. + hermes_home = tmp_path / name + monkeypatch.setenv("HERMES_HOME", str(hermes_home)) + import hermes_state + + monkeypatch.setattr(hermes_state, "DEFAULT_DB_PATH", hermes_home / "state.db") + slow = mode == "slow" + if slow: + # The compression request timeout has a 300 s production floor; pin it (and the host idle + # watchdog) to seconds so a hung summarizer is judged inside a unit-test budget. + import agent.auxiliary_client as aux + + monkeypatch.setattr(aux, "_COMPRESSION_TIMEOUT_FLOOR_SECONDS", SLOW_SUMMARY_TIMEOUT_SECONDS) + extra += f" context_timeout_seconds: {SLOW_SUMMARY_TIMEOUT_SECONDS}\n" + write_home(hermes_home, server.base_url, + compaction_config(extra=extra, slow=slow, threshold_tokens=threshold_tokens)) + sc = Scenario(server=server, dispatch=dispatch, home=hermes_home, mode=mode, platform=platform, window=window) + made.append(sc) + return sc + + yield make + for sc in made: + sc.close() diff --git a/tests/e2e/core/compaction/test_compaction_auto.py b/tests/e2e/core/compaction/test_compaction_auto.py new file mode 100644 index 0000000000..e297ad99d4 --- /dev/null +++ b/tests/e2e/core/compaction/test_compaction_auto.py @@ -0,0 +1,86 @@ +"""C4 automatic compaction: property + liveness suite on seeded random sessions (good summarizer). + +A real AIAgent + SessionDB drives seeded random sessions (tool-pair heavy with parallel calls, huge single +turns, image parts, long answers, single tool results larger than the whole compaction threshold) against +the recording fake provider. After EVERY turn and for EVERY request the shared invariants in ``_helpers`` +are asserted: + +* liveness -- the turn returns within a bounded wall time and a bounded number of summarizer calls; + a session left over the trigger surfaces a warning; a provider-side overflow ends the turn with an + explicit error or recovers, within a bounded number of retries (never a silent loop); +* safety -- no user message is lost from state.db, tool_call/tool_result pairs stay matched in the + live rows AND in every request, the system prompt stays the session's, and the in-flight ask (for a + cron run: the job prompt) reaches the model after every summary; +* persisted == sent -- state.db at request time is exactly the request's history prefix (+ only the + in-flight turn), the live rows equal the host's in-memory history after each turn, and a fresh agent + resumed from state.db continues with the same history. +""" + +from __future__ import annotations + +import pytest + +from tests.e2e.core.compaction._helpers import ( + GOOD_SUMMARY_TOKEN, + MAX_SUMMARY_CALLS_PER_TURN, + SEEDS, + THRESHOLD_TOKENS, + drive, + estimate_prompt_tokens, + generate_transcript, + job_turn, + warned, +) + + +@pytest.mark.parametrize("seed", SEEDS) +def test_good_summarizer_compacts_and_keeps_every_invariant(make_scenario, tmp_path, seed): + sc = make_scenario("good") + drive(sc, seed, tmp_path / "work") + assert sc.summary_calls, "the seeded session never crossed the compaction trigger" + assert sc.committed_compactions() > 0, "a good summary was produced but no compaction committed" + # A committed good summary is what the model sees next. + last = sc.turns[-1].main_requests[-1] + assert any(GOOD_SUMMARY_TOKEN in str(m.get("content")) for m in last["messages"]), ( + "the committed summary is not in the request after compaction") + # Ends below the trigger, or said so explicitly. + if estimate_prompt_tokens(last) >= THRESHOLD_TOKENS: + assert warned(sc), "session left over the trigger with no warning" + + +@pytest.mark.parametrize("platform", ("cron", "cli")) +def test_job_prompt_survives_mid_turn_compaction(make_scenario, tmp_path, platform): + """One job prompt, many tool rounds: compaction fires mid-turn (repeatedly). The job prompt must + still reach the model after every summary, every tool result the turn produced must stay paired, + and the run must actually finish its script (#100818: the run "succeeded" but delivered nothing).""" + sc = make_scenario("good", platform=platform) + spec = job_turn(SEEDS[0], tmp_path / "work", rounds=14) + rec = sc.run_turn(spec) + assert sc.committed_compactions() > 0, "the job run never compacted mid-turn" + assert rec.summary_calls <= MAX_SUMMARY_CALLS_PER_TURN * 2 + assert spec.marker in str(rec.result.get("final_response")), "the job run did not deliver its report" + + +# Overflow retries per turn: each rejected request may trigger one compaction retry, capped by the +# per-turn compression attempt budget (3). Anything past a small multiple is a retry loop. +MAX_OVERFLOW_REJECTIONS_PER_TURN = 6 + + +@pytest.mark.parametrize("mode", ("empty", "good")) +def test_provider_overflow_terminates(make_scenario, tmp_path, mode): + """The provider itself rejects over-window requests (the local-server shape: a real window smaller + than the configured one). Recovery is bounded per turn and either succeeds or ends the turn with an + explicit error -- never a silent loop -- and every invariant still holds afterwards.""" + sc = make_scenario(mode, window=THRESHOLD_TOKENS * 3 // 4) + specs = generate_transcript(SEEDS[1], 10, tmp_path / "work", kinds=("tail_bomb", "tools", "huge_user")) + for i, spec in enumerate(specs): + before = sc.overflow_rejections + rec = sc.run_turn(spec, allow_abort=True) + assert sc.overflow_rejections - before <= MAX_OVERFLOW_REJECTIONS_PER_TURN, ( + f"turn {i}: {sc.overflow_rejections - before} over-window requests in one turn (retry loop)") + assert rec.summary_calls <= MAX_SUMMARY_CALLS_PER_TURN, f"turn {i}: {rec.summary_calls} summarizer calls" + assert sc.overflow_rejections, "the scenario never hit the provider window" + if mode == "good": + assert sc.committed_compactions() > 0 + else: + assert sc.committed_compactions() == 0 From 9426b837cf28e7b48930b947afe237143cff88bc Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:01:15 -0700 Subject: [PATCH 008/104] test: summarizer faults never commit and stay bounded + visible (C4) Same real chain and per-turn invariants, with the summarizer answering empty / refusal / truncated (finish_reason=length) / slower than the compression timeout / HTTP 500. A bad summary must never archive history (P0 #94448 empty-summary deletes the middle; refusal accepted as a summary), never reach the model or state.db, must be surfaced to the user, and must not loop (877-summarizer-call class). A 500 may commit only the designed deterministic fallback, never with abort_on_summary_failure. --- .../test_compaction_summary_faults.py | 54 +++++++++++++++++++ 1 file changed, 54 insertions(+) create mode 100644 tests/e2e/core/compaction/test_compaction_summary_faults.py diff --git a/tests/e2e/core/compaction/test_compaction_summary_faults.py b/tests/e2e/core/compaction/test_compaction_summary_faults.py new file mode 100644 index 0000000000..f6ea343325 --- /dev/null +++ b/tests/e2e/core/compaction/test_compaction_summary_faults.py @@ -0,0 +1,54 @@ +"""C4 summarizer faults: a failed summary never commits, and failure is bounded and visible. + +Same real chain and per-turn invariants as ``test_compaction_auto`` (see ``_helpers``), with the +summarizer scripted to answer empty / refusal / truncated (finish_reason=length) / slower than the +compression timeout / HTTP 500. For every fault: the session keeps running, the summarizer is asked a +bounded number of times per turn, the user is told, and state.db never archives history behind a bad +summary (a 500 may commit only the designed deterministic fallback -- never with +``abort_on_summary_failure``). +""" + +from __future__ import annotations + +import pytest + +from tests.e2e.core.compaction._helpers import REFUSAL_TEXT, SEEDS, TRUNCATED_TEXT, db_rows, drive, warned + +FAILING_MODES = ("empty", "refusal", "truncated", "slow") + + +@pytest.mark.parametrize("mode", FAILING_MODES) +@pytest.mark.parametrize("seed", SEEDS[:2]) +def test_failed_summary_never_commits(make_scenario, tmp_path, seed, mode): + sc = make_scenario(mode) + drive(sc, seed, tmp_path / "work") + assert sc.summary_calls, "the seeded session never crossed the compaction trigger" + assert sc.committed_compactions() == 0, f"a {mode} summary committed a compaction (history archived)" + bad = {"refusal": REFUSAL_TEXT, "truncated": TRUNCATED_TEXT}.get(mode) + if bad: + for rec in sc.turns: + for body in rec.main_requests: + assert not any(bad in str(m.get("content")) for m in body["messages"]), ( + f"the {mode} summary text reached the model") + assert not any(bad in (r["content"] or "") for r in db_rows(sc.db_path, sc.session_id)), ( + f"the {mode} summary text was written to state.db") + # The session is over the trigger with no usable summary: the user must be told. + assert warned(sc), f"{mode} summarizer: compaction failed silently (no warning surfaced)" + + +@pytest.mark.parametrize("seed", SEEDS[:1]) +def test_http500_summarizer_is_bounded_and_lossless(make_scenario, tmp_path, seed): + """A 500 may abort or commit the deterministic fallback (``abort_on_summary_failure: false``), but + either way nothing is lost and it is surfaced.""" + sc = make_scenario("http500") + drive(sc, seed, tmp_path / "work") + assert sc.summary_calls + assert warned(sc) + + +@pytest.mark.parametrize("seed", SEEDS[1:2]) +def test_http500_with_abort_on_failure_never_commits(make_scenario, tmp_path, seed): + sc = make_scenario("http500", extra=" abort_on_summary_failure: true\n") + drive(sc, seed, tmp_path / "work") + assert sc.summary_calls + assert sc.committed_compactions() == 0 From 1f993e027208c6475b506a7f773fb8fd6e7bf551 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:01:15 -0700 Subject: [PATCH 009/104] test: /compress, /compress here N and micro-compaction keep persisted == sent (C4) Drives manual compaction through the TUI/Desktop choke point (tui_gateway _compress_session_history -> compress_now) and rolling micro-compaction through the real turn finalizer, then asserts no user message lost, live rows == in-memory history, kept exchanges stay live verbatim rows, failed summaries change nothing, and the next turn from memory AND from a fresh agent resumed off state.db both send exactly the persisted history. Covers the 526d135a96a (/compress here N tail lost from state.db) and 91df54184d3 (micro-compaction marks summarized rows rewind-only) data-loss class. --- .../core/compaction/test_compaction_manual.py | 131 ++++++++++++++++++ 1 file changed, 131 insertions(+) create mode 100644 tests/e2e/core/compaction/test_compaction_manual.py diff --git a/tests/e2e/core/compaction/test_compaction_manual.py b/tests/e2e/core/compaction/test_compaction_manual.py new file mode 100644 index 0000000000..14fe254fdb --- /dev/null +++ b/tests/e2e/core/compaction/test_compaction_manual.py @@ -0,0 +1,131 @@ +"""C4 manual compaction: ``/compress``, ``/compress here N`` and rolling micro-compaction, end to end. + +``/compress`` goes through the TUI/Desktop choke point (``tui_gateway`` ``_compress_session_history`` -> +``compress_now``) on a real AIAgent + SessionDB after a seeded session; micro-compaction runs from the +real turn finalizer. After the compaction the same invariants as the automatic suite are asserted: + +* no user message is lost from state.db (live, or archived as summarized history); +* the live rows equal the host's in-memory history (what the TUI keeps and sends next); +* the next turn from the in-memory history AND the next turn from a fresh agent resumed off state.db + (the gateway shape) both send exactly the persisted history (+ the new message); +* a failed summary never commits: history and state.db stay byte-identical. +""" + +from __future__ import annotations + +import threading + +import pytest + +from tests.e2e.core.compaction._helpers import ( + GOOD_SUMMARY_TOKEN, + MARKER_RE, + Scenario, + active_rows, + assert_db_matches_history, + assert_db_recoverable, + db_rows, + generate_transcript, + norm_list, + _row_msg, +) + +# Manual tests keep the automatic trigger out of the way so only the command under test compacts. +HIGH_THRESHOLD = 60_000 +PRE_TURNS = 6 +POST_TURNS = 2 + + +def _tui_session(sc: Scenario) -> dict: + return { + "agent": sc.agent, "history": list(sc.history), "history_lock": threading.Lock(), + "history_version": 1, "session_key": sc.session_id, + } + + +def _compress_via_tui(sc: Scenario, args: str) -> int: + from tui_gateway.server import _compress_session_history + + session = _tui_session(sc) + sc.install() + removed, _usage = _compress_session_history(session, args) + sc.history = session["history"] + return removed + + +def _live(sc: Scenario) -> list[tuple]: + return norm_list(_row_msg(r) for r in active_rows(db_rows(sc.db_path, sc.session_id))) + + +def _run_after(sc: Scenario, specs, where: str) -> None: + """In-memory continuation, then a fresh agent resumed from state.db: both send persisted history.""" + rows = db_rows(sc.db_path, sc.session_id) + assert_db_recoverable(rows, sc.sent_markers, where) + assert_db_matches_history(rows, sc.history, where) + for spec in specs[:-1]: + sc.run_turn(spec) + sc.resume_from_db() + sc.run_turn(specs[-1]) + + +# Each boundary form once; seeds alternate so both seeded session shapes meet full and partial compress. +@pytest.mark.parametrize(("seed", "args"), [ + (5, ""), (17, ""), (5, "here"), (17, "here 1"), (5, "here 2"), (17, "here 3"), (5, "keep the file names"), +]) +def test_manual_compress_keeps_every_invariant(make_scenario, tmp_path, seed, args): + sc = make_scenario("good", threshold_tokens=HIGH_THRESHOLD) + specs = generate_transcript(seed, PRE_TURNS + POST_TURNS + 1, tmp_path / "work", + kinds=("tools", "parallel", "chat", "image")) + for spec in specs[:PRE_TURNS]: + sc.run_turn(spec) + before_live = _live(sc) + calls_before = len(sc.summary_calls) + + removed = _compress_via_tui(sc, args) + + assert len(sc.summary_calls) - calls_before <= 2, "manual compress made a runaway number of summarizer calls" + assert removed > 0 and sc.committed_compactions() > 0, f"/compress {args!r} did not compact" + assert _live(sc) != before_live + assert any(GOOD_SUMMARY_TOKEN in str(m.get("content")) for m in sc.history), "summary missing from history" + if args.startswith("here"): + # The kept exchanges (the newest N user turns) stay verbatim and live. + n = int(args.split()[1]) if len(args.split()) > 1 else 1 + kept = [s.marker for s in specs[:PRE_TURNS]][-n:] + live_user_rows = [r for r in active_rows(db_rows(sc.db_path, sc.session_id)) if r["role"] == "user"] + for marker in kept: + assert any(MARKER_RE.findall(r["content"] or "") == [marker] for r in live_user_rows), ( + f"/compress {args}: kept exchange {marker} is not a live verbatim row in state.db") + _run_after(sc, specs[PRE_TURNS:], f"after /compress {args!r}") + + +@pytest.mark.parametrize("mode", ("empty", "refusal", "truncated")) +@pytest.mark.parametrize("args", ["", "here 2"]) +def test_manual_compress_with_failed_summary_changes_nothing(make_scenario, tmp_path, mode, args): + sc = make_scenario(mode, threshold_tokens=HIGH_THRESHOLD) + specs = generate_transcript(7, PRE_TURNS + POST_TURNS + 1, tmp_path / "work", kinds=("tools", "parallel", "chat")) + for spec in specs[:PRE_TURNS]: + sc.run_turn(spec) + rows_before = db_rows(sc.db_path, sc.session_id) + history_before = list(sc.history) + + removed = _compress_via_tui(sc, args) + + assert sc.summary_calls, "the summarizer was never asked" + assert removed == 0 + assert db_rows(sc.db_path, sc.session_id) == rows_before, f"a {mode} summary changed state.db" + assert norm_list(sc.history) == norm_list(history_before), f"a {mode} summary changed the live history" + _run_after(sc, specs[PRE_TURNS:], f"after failed /compress {args!r}") + + +@pytest.mark.parametrize("seed", (3, 29)) +def test_micro_compaction_keeps_every_invariant(make_scenario, tmp_path, seed): + """Rolling micro-compaction folds one old exchange per turn into a summary marker between turns, + carrying a non-contiguous prefix + suffix; summarized rows must be archived, carried rows kept.""" + sc = make_scenario("good", threshold_tokens=HIGH_THRESHOLD, extra=" micro_compact: true\n") + specs = generate_transcript(seed, 10, tmp_path / "work", kinds=("tools", "parallel", "chat")) + for i, spec in enumerate(specs): + if i == 6: + sc.resume_from_db() + sc.run_turn(spec) + assert sc.summary_calls, "micro-compaction never ran" + assert sc.committed_compactions() > 0, "micro-compaction never archived a summarized exchange" From 17d592053a6d298750977e6d9519025460f005c5 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:01:15 -0700 Subject: [PATCH 010/104] test: kill -9 mid-compaction leaves state.db consistent and resumable (C4) A real child process runs /compress on a seeded session's state.db; the parent SIGKILLs it while the summary request is in flight and inside the archive_and_compact write transaction (parked by a child-side hook after archive + inserts, before COMMIT). Asserts integrity_check ok, rows byte-identical to pre-compaction, message_count consistent, then a fresh agent reclaims the dead holder's lease, compacts and continues with persisted == sent. --- .../core/compaction/test_compaction_kill9.py | 187 ++++++++++++++++++ 1 file changed, 187 insertions(+) create mode 100644 tests/e2e/core/compaction/test_compaction_kill9.py diff --git a/tests/e2e/core/compaction/test_compaction_kill9.py b/tests/e2e/core/compaction/test_compaction_kill9.py new file mode 100644 index 0000000000..b38b0ac90f --- /dev/null +++ b/tests/e2e/core/compaction/test_compaction_kill9.py @@ -0,0 +1,187 @@ +"""C4 compaction atomicity: kill -9 a real child process mid-compaction, then prove state.db is intact +and the session is resumable. + +A seeded session is built in-process on a real SessionDB; then a CHILD process (real AIAgent against the +same state.db and the same fake provider) runs ``/compress``. The parent SIGKILLs it at two points: + +* ``summarizer`` — while the summary request is in flight (lease held, nothing written yet); +* ``commit`` — inside the ``archive_and_compact`` write transaction, after the archive UPDATE and the + compacted-row INSERTs but before COMMIT (the child is parked there by a test hook). + +Invariants after the kill: ``PRAGMA integrity_check`` is ok; the live rows are exactly the pre-compaction +rows (no half-archived session, nothing lost, no half-inserted summary); the session counter matches; +and a fresh agent resumed from state.db can compact (the dead holder's lease is reclaimed) and continue, +sending exactly the persisted history. +""" + +from __future__ import annotations + +import os +import signal +import sqlite3 +import subprocess +import sys +import threading +import time +from pathlib import Path + +import pytest + +from tests.e2e.core.compaction._helpers import ( + GOOD_SUMMARY_TOKEN, + Hang, + Text, + active_rows, + assert_db_matches_history, + assert_db_recoverable, + db_rows, + generate_transcript, + good_summary, +) + +REPO = Path(__file__).resolve().parents[4] +HIGH_THRESHOLD = 60_000 +KILL_DEADLINE_SECONDS = 120.0 + +CHILD = r''' +import os, sys, time +from pathlib import Path +repo, db_path, session_id, base_url, point, sentinel = sys.argv[1:7] +sys.path.insert(0, repo) +from hermes_state import SessionDB + +if point == "commit": + # Park inside archive_and_compact's write transaction (after archive + inserts, before COMMIT). + in_commit = {"on": False} + orig_aac = SessionDB.archive_and_compact + orig_insert = SessionDB._insert_message_rows + + def aac(self, *a, **k): + in_commit["on"] = True + return orig_aac(self, *a, **k) + + def insert(self, conn, *a, **k): + out = orig_insert(self, conn, *a, **k) + if in_commit["on"]: + with open(sentinel, "w") as fh: + fh.write(str(os.getpid())) + time.sleep(3600) + return out + + SessionDB.archive_and_compact = aac + SessionDB._insert_message_rows = insert + +from run_agent import AIAgent +from agent.conversation_compression_manual import compress_now, parse_compress_args + +db = SessionDB(db_path=Path(db_path)) +agent = AIAgent(provider="custom", base_url=base_url, api_key="sk-fake-e2e", model="fake-model", + session_db=db, session_id=session_id, quiet_mode=True, enabled_toolsets=["file"], + skip_context_files=True, skip_memory=True) +history = db.get_messages_as_conversation(session_id) +result = compress_now(agent, history, parse_compress_args("")) +print("CHILD-FINISHED", result.status, flush=True) +''' + + +def _child_env(home: Path) -> dict: + # Probe hygiene: tmp HOME/HERMES_HOME, no provider keys. The in-test db guard detects pytest by + # process ancestry and would read this tmp $HOME/.hermes as the production root; the documented + # child-process escape hatch is safe because every path here lives under tmp_path. + env = {k: v for k, v in os.environ.items() if not k.endswith("_API_KEY")} + env.update(HOME=str(home), HERMES_HOME=str(home / ".hermes"), PYTHONPATH=str(REPO), PYTHONUNBUFFERED="1", + HERMES_STATE_DB_GUARD_BYPASS="1") + return env + + +def _poll(pred, deadline: float, what: str, child: subprocess.Popen) -> None: + end = time.monotonic() + deadline + while time.monotonic() < end: + if pred(): + return + if child.poll() is not None: + out = child.stdout.read() if child.stdout else "" + raise AssertionError(f"child exited ({child.returncode}) before {what}: {out[-2000:]}") + time.sleep(0.05) + raise AssertionError(f"timed out waiting for {what}") + + +def _integrity(db_path: Path) -> str: + conn = sqlite3.connect(db_path, timeout=30) + try: + return conn.execute("PRAGMA integrity_check").fetchone()[0] + finally: + conn.close() + + +@pytest.mark.parametrize("point", ("summarizer", "commit")) +def test_kill9_mid_compaction_leaves_state_consistent_and_resumable(make_scenario, provider, tmp_path, point): + server, dispatch = provider + home = tmp_path / "home" + sc = make_scenario("good", threshold_tokens=HIGH_THRESHOLD, name="home/.hermes") + specs = generate_transcript(41, 9, tmp_path / "work", kinds=("tools", "parallel", "chat")) + for spec in specs[:6]: + sc.run_turn(spec) + sc.close() # hand the session to the child: no open writer in this process + rows_before = db_rows(sc.db_path, sc.session_id) + + in_flight = threading.Event() + + def summarizer(rec): + sc.summary_calls.append(rec) + if point == "summarizer": + in_flight.set() + return Hang(KILL_DEADLINE_SECONDS) + return Text(good_summary(len(sc.summary_calls))) + + dispatch.aux = summarizer + sentinel = tmp_path / "in-commit" + script = tmp_path / "child.py" + script.write_text(CHILD, encoding="utf-8") + child = subprocess.Popen( + [sys.executable, str(script), str(REPO), str(sc.db_path), sc.session_id, server.base_url, point, str(sentinel)], + env=_child_env(home), cwd=str(tmp_path), stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True) + try: + if point == "summarizer": + _poll(in_flight.is_set, KILL_DEADLINE_SECONDS, "the summary request", child) + else: + _poll(sentinel.exists, KILL_DEADLINE_SECONDS, "the commit transaction", child) + os.kill(child.pid, signal.SIGKILL) + child.wait(timeout=30) + finally: + if child.poll() is None: + child.kill() + child.wait(timeout=30) + assert child.returncode == -signal.SIGKILL + + # (3) atomicity: the store is intact and exactly the pre-compaction state. + assert _integrity(sc.db_path) == "ok" + rows_after = db_rows(sc.db_path, sc.session_id) + assert rows_after == rows_before, ( + f"kill -9 at {point} left a partial compaction: {len(rows_before)} rows before, {len(rows_after)} after; " + f"live {len(active_rows(rows_before))} -> {len(active_rows(rows_after))}") + conn = sqlite3.connect(f"file:{sc.db_path}?mode=ro", uri=True) + try: + count = conn.execute("SELECT message_count FROM sessions WHERE id = ?", (sc.session_id,)).fetchone()[0] + finally: + conn.close() + assert count == len(active_rows(rows_after)), "session message_count disagrees with the live rows" + + # Resumable: a fresh agent reclaims the dead holder's lease, compacts, and continues. + from tui_gateway.server import _compress_session_history + + sc.install() + sc.resume_from_db() + session = {"agent": sc.agent, "history": list(sc.history), "history_lock": threading.Lock(), + "history_version": 1, "session_key": sc.session_id} + removed, _ = _compress_session_history(session, "") + sc.history = session["history"] + assert removed > 0 and sc.committed_compactions() > 0, "the session could not be compacted after the crash" + assert any(GOOD_SUMMARY_TOKEN in str(m.get("content")) for m in sc.history) + rows = db_rows(sc.db_path, sc.session_id) + assert_db_recoverable(rows, sc.sent_markers, "after crash + recompress") + assert_db_matches_history(rows, sc.history, "after crash + recompress") + for spec in specs[6:8]: + sc.run_turn(spec) + sc.resume_from_db() + sc.run_turn(specs[8]) From 11230397905aca14beb4ee2514fdc68f47427da9 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:55:40 -0700 Subject: [PATCH 011/104] test: C3 transcript ledger e2e (persisted == replayed == resumed, per step) Class C3 (conversation-history persistence & replay integrity). A real AIAgent + SessionDB runs a scenario matrix (plain turns, parallel tool batches, /steer mid-turn, interrupt, stream drop + 500 retry, manual /compress, /compress here N, auto compaction, opt-in micro-compaction) against the recording fake provider. After EVERY step it asserts: exactly-once at the storage layer and in the model view, next request replays the persisted view, nothing the model saw ever leaves the durable/display history, each typed input shown once, rows monotonic, usage == provider-billed; then a FRESH 'hermes chat --resume' process must replay a byte-identical request prefix. Red-proofs: reverting 526d135a96a (compress-here tail) and 91df54184d3 (micro-compaction rewind flags) and an injected double row insert all go red. A strict xfail documents a live micro-compaction bug (merged user row duplicates both inputs in the resumed display). The fake provider now records the usage it billed per request (record['usage']). --- tests/e2e/core/history/__init__.py | 0 tests/e2e/core/history/_helpers.py | 647 ++++++++++++++++++ .../core/history/test_transcript_ledger.py | 291 ++++++++ tests/fakes/fake_llm_provider.py | 9 +- 4 files changed, 945 insertions(+), 2 deletions(-) create mode 100644 tests/e2e/core/history/__init__.py create mode 100644 tests/e2e/core/history/_helpers.py create mode 100644 tests/e2e/core/history/test_transcript_ledger.py diff --git a/tests/e2e/core/history/__init__.py b/tests/e2e/core/history/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/tests/e2e/core/history/_helpers.py b/tests/e2e/core/history/_helpers.py new file mode 100644 index 0000000000..997963f38d --- /dev/null +++ b/tests/e2e/core/history/_helpers.py @@ -0,0 +1,647 @@ +"""Transcript-ledger harness for the conversation-history (C3) and prompt-cache (C17) suites. + +Everything real except the model: a real ``AIAgent`` + ``SessionDB`` on disk in-process, the +real ``tui_gateway`` stdio JSON-RPC server and the real ``hermes chat -q`` oneshot CLI as +subprocesses, all pointed at one recording ``FakeLLMServer``. The invariants below read the +three durable/observable projections of a conversation and cross-check them: + +* what the model was sent (the fake's recorded request bodies), +* what the model will be sent on resume (``SessionDB.get_resume_conversations`` model view), +* what a user sees on resume (the display view, compaction archive included), +* what was billed (``sessions`` / ``session_model_usage`` vs the fake's reported usage). +""" + +from __future__ import annotations + +import hashlib +import json +import os +import queue +import re +import sqlite3 +import subprocess +import sys +import threading +import time +from pathlib import Path +from typing import Any, Iterable + +REPO_ROOT = Path(__file__).resolve().parents[4] + +# Children get an ALLOWLISTED env: a developer or agent shell exports TERMINAL_CWD, +# HERMES_SESSION_*, AUXILIARY_*, provider keys ... any of which changes the child's prompt or +# routing and would make a surface-switch comparison diverge for reasons outside the product. +_ENV_ALLOW = ("PATH", "LANG", "LC_ALL", "LC_CTYPE", "USER", "LOGNAME", "SHELL", "SSL_CERT_FILE", "TZ") + +# No network from probes: no update check, models.dev fetch fails fast on a closed port. +OFFLINE_CONFIG = "updates:\n check: false\nmodels_dev:\n url: http://127.0.0.1:9/api.json\n" +# The periodic memory/skill background review forks a second agent against the same provider on +# its own thread; its requests would race the scripted turns. Scenarios that want it opt in. +NO_BACKGROUND_REVIEW = "skills:\n creation_nudge_interval: 0\nmemory:\n nudge_interval: 0\n" + +SUMMARY_TEXT = "Fake summary of the earlier conversation." + + +# --------------------------------------------------------------------------------------------- +# canonical forms + + +def canon(obj: Any) -> str: + """Canonical JSON: key order is not semantic on the wire, every value byte is.""" + return json.dumps(obj, sort_keys=True, ensure_ascii=False, separators=(",", ":")) + + +def _text(content: Any) -> str: + return content if isinstance(content, str) else canon(content) + + +def _args(raw: Any) -> str: + """Tool-call arguments by meaning: the wire re-serializes the model's JSON compactly while the + row keeps the model's bytes; the request stream's own byte-stability is ``prefix_breaks``' job.""" + try: + return canon(json.loads(raw)) if isinstance(raw, str) else canon(raw) + except ValueError: + return str(raw) + + +def _calls(tool_calls: Any) -> tuple: + out = [] + for tc in tool_calls or (): + fn = tc.get("function") or {} + out.append((tc.get("id"), fn.get("name"), _args(fn.get("arguments")))) + return tuple(out) + + +def identity(msg: dict[str, Any]) -> tuple: + """Logical identity of one conversation message as the MODEL sees it. + + User rows replay their ``api_content`` sidecar when present (surface notes, attachments), + so that is the text the identity uses. + """ + content = msg.get("api_content") if msg.get("role") == "user" and msg.get("api_content") else msg.get("content") + return (msg.get("role"), _text(content or ""), msg.get("tool_call_id") or None, _calls(msg.get("tool_calls"))) + + +def short(ident: tuple) -> str: + role, text, tcid, calls = ident + extra = f" tool_call_id={tcid}" if tcid else "" + extra += f" calls={list(calls)}" if calls else "" + return f"{role}:{text[:70]!r}{extra} #{hashlib.sha1(text.encode()).hexdigest()[:8]}" + + +def first_divergence(a: str, b: str) -> str: + n = next((i for i, (x, y) in enumerate(zip(a, b)) if x != y), min(len(a), len(b))) + return f"first diverging byte at {n}: {a[max(0, n - 60):n + 60]!r} vs {b[max(0, n - 60):n + 60]!r}" + + +# --------------------------------------------------------------------------------------------- +# state.db views + + +def db_connect(hermes_home: Path) -> sqlite3.Connection: + con = sqlite3.connect(f"file:{hermes_home / 'state.db'}?mode=ro", uri=True, timeout=30) + con.row_factory = sqlite3.Row + return con + + +def views(hermes_home: Path, sid: str) -> tuple[list[dict], list[dict]]: + """(model_history, display_history) exactly as a resuming surface loads them.""" + from hermes_state import SessionDB + + db = SessionDB(db_path=hermes_home / "state.db") + try: + return db.get_resume_conversations(sid) + finally: + db.close() + + +def row_counts(hermes_home: Path, sid: str) -> dict[str, int]: + with db_connect(hermes_home) as con: + total, active = con.execute( + "SELECT count(*), coalesce(sum(active), 0) FROM messages WHERE session_id=?", (sid,)).fetchone() + return {"total": total, "active": active} + + +def integrity_ok(hermes_home: Path) -> None: + with db_connect(hermes_home) as con: + assert con.execute("PRAGMA integrity_check").fetchone()[0] == "ok" + + +def lineage(hermes_home: Path, sid: str) -> list[str]: + """Root→tip chain of session ids ending at ``sid`` (compaction rotation may create children).""" + chain = [sid] + with db_connect(hermes_home) as con: + while True: + row = con.execute("SELECT parent_session_id FROM sessions WHERE id=?", (chain[0],)).fetchone() + if not row or not row[0]: + return chain + chain.insert(0, row[0]) + + +# --------------------------------------------------------------------------------------------- +# invariants + + +def is_summary(msg: dict[str, Any]) -> bool: + return SUMMARY_TEXT in _text(msg.get("content") or "") or bool(msg.get("_compressed_summary")) + + +def assert_rows_exactly_once(hermes_home: Path, sid: str, where: str) -> None: + """(a) at the storage layer: no two ACTIVE rows of a session are the same logical message. + Readers dedupe defensively, so a double write can hide behind them; the rows cannot.""" + con = db_connect(hermes_home) + try: + rows = con.execute( + "SELECT id, role, content, tool_call_id, tool_calls FROM messages " + "WHERE session_id = ? AND active = 1 ORDER BY id", (sid,)).fetchall() + finally: + con.close() + seen: dict[tuple, int] = {} + dups = [] + for row_id, *ident in rows: + key = tuple(ident) + if key in seen: + dups.append((seen[key], row_id, ident[0], str(ident[1])[:60])) + seen.setdefault(key, row_id) + assert not dups, f"{where}: the same message is persisted as more than one active row: {dups}" + + +def assert_exactly_once(msgs: Iterable[dict[str, Any]], where: str) -> None: + """(a) No logical message appears twice in a projection (every scripted text is unique).""" + seen: dict[tuple, int] = {} + for i, m in enumerate(msgs): + if m.get("role") == "system": + continue + k = identity(m) + assert k not in seen, f"{where}: {short(k)} persisted twice (positions {seen[k]} and {i})" + seen[k] = i + + +def model_payload(msgs: Iterable[dict[str, Any]]) -> list[tuple]: + return [identity(m) for m in msgs if m.get("role") != "system"] + + +def assert_replay_equals_persisted(request: dict[str, Any], persisted: list[dict], where: str) -> None: + """(b) The first request of turn N+1 replays exactly the persisted model history + the new user msg.""" + sent = model_payload(request["messages"]) + want = model_payload(persisted) + assert sent[:-1] == want, ( + f"{where}: history sent to the model != persisted history\n" + + _list_diff(want, sent[:-1], "persisted", "sent")) + assert sent[-1][0] == "user", f"{where}: the turn's request does not end with the new user message" + + +def _list_diff(a: list[tuple], b: list[tuple], an: str, bn: str) -> str: + n = next((i for i, (x, y) in enumerate(zip(a, b)) if x != y), min(len(a), len(b))) + lines = [f" lengths {an}={len(a)} {bn}={len(b)}; first difference at index {n}"] + for label, seq in ((an, a), (bn, b)): + lines.append(f" {label}[{n}:{n + 3}] = " + "; ".join(short(x) for x in seq[n:n + 3])) + return "\n".join(lines) + + +def prefix_breaks(requests: list[dict[str, Any]]) -> list[tuple[int, str]]: + """(C17) Indices i where request i is NOT a byte-identical extension of request i-1. + + Byte-stability covers the tools array, the system prompt and every earlier message. + """ + breaks = [] + for i in range(1, len(requests)): + prev, cur = requests[i - 1], requests[i] + if canon(prev.get("tools")) != canon(cur.get("tools")): + breaks.append((i, "tools array changed: " + tools_diff(prev.get("tools"), cur.get("tools")))) + continue + pm, cm = prev["messages"], cur["messages"] + if len(cm) < len(pm): + breaks.append((i, f"message list shrank {len(pm)} -> {len(cm)}")) + continue + for j, (a, b) in enumerate(zip(pm, cm)): + ca, cb = canon(a), canon(b) + if ca != cb: + breaks.append((i, f"messages[{j}] ({a.get('role')}) changed; {first_divergence(ca, cb)}")) + break + return breaks + + +def tools_diff(a: Any, b: Any) -> str: + an = {t["function"]["name"]: canon(t) for t in a or ()} + bn = {t["function"]["name"]: canon(t) for t in b or ()} + if list(an) != list(bn): + return f"names/order differ: -{sorted(set(an) - set(bn))} +{sorted(set(bn) - set(an))}" + for name in an: + if an[name] != bn[name]: + return f"tool {name!r}: {first_divergence(an[name], bn[name])}" + return "?" + + +def assert_no_loss(ever_sent: dict[tuple, str], display: list[dict], model: list[dict], where: str) -> None: + """Loss-free persistence: every message the model was ever sent is still on disk exactly once in + the display history, and every message still in the model's working context is there too. + Compaction may move a message out of the model view (summarized) but never out of display.""" + shown = [identity(m) for m in display if m.get("role") != "system" and not is_summary(m)] + counts: dict[tuple, int] = {} + for k in shown: + counts[k] = counts.get(k, 0) + 1 + dup = [short(k) for k, c in counts.items() if c > 1] + assert not dup, f"{where}: display history shows messages twice: {dup}" + missing = [label for k, label in ever_sent.items() if k not in counts] + assert not missing, (f"{where}: messages the model saw are gone from the durable transcript: {missing}\n" + f" display now: {[short(k) for k in shown]}") + model_missing = [short(identity(m)) for m in model + if m.get("role") != "system" and not is_summary(m) and identity(m) not in counts] + assert not model_missing, f"{where}: model-view rows absent from display: {model_missing}" + + +def assert_inputs_shown_once(inputs: list[str], display: list[dict], model: list[dict], where: str) -> None: + """Every text the user typed is on screen exactly once after a resume, and at most once in + what the model is fed (compaction may summarize it away, never duplicate or merge-copy it).""" + def hits(msgs: list[dict], text: str) -> int: + return sum(_text(m.get("content") or "").count(text) for m in msgs if m.get("role") == "user") + for text in inputs: + shown, fed = hits(display, text), hits(model, text) + assert shown == 1, f"{where}: user input {text[:60]!r} shown {shown}x in the resumed display history" + assert fed <= 1, f"{where}: user input {text[:60]!r} replayed {fed}x to the model" + + +def billed_usage(records: list[dict[str, Any]], kind: str = "main") -> dict[str, int]: + """What the fake provider reported for completed requests of ``kind``.""" + tot = {"calls": 0, "prompt": 0, "completion": 0, "cached": 0} + for r in records: + u = r.get("usage") + if r["kind"] != kind or not u: + continue + tot["calls"] += 1 + tot["prompt"] += u.get("prompt_tokens", 0) + tot["completion"] += u.get("completion_tokens", 0) + tot["cached"] += (u.get("prompt_tokens_details") or {}).get("cached_tokens", 0) + return tot + + +def assert_usage_matches(hermes_home: Path, sids: list[str], records: list[dict[str, Any]], where: str) -> None: + """(f) state.db usage equals what the provider billed: main-loop totals on ``sessions`` and in + the per-model ledger, no double count, no drop; cache reads are split out of input tokens.""" + want = billed_usage(records, "main") + marks = ",".join("?" * len(sids)) + with db_connect(hermes_home) as con: + s = con.execute( + f"SELECT sum(input_tokens), sum(output_tokens), sum(cache_read_tokens), sum(api_call_count) " + f"FROM sessions WHERE id IN ({marks})", sids).fetchone() + u = con.execute( + f"SELECT sum(input_tokens), sum(output_tokens), sum(cache_read_tokens), sum(api_call_count) " + f"FROM session_model_usage WHERE session_id IN ({marks}) AND task=''", sids).fetchone() + expect = (want["prompt"] - want["cached"], want["completion"], want["cached"], want["calls"]) + got_s = tuple(int(x or 0) for x in s) + got_u = tuple(int(x or 0) for x in u) + cols = "(input, output, cache_read, api_calls)" + assert got_s == expect, f"{where}: sessions usage {cols} {got_s} != provider-billed {expect}" + assert got_u == expect, f"{where}: session_model_usage main rows {cols} {got_u} != provider-billed {expect}" + + +# --------------------------------------------------------------------------------------------- +# subprocess surfaces + + +def child_env(home: Path, hermes_home: Path, workdir: Path) -> dict[str, str]: + env = {k: os.environ[k] for k in _ENV_ALLOW if k in os.environ} + env.update( + HOME=str(home), HERMES_HOME=str(hermes_home), PWD=str(workdir), + HERMES_TEST_ISOLATION=str(hermes_home), + # The child's state.db IS this test's sandbox db; the live-DB guard would refuse it + # because a pytest ancestor is present. + HERMES_STATE_DB_GUARD_BYPASS="1", + PYTHONPATH=str(REPO_ROOT), PYTHONUNBUFFERED="1", NO_COLOR="1", TMPDIR=str(home), + TZ="UTC", LANG="C.UTF-8", PYTHONHASHSEED="0", + ) + return env + + +class Spawned: + """PIDs this test started; only these are ever signalled.""" + + def __init__(self) -> None: + self.procs: list[subprocess.Popen] = [] + + def add(self, p: subprocess.Popen) -> subprocess.Popen: + self.procs.append(p) + return p + + def reap(self) -> list[int]: + leaked = [] + for p in self.procs: + if p.poll() is None: + leaked.append(p.pid) + p.kill() + p.wait(timeout=10) + return leaked + + +_SID_RE = re.compile(r"^session_id:\s*(\S+)\s*$", re.M) + + +def run_oneshot(env: dict[str, str], cwd: Path, prompt: str, spawned: Spawned, *, + resume: str | None = None, timeout: float = 180.0) -> tuple[str, str]: + """``hermes chat -q PROMPT -Q [--resume SID]`` in a fresh process -> (stdout, durable sid).""" + cmd = [sys.executable, "-m", "hermes_cli.main", "chat", "-q", prompt, "-Q"] + if resume: + cmd += ["--resume", resume] + p = spawned.add(subprocess.Popen(cmd, cwd=str(cwd), env={**env, "PWD": str(cwd)}, stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, encoding="utf-8")) + try: + out, err = p.communicate(timeout=timeout) + except subprocess.TimeoutExpired: + p.kill() + out, err = p.communicate(timeout=10) + raise AssertionError(f"oneshot did not finish in {timeout}s; stderr tail: {err[-2000:]}") + assert p.returncode == 0, f"oneshot rc={p.returncode}; stderr tail: {err[-3000:]}" + sids = _SID_RE.findall(err) + assert sids, f"oneshot printed no session_id line; stderr tail: {err[-2000:]}" + return out, sids[-1] + + +class GatewayError(RuntimeError): + pass + + +class TuiGateway: + """The real ``python -m tui_gateway.entry`` stdio JSON-RPC server (what the TUI spawns).""" + + def __init__(self, env: dict[str, str], cwd: Path, spawned: Spawned, *, ready_timeout: float = 120.0): + self.env, self.cwd, self.spawned, self.ready_timeout = env, str(cwd), spawned, ready_timeout + self.proc: subprocess.Popen | None = None + self.events: list[dict] = [] + self.stderr_lines: list[str] = [] + self._pending: dict[int, queue.Queue] = {} + self._cond = threading.Condition() + self._next_id = 0 + self._wlock = threading.Lock() + self.stored: dict[str, str] = {} + + def __enter__(self) -> "TuiGateway": + self.proc = self.spawned.add(subprocess.Popen( + [sys.executable, "-m", "tui_gateway.entry"], cwd=self.cwd, env={**self.env, "PWD": self.cwd}, + stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, + text=True, encoding="utf-8", bufsize=1)) + threading.Thread(target=self._read_stdout, daemon=True).start() + threading.Thread(target=self._read_stderr, daemon=True).start() + self.wait_event("gateway.ready", timeout=self.ready_timeout) + return self + + def __exit__(self, *_exc: object) -> None: + self.close() + + def close(self, timeout: float = 30.0) -> None: + p = self.proc + if p is None or p.poll() is not None: + return + try: + p.stdin.close() # EOF -> orderly shutdown + except OSError: + pass + try: + p.wait(timeout=timeout) + except subprocess.TimeoutExpired: + p.kill() + p.wait(timeout=10) + + def _read_stdout(self) -> None: + for raw in self.proc.stdout: + line = raw.strip() + if not line: + continue + try: + frame = json.loads(line) + except json.JSONDecodeError: + continue + if frame.get("method") == "event": + with self._cond: + self.events.append(frame.get("params") or {}) + self._cond.notify_all() + elif "method" in frame and "id" in frame: # server->client request: decline + self._write({"jsonrpc": "2.0", "id": frame["id"], + "error": {"code": -32601, "message": "test client declines server requests"}}) + elif "id" in frame and frame["id"] in self._pending: + self._pending[frame["id"]].put(frame) + with self._cond: + self._cond.notify_all() + + def _read_stderr(self) -> None: + for raw in self.proc.stderr: + self.stderr_lines.append(raw.rstrip("\n")) + + def _write(self, frame: dict) -> None: + with self._wlock: + self.proc.stdin.write(json.dumps(frame) + "\n") + self.proc.stdin.flush() + + def call(self, method: str, params: dict | None = None, timeout: float = 120.0) -> Any: + self._next_id += 1 + rid = self._next_id + q: queue.Queue = queue.Queue() + self._pending[rid] = q + try: + self._write({"jsonrpc": "2.0", "id": rid, "method": method, "params": params or {}}) + try: + frame = q.get(timeout=timeout) + except queue.Empty: + raise TimeoutError(f"{method}: no reply in {timeout}s; stderr tail={self.stderr_lines[-8:]}") + finally: + self._pending.pop(rid, None) + if "error" in frame: + raise GatewayError(f"{method}: {frame['error']}") + return frame.get("result") + + def mark(self) -> int: + with self._cond: + return len(self.events) + + def wait_event(self, etype: str, sid: str | None = None, *, start: int = 0, timeout: float = 120.0) -> dict: + deadline = time.monotonic() + timeout + with self._cond: + while True: + for ev in self.events[start:]: + if ev.get("type") == etype and (sid is None or ev.get("session_id") == sid): + return ev + if self.proc.poll() is not None: + raise RuntimeError(f"gateway exited rc={self.proc.returncode} waiting for {etype}; " + f"stderr tail={self.stderr_lines[-10:]}") + left = deadline - time.monotonic() + if left <= 0: + raise TimeoutError(f"no {etype} in {timeout}s; recent={[e.get('type') for e in self.events[-15:]]}") + self._cond.wait(min(left, 0.5)) + + def create(self) -> str: + res = self.call("session.create", {"cwd": self.cwd}) + self.stored[res["session_id"]] = res["stored_session_id"] + return res["session_id"] + + def resume(self, stored_sid: str) -> str: + res = self.call("session.resume", {"session_id": stored_sid}) + self.stored[res["session_id"]] = res.get("stored_session_id") or stored_sid + return res["session_id"] + + def call_when_idle(self, method: str, params: dict, timeout: float = 180.0) -> Any: + """Idle-gated methods answer 4009 while the previous turn is still finalizing (after + ``message.complete``); retry until the deadline instead of sleeping a fixed time.""" + deadline = time.monotonic() + timeout + while True: + try: + return self.call(method, params, timeout=timeout) + except GatewayError as exc: + if "4009" not in str(exc) or time.monotonic() > deadline: + raise + time.sleep(0.2) + + def submit(self, sid: str, text: str, timeout: float = 120.0) -> dict: + start = self.mark() + self.call("prompt.submit", {"session_id": sid, "text": text}, timeout=timeout) + return self.wait_event("message.complete", sid, start=start, timeout=timeout).get("payload") or {} + + +# --------------------------------------------------------------------------------------------- +# scripted model + + +class Script: + """Main-turn responder: pops scripted actions; an action may be a callable run AT request + arrival (so /steer and interrupt land while the request is genuinely in flight).""" + + def __init__(self) -> None: + self.actions: list[Any] = [] + self.n = 0 + self.session: Any = None + self.steered: list[str] = [] + + def __call__(self, record: dict[str, Any]) -> Any: + self.n += 1 + if self.actions: + act = self.actions.pop(0) + return act(record) if callable(act) else act + # Unique text + varying usage per answer, so a duplicated row or a double-counted + # request is always distinguishable. + from tests.fakes.fake_llm_provider import Text + + return Text(f"answer #{self.n}", prompt_tokens=900 + 13 * self.n, completion_tokens=5 + self.n, + cached_tokens=400 + self.n) + + +def big(label: str, n: int = 12000) -> str: + """A user message large enough that summarizing a few of them genuinely shrinks the context.""" + body = " ".join(f"{label}-{i}" for i in range(n // (len(label) + 4))) + return f"{label}: {body}"[:n] + + +# --------------------------------------------------------------------------------------------- +# in-process surface + per-step ledger + + +class InProcessSession: + """A real ``AIAgent`` + ``SessionDB`` driven the way the classic CLI drives it: the surface + owns ``history`` and hands it back each turn; ``/compress`` goes through ``compress_now``.""" + + def __init__(self, base_url: str, hermes_home: Path, sid: str, *, platform: str = "cli") -> None: + from hermes_cli.config import load_config + from hermes_cli.tools_config import _get_platform_tools + from hermes_state import SessionDB + from run_agent import AIAgent + + self.db = SessionDB(db_path=hermes_home / "state.db") + # The CLI entrypoints enable the platform's configured toolsets; a bare AIAgent would + # advertise a different tools array than every real surface resuming this session. + toolsets = sorted(_get_platform_tools(load_config(), platform)) + self.agent = AIAgent(provider="custom", base_url=base_url, api_key="sk-fake-e2e", model="fake-model", + session_db=self.db, session_id=sid, quiet_mode=True, platform=platform, + enabled_toolsets=toolsets) + self.history: list[dict[str, Any]] = [] + + @property + def sid(self) -> str: + return self.agent.session_id + + def turn(self, text: str) -> dict[str, Any]: + result = self.agent.run_conversation(text, conversation_history=self.history, task_id=self.sid) + self.history = result.get("messages") or self.history + return result + + def compress(self, args: str = "") -> Any: + """Mirror ``hermes_cli.cli_session_mixin`` ``/compress``: install ``after_messages``, follow a + rotated session id, re-flush the handoff on rotation, finalize the engine notification.""" + from agent.conversation_compression import finalize_context_engine_compression_notification + from agent.conversation_compression_manual import compress_now, parse_compress_args + + before_sid = self.sid + result = compress_now(self.agent, self.history, parse_compress_args(args), task_id=before_sid) + if result.status != "compressed": + return result + self.history = result.after_messages + if self.sid != before_sid: + self.agent._flush_messages_to_session_db(self.history, None) + finalize_context_engine_compression_notification(self.agent, committed=True) + return result + + def close(self) -> None: + try: + self.agent.close() + finally: + self.db.close() + + +class Ledger: + """Cross-checks request stream, model view, display view, row counts and usage after each step.""" + + def __init__(self, srv: Any, hermes_home: Path, *, check_inputs: bool = True) -> None: + self.srv, self.home, self.check_inputs = srv, hermes_home, check_inputs + self.ever: dict[tuple, str] = {} + self.inputs: list[str] = [] + self.model_view: list[dict] | None = None + self.counts: dict[str, int] | None = None + self.compaction_request_idx: list[int] = [] + self._seen_main = 0 + + def main(self) -> list[dict[str, Any]]: + return self.srv.main_requests() + + def step(self, sid: str, label: str, *, compaction: bool = False, turn: bool = True) -> None: + reqs = self.main() + new = reqs[self._seen_main:] + where = f"after step {label!r}" + model, display = views(self.home, sid) + if turn: + assert new, f"{where}: the turn issued no model request" + first = new[0] + if compaction: + # Pre-turn compaction rewrites what this request carries; a post-turn pass (micro) + # rewrites the durable view after it. Either way the break lands on this request or + # the next turn's first request; both are declared. + self.compaction_request_idx += [self._seen_main, len(reqs)] + else: + if self.model_view is not None: + assert_replay_equals_persisted(first, self.model_view, where) + sent = model_payload(first["messages"]) + persisted = model_payload(model) + assert persisted[:len(sent)] == sent, ( + f"{where}: the durable model view does not start with what this turn sent\n" + + _list_diff(sent, persisted[:len(sent)], "sent", "persisted")) + elif compaction: + self.compaction_request_idx.append(len(reqs)) + for r in new: + for m in r["messages"]: + if m.get("role") != "system" and not is_summary(m): + self.ever.setdefault(identity(m), short(identity(m))) + for m in model: + if m.get("role") != "system" and not is_summary(m): + self.ever.setdefault(identity(m), short(identity(m))) + assert_exactly_once(model, f"{where} (model view)") + assert_rows_exactly_once(self.home, sid, where) + assert_no_loss(self.ever, display, model, where) + if self.check_inputs: + assert_inputs_shown_once(self.inputs, display, model, where) + counts = row_counts(self.home, sid) + if self.counts is not None and sid == self._sid: + assert counts["total"] >= self.counts["total"], f"{where}: durable rows were deleted {self.counts} -> {counts}" + if not compaction: + assert counts["active"] >= self.counts["active"], ( + f"{where}: active rows shrank outside compaction {self.counts} -> {counts}") + self.counts, self._sid = counts, sid + self.model_view = model + self._seen_main = len(reqs) + + _sid: str | None = None diff --git a/tests/e2e/core/history/test_transcript_ledger.py b/tests/e2e/core/history/test_transcript_ledger.py new file mode 100644 index 0000000000..c4d3797aa2 --- /dev/null +++ b/tests/e2e/core/history/test_transcript_ledger.py @@ -0,0 +1,291 @@ +"""C3 transcript ledger: what is persisted == what is replayed == what a resume shows, per step. + +A real ``AIAgent`` + ``SessionDB`` runs scripted multi-turn conversations against the recording +fake provider. After EVERY step (turn, tool batch, /steer, interrupt, stream drop, /compress, +auto-compaction) the ledger asserts: + +(a) exactly-once: no logical message appears twice in the model view that resume replays; +(b) the first request of turn N+1 replays exactly the persisted model view after turn N plus the + new user message (a compaction step is the only sanctioned rewrite), and the durable view + starts with what the turn sent (nothing sent to the model is left unpersisted); +(c) durable row counts never shrink, active rows only shrink at compaction, and every message + the model ever saw is still in the display history exactly once (compaction archives, it + never drops or duplicates — the kept window survives); +(f) sessions + session_model_usage totals equal what the provider billed; +and at the end of every scenario a FRESH ``hermes chat --resume`` process replays the persisted +history as a byte-identical request prefix (d), and the whole request stream only breaks its +prefix at the declared compaction boundaries (C17). +""" + +from __future__ import annotations + +import os +from pathlib import Path +from typing import Any, Callable + +import pytest + +from tests.e2e.core.history._helpers import ( + NO_BACKGROUND_REVIEW, + Script, + big, + OFFLINE_CONFIG, + InProcessSession, + Ledger, + assert_inputs_shown_once, + is_summary, + lineage, + Spawned, + assert_replay_equals_persisted, + assert_usage_matches, + canon, + child_env, + first_divergence, + integrity_ok, + prefix_breaks, + run_oneshot, + views, +) +from tests.fakes.fake_llm_provider import ( + DropMidStream, + Error, + FakeLLMServer, + Hang, + Text, + ToolCall, + write_hermes_home, +) + +# --------------------------------------------------------------------------------------------- +# scripted model + + +Step = tuple[str, Any] + + +def bulky(*labels: str) -> list[Step]: + """Turns whose ANSWERS are large: summarizing them genuinely shrinks the context (the lean tail + keeps user messages verbatim, so bulk in user text would not compress).""" + steps: list[Step] = [] + for label in labels: + steps += [("script", [Text(big(f"answer-{label}"))]), ("turn", f"tell me about {label}")] + return steps + + +def _steer_then(tool: ToolCall, text: str, script: Script) -> Callable[[dict], Any]: + def act(_record: dict) -> Any: + assert script.session is not None + assert script.session.agent.steer(text) + script.steered.append(text) + return tool + return act + + +def _interrupt_during_request(script: Script) -> Callable[[dict], Any]: + def act(_record: dict) -> Any: + assert script.session is not None + import threading + + threading.Thread(target=script.session.agent.interrupt, daemon=True).start() + return Hang(seconds=60) + return act + + +def scenario(name: str, script: Script) -> list[Step]: + """Steps: ("turn", text) / ("compress", args) / ("script", [actions for the next turn]).""" + s = script + if name == "plain": + return [("turn", "hello there"), + ("script", [Text("thinking out loud answer", reasoning="private chain of thought")]), + ("turn", "second question"), ("turn", "third question"), ("turn", "fourth question")] + if name == "tool_batches": + return [("script", [ToolCall("terminal", {"command": "echo single-1"}), Text("single done")]), + ("turn", "run one tool"), + ("script", [ToolCall("terminal", {"command": "echo par-a"}, + parallel=[("terminal", {"command": "echo par-b"}), + ("terminal", {"command": "echo par-c"})]), + ToolCall("terminal", {"command": "echo chained"}), Text("batch done")]), + ("turn", "run a parallel batch then one more"), + ("turn", "and a plain follow-up")] + if name == "steer": + return [("turn", "warm up"), + ("script", [_steer_then(ToolCall("terminal", {"command": "echo steered-tool"}), + "also mention the steer marker", s), Text("steer seen")]), + ("turn", "do a tool while I steer"), + ("turn", "after the steer")] + if name == "interrupt": + return [("turn", "warm up"), + ("script", [_interrupt_during_request(s)]), + ("turn", "this request gets interrupted"), + ("turn", "the follow-up after the interrupt"), + ("turn", "one more")] + if name == "stream_faults": + return [("turn", "warm up"), + ("script", [DropMidStream("a partial reply that the network cut", after_chars=18), + Text("recovered after the drop")]), + ("turn", "stream this and drop"), + ("script", [Error(500, "scripted outage", retry_after=0.2), Text("recovered after a 500")]), + ("turn", "this one hits a 500 first"), + ("turn", "and a clean one")] + if name == "manual_compress": + return [*bulky("alpha", "bravo", "charlie", "delta", "echo", "foxtrot"), ("compress", ""), + ("turn", "first turn past full compress"), ("turn", "next turn past full compress")] + if name == "compress_here": + return [*bulky("golf", "hotel", "india", "juliet", "kilo", "lima"), ("compress", "here 1"), + ("turn", "first turn past compress here"), ("turn", "next turn past compress here")] + if name == "auto_compaction": + # The provider reports a prompt over the compaction threshold; the NEXT turn must compact + # before it is sent, and nothing the user saw may be lost from the durable transcript. + return [*bulky("mike", "november", "oscar", "papa", "quebec"), + ("script", [Text("reply under pressure", prompt_tokens=120_000, completion_tokens=9)]), + ("turn", "this reply reports context pressure"), + ("turn", "the turn that must compact first"), ("turn", "a turn past auto compaction")] + if name == "micro_compaction": + # Opt-in rolling micro-compaction folds the oldest exchange (tool rows included) into a + # summary after every turn; summarized rows must stay in the durable archive. + steps: list[Step] = [] + for label in ("romeo", "sierra", "tango", "uniform", "victor", "whiskey"): + steps += [("script", [ToolCall("terminal", {"command": f"echo tool-{label}"}), + Text(big(f"answer-{label}", 6000))]), + ("turn", f"run a tool about {label}")] + return steps + raise AssertionError(name) + + +SCENARIO_CONFIG = { + "micro_compaction": "compression:\n micro_compact: true\n micro_compact_every_n_turns: 1\n", +} + +SCENARIOS = ["plain", "tool_batches", "steer", "interrupt", "stream_faults", + "manual_compress", "compress_here", "auto_compaction", "micro_compaction"] +COMPACTING = {"manual_compress", "compress_here", "auto_compaction", "micro_compaction"} + + +# --------------------------------------------------------------------------------------------- + + +@pytest.fixture() +def world(tmp_path, monkeypatch): + home = tmp_path / "home" + home.mkdir() + # The hermetic conftest's per-test HERMES_HOME (a tmp dir): the in-process live-system guard + # refuses a /.hermes layout, and the children must share this very state.db. + hermes_home = Path(os.environ["HERMES_HOME"]) + workdir = tmp_path / "work" + workdir.mkdir() + monkeypatch.setenv("HOME", str(home)) + monkeypatch.chdir(workdir) + script = Script() + spawned = Spawned() + with FakeLLMServer(script) as srv: + write_hermes_home(hermes_home, srv.base_url, extra_config=OFFLINE_CONFIG + NO_BACKGROUND_REVIEW) + monkeypatch.setenv("OPENAI_API_KEY", "sk-fake-e2e") + yield {"srv": srv, "script": script, "home": home, "hermes_home": hermes_home, + "workdir": workdir, "spawned": spawned} + leaked = spawned.reap() + assert not leaked, f"subprocesses outlived the test: {leaked}" + + +@pytest.mark.parametrize("name", SCENARIOS) +def test_transcript_ledger(world, name): + srv, script, hermes_home = world["srv"], world["script"], world["hermes_home"] + if name in SCENARIO_CONFIG: + write_hermes_home(hermes_home, srv.base_url, + extra_config=OFFLINE_CONFIG + NO_BACKGROUND_REVIEW + SCENARIO_CONFIG[name]) + typed = [arg for kind, arg in scenario(name, Script()) if kind == "turn"] + assert not [(a, b) for a in typed for b in typed if a != b and a in b], "scenario inputs must be unique" + session = InProcessSession(srv.base_url, hermes_home, f"ledger-{name}") + script.session = session + # Micro-compaction's merged-user-row display duplication is tracked separately (strict xfail + # below) so the scenario still guards every other invariant. + ledger = Ledger(srv, hermes_home, check_inputs=name != "micro_compaction") + try: + for i, (kind, arg) in enumerate(scenario(name, script)): + label = f"{i}:{kind}:{str(arg)[:24]}" + if kind == "script": + script.actions.extend(arg) + elif kind == "turn": + ledger.inputs.append(arg) + result = session.turn(arg) + ledger.inputs += [x for x in script.steered if x not in ledger.inputs] + assert result.get("final_response") is not None or name == "interrupt", f"{label}: {result}" + compaction = name == "micro_compaction" or ( + name == "auto_compaction" and arg == "the turn that must compact first") + ledger.step(session.sid, label, compaction=compaction) + elif kind == "compress": + res = session.compress(arg) + flags = {k: res.summary.get(k) for k in ("noop", "aborted", "fallback_used", "refused_would_grow")} + assert res.status == "compressed" and not any(flags.values()), ( + f"{label}: /compress {arg!r} did not compact: {res.status} {flags} " + f"{res.before_tokens}->{res.after_tokens} tokens") + ledger.step(session.sid, label, compaction=True, turn=False) + sid = session.sid + finally: + session.close() + assert not script.actions, f"scripted responses never consumed: {script.actions}" + + # C17: the request stream only breaks its prefix at declared compaction boundaries, and a + # one-shot compaction breaks it exactly once. + main = srv.main_requests() + breaks = prefix_breaks(main) + allowed = set(ledger.compaction_request_idx) + unexpected = [(i, why) for i, why in breaks if i not in allowed] + assert not unexpected, "prompt-cache prefix broke outside compaction:\n" + "\n".join( + f" request {i}: {why}" for i, why in unexpected) + if name in COMPACTING: + assert len(breaks) == (len(breaks) if name == "micro_compaction" else 1) and breaks, ( + f"expected the compaction boundary to break the prefix (exactly once for one-shot compaction), " + f"got {breaks} (declared at {sorted(allowed)})") + model, _ = views(hermes_home, sid) + assert any(is_summary(m) for m in model), "compaction left no summary in the model view" + + # (f) usage: what the provider billed == what state.db accounts for the whole lineage. + assert_usage_matches(hermes_home, lineage(hermes_home, sid), srv.requests, f"scenario {name}") + integrity_ok(hermes_home) + + # (d) resume in a FRESH `hermes chat --resume` process: its first request replays the persisted + # model view and is a byte-identical extension of the last in-process request. Messages only: + # the tools array of an in-process AIAgent is a harness construction choice; the cross-surface + # tools comparison lives in test_prefix_stability.py (real surfaces only). + persisted, _ = views(hermes_home, sid) + n_before = len(main) + env = child_env(world["home"], hermes_home, world["workdir"]) + _out, resumed_sid = run_oneshot(env, world["workdir"], f"resume check for {name}", world["spawned"], resume=sid) + assert resumed_sid == sid, f"--resume {sid} continued a different session {resumed_sid}" + after = srv.main_requests() + assert len(after) > n_before, "the resumed process sent no request" + first, last = after[n_before], main[-1] + assert_replay_equals_persisted(first, persisted, f"fresh-process resume of {name}") + if name != "micro_compaction": # a post-turn micro pass legitimately rewrote history after `last` + for j, (a, b) in enumerate(zip(last["messages"], first["messages"])): + assert canon(a) == canon(b), ( + f"fresh-process resume changed messages[{j}] ({a.get('role')}): " + f"{first_divergence(canon(a), canon(b))}") + assert len(first["messages"]) > len(last["messages"]) + ledger.step(sid, "fresh-process resume", compaction=name == "micro_compaction") + assert_usage_matches(hermes_home, lineage(hermes_home, sid), srv.requests, f"scenario {name} after resume") + + +@pytest.mark.xfail(strict=True, reason=( + "PRODUCTION BUG (opt-in micro-compaction): when a newer micro marker supersedes the old one, " + "_merge_adjacent_user_turns persists a merged 'A\\n\\nB' user row while the originals stay " + "compacted=1, so the resumed display history shows both user inputs twice. Flip to a plain test " + "once the display projection (or the merge) stops duplicating them.")) +def test_micro_compaction_resumed_display_shows_each_input_once(world): + srv, script, hermes_home = world["srv"], world["script"], world["hermes_home"] + write_hermes_home(hermes_home, srv.base_url, + extra_config=OFFLINE_CONFIG + NO_BACKGROUND_REVIEW + SCENARIO_CONFIG["micro_compaction"]) + session = InProcessSession(srv.base_url, hermes_home, "ledger-micro-display") + inputs: list[str] = [] + try: + for kind, arg in scenario("micro_compaction", script): + if kind == "script": + script.actions.extend(arg) + else: + inputs.append(arg) + session.turn(arg) + sid = session.sid + finally: + session.close() + model, display = views(hermes_home, sid) + assert_inputs_shown_once(inputs, display, model, "after six micro-compacted turns") diff --git a/tests/fakes/fake_llm_provider.py b/tests/fakes/fake_llm_provider.py index 4cafadab3b..50e5ad3deb 100644 --- a/tests/fakes/fake_llm_provider.py +++ b/tests/fakes/fake_llm_provider.py @@ -261,10 +261,11 @@ def _handler_for(server: FakeLLMServer) -> type[BaseHTTPRequestHandler]: resp = server._next_main(record) if kind == "main" else server._aux(record) record["response"] = type(resp).__name__ prompt_tokens = server.prompt_tokens_fn(body) if server.prompt_tokens_fn else None - self._respond(resp, bool(body.get("stream")), prompt_tokens) + self._respond(resp, bool(body.get("stream")), prompt_tokens, record) # response rendering - def _respond(self, resp: Response, stream: bool, prompt_tokens: int | None = None) -> None: + def _respond(self, resp: Response, stream: bool, prompt_tokens: int | None = None, + record: dict[str, Any] | None = None) -> None: if isinstance(resp, Error): headers = {"Retry-After": str(resp.retry_after)} if resp.retry_after is not None else {} self._send_json(resp.status, {"error": {"message": resp.message, "type": "server_error"}}, headers) @@ -295,6 +296,10 @@ def _handler_for(server: FakeLLMServer) -> type[BaseHTTPRequestHandler]: self.close_connection = True return message, finish, usage = _message_for(resp, server, prompt_tokens) + # What the provider billed for this request, so usage/cost accounting can be checked + # against state.db (faulted requests never get a ``usage`` key). + if record is not None: + record["usage"] = usage if not stream: self._send_json(200, { "id": "chatcmpl-fake", "object": "chat.completion", "created": int(time.time()), From 51a1205be0bb0c45478fdf68252f0fbcf0d45994 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:57:53 -0700 Subject: [PATCH 012/104] test: C17 request-prefix byte-stability + usage across real processes Class C17 (prompt-cache prefix byte-stability + usage/cost accounting). One durable session is driven for 10+ turns through FRESH processes of the real entrypoints (tui_gateway stdio JSON-RPC server, 'hermes chat -q --resume' oneshot), flipping cwd between hops, with a parallel tool batch and a manual session.compress. Asserts every main request is a byte-identical extension of the previous one (tools array, system prompt, earlier messages) with exactly one sanctioned break at the compaction, every process boundary replays the persisted model view, storage rows are exactly-once and monotonic, and sessions / session_model_usage equal what the fake provider billed. tui_gateway_restarts went red on base (the gateway's manual /compress persisted a prompt with the process HOME as cwd; fixed in the previous commit). surface_switch is a strict xfail documenting live TUI<->oneshot tools-array divergence (tool_search catalog rebuilt per surface; pinned tools lose dynamic schema overrides; -q --resume prunes skill_manage). --- .../e2e/core/history/test_prefix_stability.py | 197 ++++++++++++++++++ 1 file changed, 197 insertions(+) create mode 100644 tests/e2e/core/history/test_prefix_stability.py diff --git a/tests/e2e/core/history/test_prefix_stability.py b/tests/e2e/core/history/test_prefix_stability.py new file mode 100644 index 0000000000..9e3476ffa3 --- /dev/null +++ b/tests/e2e/core/history/test_prefix_stability.py @@ -0,0 +1,197 @@ +"""C17 prompt-cache prefix byte-stability + usage accounting across REAL processes and surfaces. + +One durable session is driven for 10+ turns through fresh processes of the real entrypoints that +can continue it — the ``tui_gateway`` stdio JSON-RPC server (what the TUI/Desktop spawn) and the +``hermes chat -q --resume`` oneshot CLI — switching process and working directory between turns, +with a parallel tool batch and (gateway journeys) one manual ``session.compress`` on the way. + +Invariants over the fake provider's recorded request stream: + +* every main request is a byte-identical extension of the previous one — same tools array, same + system prompt, same earlier messages — across resume, process restart and cwd change; the ONLY + sanctioned break is the first request after the compaction, and it breaks exactly once; +* every process boundary replays exactly the persisted model view (nothing re-rendered); +* each logical message is persisted once (storage rows, not reader output); rows never shrink; +* ``sessions`` / ``session_model_usage`` equal what the provider billed, request by request. +""" + +from __future__ import annotations + +import os +from pathlib import Path + +import pytest + +from tests.e2e.core.history._helpers import ( + NO_BACKGROUND_REVIEW, + OFFLINE_CONFIG, + Script, + Spawned, + TuiGateway, + assert_replay_equals_persisted, + assert_rows_exactly_once, + assert_usage_matches, + big, + child_env, + integrity_ok, + lineage, + prefix_breaks, + row_counts, + run_oneshot, + views, +) +from tests.fakes.fake_llm_provider import FakeLLMServer, Text, ToolCall, write_hermes_home + +# A hop is (surface, cwd, turns); a turn is its prompt, "TOOLS:" (parallel tool batch +# first) or "/compress" (gateway only). Every hop is a FRESH process. +Hop = tuple[str, str, list[str]] + +JOURNEYS: dict[str, list[Hop]] = { + # Same surface, four gateway processes, two cwds, compaction mid-way, resume after it. + "tui_gateway_restarts": [ + ("gw", "a", ["gw1 first turn", "TOOLS:gw1 runs a parallel batch", "gw1 third turn"]), + ("gw", "b", ["gw2 resumed turn", "gw2 another turn"]), + ("gw", "a", ["gw3 turn before compress", "/compress", "gw3 turn after compress", "gw3 more"]), + ("gw", "b", ["gw4 resumed after compaction", "gw4 last turn"]), + ], + # Same surface, eight oneshot processes, cwd flips between them. + "oneshot_restarts": [ + ("cli", "a", ["cli1 first turn"]), + ("cli", "a", ["TOOLS:cli2 runs a parallel batch"]), + ("cli", "b", ["cli3 from another cwd"]), + ("cli", "a", ["cli4 back home"]), + ("cli", "b", ["TOOLS:cli5 tools away again"]), + ("cli", "a", ["cli6 turn"]), + ("cli", "b", ["cli7 turn"]), + ("cli", "a", ["cli8 last turn"]), + ], + # Surface switch on one session: TUI <-> oneshot, cwd change, compaction, resume after it. + "surface_switch": [ + ("gw", "a", ["sw1 first turn", "TOOLS:sw1 runs a parallel batch", "sw1 third turn"]), + ("cli", "a", ["sw2 oneshot turn"]), + ("cli", "b", ["sw3 oneshot from another cwd"]), + ("gw", "b", ["sw4 gateway resumes", "sw4 another turn"]), + ("cli", "a", ["sw5 oneshot turn"]), + ("gw", "a", ["sw6 before compress", "/compress", "sw6 after compress", "sw6 more"]), + ("cli", "b", ["sw7 oneshot after compaction"]), + ], +} + +KNOWN_BROKEN = { + "surface_switch": ( + "PRODUCTION BUGS (C17): the tools array differs per entrypoint for one durable session, so " + "every TUI<->oneshot hop is a full prompt-cache miss: (1) tool_search's deferred catalog is " + "rebuilt per process from that surface's toolsets (gateway defers the GUI-only project tool: " + "'6 additional tools' vs '5') while the session pin stores names only " + "(tools/tool_search.py, tools/mcp_tool_agent.py restore_agent_tool_prefix); (2) a pinned tool " + "re-materialized from the registry skips dynamic_schema_overrides, so skill_manage's " + "description changes (tools/mcp_tool_agent.py vs tools/registry.py get_definitions); " + "(3) `-q --resume` prunes skill_manage (agent/oneshot_footprint.py) whenever the stored " + "prompt is rebuilt, and persists the pruned pin."), +} + + +@pytest.fixture +def world(tmp_path): + home = tmp_path / "home" + home.mkdir() + hermes_home = Path(os.environ["HERMES_HOME"]) # hermetic per-test tmp dir from the conftest + cwds = {"a": tmp_path / "work-a", "b": tmp_path / "work-b"} + for d in cwds.values(): + d.mkdir() + script = Script() + spawned = Spawned() + with FakeLLMServer(script) as srv: + write_hermes_home(hermes_home, srv.base_url, extra_config=OFFLINE_CONFIG + NO_BACKGROUND_REVIEW) + yield {"srv": srv, "script": script, "home": home, "hermes_home": hermes_home, + "cwds": cwds, "spawned": spawned} + leaked = spawned.reap() + assert not leaked, f"subprocesses outlived the test: {leaked}" + + +def _queue_turn(script: Script, prompt: str) -> str: + """Script the model for one turn; returns the text the user types.""" + if prompt.startswith("TOOLS:"): + prompt = prompt[len("TOOLS:"):] + tag = prompt.split()[0] + script.actions.append(ToolCall("terminal", {"command": f"echo {tag}-a"}, + parallel=[("terminal", {"command": f"echo {tag}-b"})])) + # Bulky answers so the later /compress genuinely shrinks the context; distinct usage per + # answer so a double-counted or dropped request is visible. + n = len(prompt) + script.actions.append(Text(big(f"answer-{prompt.replace(' ', '-')}", 12000), + prompt_tokens=2000 + n, completion_tokens=40 + n, cached_tokens=1500)) + return prompt + + +def run_journey(world: dict, hops: list[Hop]) -> tuple[str, list[tuple[int, str]], int | None]: + """Drive the hops; returns (durable sid, [(first request idx, hop label)], compaction idx).""" + srv, script, home = world["srv"], world["script"], world["hermes_home"] + sid: str | None = None + openings: list[tuple[int, str]] = [] + compaction_idx: int | None = None + rows_total = 0 + for n, (surface, cwd_key, turns) in enumerate(hops, 1): + cwd = world["cwds"][cwd_key] + label = f"hop {n} ({surface} in work-{cwd_key})" + persisted = views(home, sid)[0] if sid else None + opened_at = len(srv.main_requests()) + env = child_env(world["home"], home, cwd) + if surface == "gw": + with TuiGateway(env, cwd, world["spawned"]) as gw: + live = gw.resume(sid) if sid else gw.create() + for turn in turns: + if turn == "/compress": + res = gw.call_when_idle("session.compress", {"session_id": live}) or {} + summary = res.get("summary") or {} + flags = {k: summary.get(k) for k in ("noop", "aborted", "refused_would_grow")} + assert res.get("status") == "compressed" and not any(flags.values()) and ( + (res.get("after_messages") or 0) < (res.get("before_messages") or 0)), ( + f"{label}: session.compress did not compact: " + f"{ {k: v for k, v in res.items() if k != 'messages'} }") + compaction_idx = len(srv.main_requests()) + continue + gw.submit(live, _queue_turn(script, turn)) + sid = sid or gw.stored[live] + else: + for turn in turns: + _out, got = run_oneshot(env, cwd, _queue_turn(script, turn), world["spawned"], resume=sid) + assert sid is None or got == sid, f"{label}: --resume {sid} continued {got}" + sid = got + main = srv.main_requests() + assert len(main) > opened_at, f"{label}: sent no model request" + if persisted is not None: + assert_replay_equals_persisted(main[opened_at], persisted, f"{label} opening request") + assert_rows_exactly_once(home, sid, label) + total = row_counts(home, sid)["total"] + assert total >= rows_total, f"{label}: durable rows shrank {rows_total} -> {total}" + rows_total = total + openings.append((opened_at, label)) + assert not script.actions, f"scripted responses never consumed: {script.actions}" + return sid, openings, compaction_idx + + +@pytest.mark.parametrize("journey", [ + pytest.param(name, marks=pytest.mark.xfail(strict=True, reason=KNOWN_BROKEN[name])) + if name in KNOWN_BROKEN else name + for name in JOURNEYS +]) +def test_request_prefix_is_byte_stable_across_processes(world, journey): + sid, openings, compaction_idx = run_journey(world, JOURNEYS[journey]) + srv, home = world["srv"], world["hermes_home"] + main = srv.main_requests() + assert len(main) >= 10 + + def where(i: int) -> str: + return f"request {i} (in {max((o for o in openings if o[0] <= i), default=(0, '?'))[1]})" + + breaks = prefix_breaks(main) + unexpected = [(i, why) for i, why in breaks if i != compaction_idx] + assert not unexpected, "prompt-cache prefix broke outside the compaction boundary:\n" + "\n".join( + f" {where(i)}: {why}" for i, why in unexpected) + if compaction_idx is not None: + assert [i for i, _ in breaks] == [compaction_idx], ( + f"the manual /compress must break the prefix exactly once, at request {compaction_idx}: {breaks}") + + assert_usage_matches(home, lineage(home, sid), srv.requests, f"after journey {journey}") + integrity_ok(home) From ae4c5a0874c7f8df6f9f723cf41a412d1ab96a2b Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:09:27 -0700 Subject: [PATCH 013/104] test: entrypoint parity matrix with a real stdio MCP server (C19, C15) One fixture HERMES_HOME (shell + plugin pre_llm_call hooks, AGENTS.md, a skill, a memory entry, a stdio MCP server that spawns a grandchild, the recording fake provider, a disabled toolset) and one scripted turn per entrypoint: hermes -z, hermes chat -q, tui_gateway stdio, hermes serve over the Desktop WS, GatewayRunner with a fake adapter, api_server, ACP stdio, cron run-now. The fake model calls the MCP canary tool, then answers. Same invariants on every surface, first turn after a cold start: context file / skill index / memory in the recorded system prompt, both hooks fired and injected, MCP tool offered and a REAL call returned the canary (the inverted stdio-liveness burst failed exactly here everywhere), documented toolset's feature tools present and the disabled toolset absent, answer delivered to the client, zero MCP server/grandchild survivors after the surface's normal shutdown. Catches the 'works in the CLI, missing in Desktop/gateway/ACP/cron' class; found #95577 (oneshot row). ACP's disabled_toolsets gap (#74582) is a strict known-red cell. --- tests/e2e/core/parity/__init__.py | 0 tests/e2e/core/parity/_boot_contract.py | 54 +++ tests/e2e/core/parity/_drive_acp.py | 149 ++++++ tests/e2e/core/parity/_drive_cli.py | 43 ++ tests/e2e/core/parity/_drive_cron.py | 70 +++ tests/e2e/core/parity/_drive_gateway.py | 246 ++++++++++ tests/e2e/core/parity/_drive_rpc.py | 272 +++++++++++ tests/e2e/core/parity/_gateway_child.py | 118 +++++ tests/e2e/core/parity/_helpers.py | 453 ++++++++++++++++++ tests/e2e/core/parity/desktop_ready_parse.mjs | 30 ++ tests/e2e/core/parity/fixture_mcp_server.py | 60 +++ .../e2e/core/parity/test_entrypoint_parity.py | 177 +++++++ 12 files changed, 1672 insertions(+) create mode 100644 tests/e2e/core/parity/__init__.py create mode 100644 tests/e2e/core/parity/_boot_contract.py create mode 100644 tests/e2e/core/parity/_drive_acp.py create mode 100644 tests/e2e/core/parity/_drive_cli.py create mode 100644 tests/e2e/core/parity/_drive_cron.py create mode 100644 tests/e2e/core/parity/_drive_gateway.py create mode 100644 tests/e2e/core/parity/_drive_rpc.py create mode 100644 tests/e2e/core/parity/_gateway_child.py create mode 100644 tests/e2e/core/parity/_helpers.py create mode 100644 tests/e2e/core/parity/desktop_ready_parse.mjs create mode 100644 tests/e2e/core/parity/fixture_mcp_server.py create mode 100644 tests/e2e/core/parity/test_entrypoint_parity.py diff --git a/tests/e2e/core/parity/__init__.py b/tests/e2e/core/parity/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/tests/e2e/core/parity/_boot_contract.py b/tests/e2e/core/parity/_boot_contract.py new file mode 100644 index 0000000000..f693e7973e --- /dev/null +++ b/tests/e2e/core/parity/_boot_contract.py @@ -0,0 +1,54 @@ +"""Bridge to the Desktop's real READY-sentinel parser (``apps/desktop/electron/backend-ready.ts``). + +There is no shared constant for the sentinel format, and regexing the TS source +would test a copy of the contract. Instead the captured backend stdout is fed +through the parser itself under Node (``desktop_ready_parse.mjs``), so the Python +side asserts exactly what the Desktop would conclude from those bytes. +""" + +from __future__ import annotations + +import json +import re +import shutil +import subprocess +from pathlib import Path + +import pytest + +from tests.e2e.core.parity._helpers import REPO_ROOT + +PARSER_TS = REPO_ROOT / "apps" / "desktop" / "electron" / "backend-ready.ts" +BRIDGE_MJS = Path(__file__).with_name("desktop_ready_parse.mjs") +# Native TypeScript type-stripping landed unflagged in Node 22.18 / 23.6. +_MIN_NODE = (22, 18) + + +def node_binary() -> str | None: + node = shutil.which("node") + if not node: + return None + out = subprocess.run([node, "--version"], capture_output=True, text=True, timeout=30, + stdin=subprocess.DEVNULL).stdout.strip() + m = re.match(r"v(\d+)\.(\d+)", out) + if not m or (int(m.group(1)), int(m.group(2))) < _MIN_NODE: + return None + return node + + +def require_node() -> str: + node = node_binary() + if node is None: + pytest.skip(f"needs node >= {'.'.join(map(str, _MIN_NODE))} to run the Desktop READY parser") + return node + + +def desktop_parse(stdout_bytes: str) -> dict: + """What ``waitForDashboardPort`` resolves/rejects with for this exact stdout stream.""" + proc = subprocess.run( + [require_node(), str(BRIDGE_MJS), str(PARSER_TS), "5000"], input=stdout_bytes, + capture_output=True, text=True, timeout=60, + ) + assert proc.returncode == 0, f"desktop parser bridge crashed: {proc.stderr[-2000:]}" + return json.loads(proc.stdout.strip().splitlines()[-1]) + diff --git a/tests/e2e/core/parity/_drive_acp.py b/tests/e2e/core/parity/_drive_acp.py new file mode 100644 index 0000000000..25bd77afef --- /dev/null +++ b/tests/e2e/core/parity/_drive_acp.py @@ -0,0 +1,149 @@ +"""Entrypoint driver for the parity matrix: the ACP stdio server (``hermes acp``). + +Follows the driver contract in ``_drive_cli``. The client side speaks +newline-delimited JSON-RPC 2.0 itself (no SDK client) so the driver controls +every byte and the shutdown path: ``initialize`` -> ``session/new`` (ACP's +documented cwd channel) -> ``session/prompt``, collecting +``agent_message_chunk`` text from ``session/update`` notifications, then stdin +EOF — the way an editor host closes the server. +""" + +from __future__ import annotations + +import json +import queue +import subprocess +import threading +import time +from typing import Any + +from tests.e2e.core.parity._helpers import TURN_TIMEOUT, DriveResult, ParityHome, hermes_argv, terminate +from tests.fakes.fake_llm_provider import FakeLLMServer + +ACP_TOOLSET = "hermes-acp" # acp_adapter/session.py::_expand_acp_enabled_toolsets default +EOF_EXIT_TIMEOUT = 30.0 + + +class _AcpClient: + """Minimal ACP client over a subprocess's stdio (one reader thread).""" + + def __init__(self, proc: subprocess.Popen) -> None: + self.proc = proc + self._next_id = 0 + self._inbox: queue.Queue[dict[str, Any] | None] = queue.Queue() + self.chunks: list[str] = [] + self.updates: list[dict[str, Any]] = [] + self.server_requests: list[str] = [] + self.stray_stdout: list[str] = [] + threading.Thread(target=self._pump, daemon=True).start() + + def _pump(self) -> None: + for line in self.proc.stdout: # type: ignore[union-attr] + line = line.strip() + if not line: + continue + try: + self._inbox.put(json.loads(line)) + except json.JSONDecodeError: + # stdout must stay pure JSON-RPC; keep evidence instead of crashing. + self.stray_stdout.append(line) + self._inbox.put(None) + + def _send(self, msg: dict[str, Any]) -> None: + self.proc.stdin.write(json.dumps(msg) + "\n") # type: ignore[union-attr] + self.proc.stdin.flush() # type: ignore[union-attr] + + def request(self, method: str, params: dict[str, Any], timeout: float) -> dict[str, Any]: + self._next_id += 1 + rid = self._next_id + self._send({"jsonrpc": "2.0", "id": rid, "method": method, "params": params}) + deadline = time.monotonic() + timeout + while True: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise AssertionError(f"ACP {method} got no response within {timeout}s") + try: + msg = self._inbox.get(timeout=remaining) + except queue.Empty: + continue + if msg is None: + raise AssertionError(f"ACP server closed stdout during {method} (rc={self.proc.poll()})") + if "method" in msg: + self._handle_incoming(msg) + elif msg.get("id") == rid: + if "error" in msg: + raise AssertionError(f"ACP {method} failed: {msg['error']}") + return msg.get("result") or {} + + def _handle_incoming(self, msg: dict[str, Any]) -> None: + method, params = msg["method"], msg.get("params") or {} + if "id" not in msg: # notification + if method == "session/update": + update = params.get("update") or {} + self.updates.append(update) + content = update.get("content") or {} + if update.get("sessionUpdate") == "agent_message_chunk" and content.get("type") == "text": + self.chunks.append(content.get("text", "")) + return + self.server_requests.append(method) + if method == "session/request_permission": + # An editor user clicking "allow once" (fall back to the first offered option). + options = params.get("options") or [] + pick = next((o for o in options if o.get("kind") == "allow_once"), options[0] if options else None) + outcome = ({"outcome": "selected", "optionId": pick["optionId"]} if pick + else {"outcome": "cancelled"}) + self._send({"jsonrpc": "2.0", "id": msg["id"], "result": {"outcome": outcome}}) + return + # We advertise no fs/terminal client capabilities, so anything else is unexpected. + self._send({"jsonrpc": "2.0", "id": msg["id"], + "error": {"code": -32601, "message": f"client does not implement {method}"}}) + + +def drive_acp(ph: ParityHome, srv: FakeLLMServer, prompt: str) -> DriveResult: + stderr_path = ph.root / "acp_stderr.log" + with open(stderr_path, "w", encoding="utf-8") as stderr_fh: + proc = subprocess.Popen( + hermes_argv("acp"), cwd=ph.project, env=ph.env(), stdin=subprocess.PIPE, + stdout=subprocess.PIPE, stderr=stderr_fh, text=True, bufsize=1, + ) + client = _AcpClient(proc) + graceful = True + stop_reason = None + try: + init = client.request("initialize", { + "protocolVersion": 1, + "clientCapabilities": {"fs": {"readTextFile": False, "writeTextFile": False}, "terminal": False}, + "clientInfo": {"name": "parity-suite", "version": "1.0"}, + }, timeout=60) + session = client.request("session/new", {"cwd": str(ph.project), "mcpServers": []}, + timeout=120) + session_id = session["sessionId"] + result = client.request("session/prompt", { + "sessionId": session_id, "prompt": [{"type": "text", "text": prompt}], + }, timeout=TURN_TIMEOUT) + stop_reason = result.get("stopReason") + finally: + # Normal host shutdown: close the pipe and let the server exit on EOF. + try: + proc.stdin.close() # type: ignore[union-attr] + except OSError: + pass + try: + proc.wait(timeout=EOF_EXIT_TIMEOUT) + except subprocess.TimeoutExpired: + graceful = False + terminate(proc) + return DriveResult( + final_text="".join(client.chunks), + toolset=ACP_TOOLSET, + cwd_channel="session/new cwd", + graceful_exit=graceful, + extra={ + "stop_reason": stop_reason, + "returncode": proc.returncode, + "agent_info": (init or {}).get("agentInfo"), + "server_requests": client.server_requests, + "stray_stdout": client.stray_stdout[:20], + "stderr_log": str(stderr_path), + }, + ) diff --git a/tests/e2e/core/parity/_drive_cli.py b/tests/e2e/core/parity/_drive_cli.py new file mode 100644 index 0000000000..1d87cd4665 --- /dev/null +++ b/tests/e2e/core/parity/_drive_cli.py @@ -0,0 +1,43 @@ +"""Entrypoint drivers for the parity matrix: CLI subprocess entrypoints. + +Driver contract (every ``_drive_*.py`` module follows it):: + + def drive(ph: ParityHome, srv: FakeLLMServer, prompt: str) -> DriveResult + +* spawn the REAL entrypoint with ``ph.env()`` (hermetic fake HOME, no real + credentials) — cwd ``ph.project`` unless the surface documents another cwd + channel (then use that channel and say so in ``DriveResult.cwd_channel``); +* run exactly ONE user turn with ``prompt`` against ``srv`` (the scripted + responder makes the model call the MCP canary tool, then answer + ``FINAL_ANSWER``); +* return what the surface delivered to ITS client in ``final_text``; +* stop the entrypoint through its NORMAL shutdown path (exit, stdin EOF, + SIGTERM, RPC) before returning; never SIGKILL except as a last resort after a + bounded graceful wait (and then report it via ``graceful_exit=False``). +""" + +from __future__ import annotations + +import subprocess + +from tests.e2e.core.parity._helpers import TURN_TIMEOUT, DriveResult, ParityHome, hermes_argv +from tests.fakes.fake_llm_provider import FakeLLMServer + + +def _run_cli(ph: ParityHome, *args: str) -> subprocess.CompletedProcess: + return subprocess.run( + hermes_argv(*args), cwd=ph.project, env=ph.env(), capture_output=True, text=True, + timeout=TURN_TIMEOUT, stdin=subprocess.DEVNULL, + ) + + +def drive_oneshot(ph: ParityHome, srv: FakeLLMServer, prompt: str) -> DriveResult: + proc = _run_cli(ph, "-z", prompt) + assert proc.returncode == 0, f"hermes -z exited {proc.returncode}: {proc.stderr[-2000:]}" + return DriveResult(final_text=proc.stdout.strip(), toolset="hermes-cli") + + +def drive_chat_q(ph: ParityHome, srv: FakeLLMServer, prompt: str) -> DriveResult: + proc = _run_cli(ph, "chat", "-q", prompt, "-Q") + assert proc.returncode == 0, f"hermes chat -q exited {proc.returncode}: {proc.stderr[-2000:]}" + return DriveResult(final_text=proc.stdout.strip(), toolset="hermes-cli") diff --git a/tests/e2e/core/parity/_drive_cron.py b/tests/e2e/core/parity/_drive_cron.py new file mode 100644 index 0000000000..83bf4faac2 --- /dev/null +++ b/tests/e2e/core/parity/_drive_cron.py @@ -0,0 +1,70 @@ +"""Entrypoint driver for the parity matrix: cron run-now (``hermes cron run ``). + +Follows the driver contract in ``_drive_cli``. ``hermes cron create`` writes the +job (``--deliver local``: no platform; ``--workdir``: cron's documented +per-job cwd/context channel, see ``hermes cron create --help``), then +``hermes cron run `` executes it. With no gateway owning the store the CLI +runs the job synchronously through ``cron.scheduler.run_job`` — the same agent +build the ticker uses (``hermes_cli/cron.py::_job_action`` forces the +synchronous path) — and prints ``Ran now: succeeded.``. The job's saved output +file (``cron/output//.md``, ``## Response`` section) is what cron +delivers locally, so that is ``final_text``. + +Both CLI calls run from the fake HOME, not the project: a scheduled job's +process cwd is whatever the ticker had, so the workdir is the only channel +that may carry the project context. +""" + +from __future__ import annotations + +import re +import subprocess + +from tests.e2e.core.parity._helpers import TURN_TIMEOUT, DriveResult, ParityHome, hermes_argv +from tests.fakes.fake_llm_provider import FakeLLMServer + +# cron/scheduler.py::_resolve_cron_enabled_toolsets -> _get_platform_tools(cfg, "cron") default. +CRON_TOOLSET = "hermes-cron" +_JOB_ID = re.compile(r"Created job: (\S+)") +_RESPONSE = re.compile(r"^## Response\s*\n(.*)\Z", re.S | re.M) + + +def _cron(ph: ParityHome, *args: str) -> subprocess.CompletedProcess: + return subprocess.run( + hermes_argv("cron", *args), cwd=ph.home, env=ph.env(), capture_output=True, text=True, + timeout=TURN_TIMEOUT, stdin=subprocess.DEVNULL, + ) + + +def drive_cron(ph: ParityHome, srv: FakeLLMServer, prompt: str) -> DriveResult: + created = _cron(ph, "create", "1d", prompt, "--name", "parity", "--deliver", "local", + "--workdir", str(ph.project)) + match = _JOB_ID.search(created.stdout) + assert created.returncode == 0 and match, ( + f"hermes cron create exited {created.returncode}: {created.stdout[-1000:]} {created.stderr[-2000:]}") + job_id = match.group(1) + + ran = _cron(ph, "run", job_id) + assert ran.returncode == 0, f"hermes cron run exited {ran.returncode}: {ran.stderr[-2000:]}" + # Anything but a synchronous verdict means the run was handed to a ticker/background + # worker this driver cannot observe — a harness problem, not a parity result. + assert "Ran now:" in ran.stdout, f"hermes cron run did not execute synchronously: {ran.stdout[-1000:]}" + + outputs = sorted((ph.hermes_home / "cron" / "output" / job_id).glob("*.md")) + final_text = None + if outputs: + body = outputs[-1].read_text(encoding="utf-8") + m = _RESPONSE.search(body) + final_text = (m.group(1) if m else body).strip() + return DriveResult( + final_text=final_text, + toolset=CRON_TOOLSET, + cwd_channel="job workdir", + graceful_exit=True, # both CLI invocations exited on their own (subprocess.run, no signal) + extra={ + "job_id": job_id, + "run_stdout": ran.stdout.strip()[-500:], + "run_succeeded": "Ran now: succeeded." in ran.stdout, + "output_file": str(outputs[-1]) if outputs else None, + }, + ) diff --git a/tests/e2e/core/parity/_drive_gateway.py b/tests/e2e/core/parity/_drive_gateway.py new file mode 100644 index 0000000000..3f2862e78c --- /dev/null +++ b/tests/e2e/core/parity/_drive_gateway.py @@ -0,0 +1,246 @@ +"""Entrypoint drivers for the parity matrix: messaging gateway + OpenAI-compatible API server. + +Both follow the driver contract in ``_drive_cli``. Daemon surfaces have no launch +directory, so the documented cwd channel ``terminal.cwd`` is pinned in config.yaml +(``DriveResult.cwd_channel = "terminal.cwd"``) and credentials go in the fixture +home's ``.env`` exactly as a user configures them. + +* ``drive_gateway`` — the REAL gateway entrypoint (``gateway.run.main``: host-attach + decision, PID claim, MCP discovery, ``GatewayRunner.start()``, signal handlers, + graceful shutdown tail) in a subprocess; only the Telegram adapter instance is a + recording fake (``_gateway_child.py``), fed one inbound DM through its normal + ``handle_message`` path. +* ``drive_api_server`` — the stock ``python -m gateway.run`` with only the + ``api_server`` platform enabled (loopback, free port, strong key); one + non-streaming ``POST /v1/chat/completions``. + +Both stop through the normal operator path of ``hermes gateway stop``: write the +planned-stop marker for the gateway PID, then SIGTERM (a bare SIGTERM is treated +as an unexpected kill and exits non-zero on purpose so supervisors revive it). +""" + +from __future__ import annotations + +import contextlib +import json +import os +import queue +import secrets +import signal +import socket +import subprocess +import sys +import threading +import time +import urllib.error +import urllib.request +from pathlib import Path +from typing import Any + +from tests.e2e.core.parity._helpers import TURN_TIMEOUT, DriveResult, ParityHome +from tests.fakes.fake_llm_provider import FakeLLMServer + +GATEWAY_CHILD = Path(__file__).with_name("_gateway_child.py") +_LINE_PREFIX = "PARITY-GW " +STOP_TIMEOUT = 60.0 +# Telegram private chats use the user id as chat id; the child reads it from the env +# (importing the child here would import gateway.run into the test process). +PARITY_USER_ID = "424242" + + +def _append_env(ph: ParityHome, values: dict[str, str]) -> None: + path = ph.hermes_home / ".env" + existing = path.read_text(encoding="utf-8") if path.exists() else "" + if existing and not existing.endswith("\n"): + existing += "\n" + path.write_text(existing + "".join(f"{k}={v}\n" for k, v in values.items()), encoding="utf-8") + + +def _spawn(ph: ParityHome, argv: list[str], log_name: str, extra_env: dict[str, str] | None = None): + log_path = ph.root / log_name + log = open(log_path, "wb") # noqa: SIM115 - closed by _stop + proc = subprocess.Popen( + argv, cwd=ph.project, env=ph.env(extra_env), stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, stderr=log, text=True, bufsize=1, + ) + proc._parity_log = log # type: ignore[attr-defined] + proc._parity_log_path = log_path # type: ignore[attr-defined] + return proc + + +def _stdout_pump(proc: subprocess.Popen) -> "queue.Queue[str | None]": + lines: queue.Queue[str | None] = queue.Queue() + + def pump() -> None: + assert proc.stdout is not None + for line in proc.stdout: + lines.put(line) + lines.put(None) + + threading.Thread(target=pump, daemon=True, name="parity-gw-stdout").start() + return lines + + +def _log_tail(proc: subprocess.Popen, n: int = 4000) -> str: + try: + return proc._parity_log_path.read_text(encoding="utf-8", errors="replace")[-n:] # type: ignore[attr-defined] + except OSError: + return "" + + +def _stop(ph: ParityHome, proc: subprocess.Popen) -> bool: + """``hermes gateway stop`` semantics (planned-stop marker for the PID, then SIGTERM), + in its worst-case interleaving: the SIGTERM lands only AFTER the gateway's + planned-stop watcher has already consumed the marker. Any interleaving of the + CLI's two steps must still be a clean planned exit (exit 0; a non-zero exit + makes systemd/launchd revive a gateway the operator just stopped). + + Returns True when the gateway exited 0 on its own within ``STOP_TIMEOUT``. + """ + graceful = False + try: + if proc.poll() is None: + marker = subprocess.run( + [sys.executable, "-c", + "import sys; from gateway.status import write_planned_stop_marker as w, " + "_get_planned_stop_marker_path as p; ok = w(int(sys.argv[1])); print(p()); " + "sys.exit(0 if ok else 1)", + str(proc.pid)], + cwd=ph.project, env=ph.env(), capture_output=True, text=True, timeout=60, + stdin=subprocess.DEVNULL, + ) + assert marker.returncode == 0, f"planned-stop marker write failed: {marker.stderr[-1000:]}" + marker_path = Path(marker.stdout.strip().splitlines()[-1]) + deadline = time.monotonic() + STOP_TIMEOUT + while marker_path.exists() and proc.poll() is None and time.monotonic() < deadline: + time.sleep(0.05) + if proc.poll() is None: + with contextlib.suppress(ProcessLookupError): + os.kill(proc.pid, signal.SIGTERM) + try: + proc.wait(timeout=STOP_TIMEOUT) + graceful = proc.returncode == 0 + except subprocess.TimeoutExpired: + proc.kill() + proc.wait(timeout=10) + finally: + proc._parity_log.close() # type: ignore[attr-defined] + return graceful + + +# Messaging gateway (Telegram-shaped fake adapter) ------------------------------------ + + +def _collect_delivered(events: list[dict[str, Any]]) -> str: + """Final text of every message the client would see (sends, updated by later edits).""" + messages: dict[str, str] = {} + for ev in events: + if ev["kind"] in ("send", "edit"): + messages[ev["message_id"]] = ev["content"] + return "\n".join(messages.values()) + + +def drive_gateway(ph: ParityHome, srv: FakeLLMServer, prompt: str) -> DriveResult: + ph.pin_terminal_cwd() + # A real Telegram setup: bot token + allowlisted user in the profile .env. + _append_env(ph, {"TELEGRAM_BOT_TOKEN": "123456:parity-fake-token", + "TELEGRAM_ALLOWED_USERS": PARITY_USER_ID}) + proc = _spawn(ph, [sys.executable, str(GATEWAY_CHILD)], "gateway.stderr.log", + {"PARITY_GATEWAY_PROMPT": prompt, "PARITY_GATEWAY_USER": PARITY_USER_ID}) + lines = _stdout_pump(proc) + events: list[dict[str, Any]] = [] + outcome: str | None = None + graceful = False + try: + deadline = time.monotonic() + TURN_TIMEOUT + while outcome is None: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise AssertionError(f"gateway turn timed out; events={events}\n{_log_tail(proc)}") + try: + line = lines.get(timeout=remaining) + except queue.Empty: + continue + if line is None: + raise AssertionError( + f"gateway exited {proc.wait()} before the turn completed; events={events}\n{_log_tail(proc)}") + if not line.startswith(_LINE_PREFIX): + continue + ev = json.loads(line[len(_LINE_PREFIX):]) + events.append(ev) + if ev["kind"] == "complete": + outcome = ev["outcome"] + finally: + graceful = _stop(ph, proc) + return DriveResult( + final_text=_collect_delivered(events), toolset="hermes-telegram", cwd_channel="terminal.cwd", + graceful_exit=graceful, + extra={"outcome": outcome, "exit_code": proc.returncode, "events": events, + "stderr_log": str(proc._parity_log_path)}, # type: ignore[attr-defined] + ) + + +# OpenAI-compatible API server ------------------------------------------------------- + + +def _free_loopback_port() -> int: + with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s: + s.bind(("127.0.0.1", 0)) + return s.getsockname()[1] + + +def _http(method: str, url: str, *, key: str | None = None, body: dict | None = None, + timeout: float = 5.0) -> tuple[int, str]: + data = json.dumps(body).encode() if body is not None else None + req = urllib.request.Request(url, data=data, method=method) + if data is not None: + req.add_header("Content-Type", "application/json") + if key: + req.add_header("Authorization", f"Bearer {key}") + try: + with urllib.request.urlopen(req, timeout=timeout) as resp: + return resp.status, resp.read().decode("utf-8", "replace") + except urllib.error.HTTPError as exc: + return exc.code, exc.read().decode("utf-8", "replace") + + +def drive_api_server(ph: ParityHome, srv: FakeLLMServer, prompt: str) -> DriveResult: + ph.pin_terminal_cwd() + port = _free_loopback_port() + key = secrets.token_hex(32) + _append_env(ph, {"API_SERVER_ENABLED": "true", "API_SERVER_KEY": key, + "API_SERVER_HOST": "127.0.0.1", "API_SERVER_PORT": str(port)}) + proc = _spawn(ph, [sys.executable, "-m", "gateway.run"], "api_server.stderr.log") + base = f"http://127.0.0.1:{port}" + status: int | None = None + payload: dict[str, Any] = {} + graceful = False + try: + deadline = time.monotonic() + TURN_TIMEOUT + while True: + if proc.poll() is not None: + raise AssertionError(f"api server exited {proc.returncode} before ready\n{_log_tail(proc)}") + if time.monotonic() >= deadline: + raise AssertionError(f"api server never became ready on {base}\n{_log_tail(proc)}") + try: + if _http("GET", f"{base}/health", timeout=2.0)[0] == 200: + break + except (OSError, urllib.error.URLError): + pass + time.sleep(0.2) + status, raw = _http( + "POST", f"{base}/v1/chat/completions", key=key, timeout=TURN_TIMEOUT, + body={"model": "hermes-agent", "messages": [{"role": "user", "content": prompt}], + "stream": False}, + ) + payload = json.loads(raw) if raw.strip().startswith("{") else {"raw": raw} + assert status == 200, f"/v1/chat/completions -> {status}: {raw[-2000:]}\n{_log_tail(proc)}" + finally: + graceful = _stop(ph, proc) + choices = payload.get("choices") or [{}] + text = (choices[0].get("message") or {}).get("content") + return DriveResult( + final_text=text, toolset="hermes-api-server", cwd_channel="terminal.cwd", graceful_exit=graceful, + extra={"http_status": status, "exit_code": proc.returncode, + "stderr_log": str(proc._parity_log_path)}, # type: ignore[attr-defined] + ) diff --git a/tests/e2e/core/parity/_drive_rpc.py b/tests/e2e/core/parity/_drive_rpc.py new file mode 100644 index 0000000000..2dcc71c16d --- /dev/null +++ b/tests/e2e/core/parity/_drive_rpc.py @@ -0,0 +1,272 @@ +"""Entrypoint drivers for the parity matrix: the JSON-RPC surfaces. + +* ``drive_tui_gateway`` — ``python -m tui_gateway.entry`` over stdio, exactly how + the Ink TUI spawns it (launch dir = the user's shell cwd); stops on stdin EOF. +* ``drive_serve`` — ``hermes serve --host 127.0.0.1 --port 0`` spawned the way the + Desktop spawns it (``TERMINAL_CWD`` + ``HERMES_DASHBOARD_SESSION_TOKEN`` + + ``HERMES_DESKTOP=1`` env, stdin closed), port read from the READY sentinel on + stdout, one turn over the ``/api/ws`` JSON-RPC WebSocket; stops on SIGTERM. + +Both speak the same contract (``tui_gateway/contracts``): ``session.create`` → +``prompt.submit`` → ``message.complete`` event. Follows the driver contract in +``_drive_cli``. +""" + +from __future__ import annotations + +import json +import queue +import subprocess +import threading +import time +import uuid +from dataclasses import dataclass, field +from typing import Any, Callable + +from tests.e2e.core.parity._helpers import ( + TURN_TIMEOUT, + DriveResult, + ParityHome, + hermes_argv, + terminate, +) +from tests.fakes.fake_llm_provider import FakeLLMServer + +import sys + +READY_TIMEOUT = 120.0 + + +class RpcClient: + """Minimal JSON-RPC 2.0 client over a line/frame transport (``send`` + ``recv`` callables).""" + + def __init__(self, send: Callable[[str], None], frames: "queue.Queue[str | None]") -> None: + self._send = send + self._frames = frames + self._next_id = 0 + self.events: list[dict[str, Any]] = [] + self._responses: dict[Any, dict[str, Any]] = {} + + def _pump(self, timeout: float) -> bool: + try: + raw = self._frames.get(timeout=timeout) + except queue.Empty: + return False + if raw is None: + raise AssertionError("JSON-RPC transport closed") + msg = json.loads(raw) + if msg.get("method") == "event": + self.events.append(msg.get("params") or {}) + elif "id" in msg and "method" in msg: + # Server→client request (approval/clarify/...): this scripted turn never needs one. + self._send(json.dumps({"jsonrpc": "2.0", "id": msg["id"], + "error": {"code": -32601, "message": "parity driver: unsupported"}})) + elif "id" in msg: + self._responses[msg["id"]] = msg + return True + + def call(self, method: str, params: dict[str, Any] | None = None, timeout: float = 60.0) -> Any: + self._next_id += 1 + rid = f"parity-{self._next_id}" + self._send(json.dumps({"jsonrpc": "2.0", "id": rid, "method": method, "params": params or {}})) + deadline = time.monotonic() + timeout + while rid not in self._responses: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise AssertionError(f"{method}: no response within {timeout}s") + self._pump(min(remaining, 0.5)) + resp = self._responses.pop(rid) + if "error" in resp: + raise AssertionError(f"{method} failed: {resp['error']}") + return resp.get("result") + + def wait_event(self, etype: str, pred: Callable[[dict], bool] = lambda _e: True, + timeout: float = TURN_TIMEOUT) -> dict[str, Any]: + deadline = time.monotonic() + timeout + seen = 0 + while True: + for ev in self.events[seen:]: + if ev.get("type") == etype and pred(ev): + return ev + seen = len(self.events) + remaining = deadline - time.monotonic() + if remaining <= 0: + raise AssertionError( + f"no {etype!r} event within {timeout}s; saw {[e.get('type') for e in self.events][-30:]}") + self._pump(min(remaining, 0.5)) + + +def run_turn(rpc: RpcClient, prompt: str, create_params: dict[str, Any]) -> str: + created = rpc.call("session.create", create_params, timeout=TURN_TIMEOUT) + sid = created["session_id"] + rpc.call("prompt.submit", {"session_id": sid, "text": prompt}, timeout=TURN_TIMEOUT) + done = rpc.wait_event("message.complete", lambda e: e.get("session_id") == sid) + payload = done.get("payload") or {} + text = payload.get("text") + return text if isinstance(text, str) else json.dumps(text) + + +@dataclass +class StreamCapture: + """Every stdout line (in order) plus the stderr text of a child, pumped on threads.""" + + stdout_lines: "queue.Queue[str | None]" = field(default_factory=queue.Queue) + stdout_seen: list[str] = field(default_factory=list) + stderr_chunks: list[str] = field(default_factory=list) + + def start(self, proc: subprocess.Popen) -> "StreamCapture": + def out() -> None: + assert proc.stdout is not None + for line in proc.stdout: + self.stdout_seen.append(line) + self.stdout_lines.put(line) + self.stdout_lines.put(None) + + def err() -> None: + assert proc.stderr is not None + for line in proc.stderr: + self.stderr_chunks.append(line) + + threading.Thread(target=out, daemon=True, name="parity-stdout").start() + threading.Thread(target=err, daemon=True, name="parity-stderr").start() + return self + + @property + def stderr(self) -> str: + return "".join(self.stderr_chunks) + + +# tui_gateway (stdio) --------------------------------------------------------------- + + +def spawn_tui_gateway(ph: ParityHome) -> tuple[subprocess.Popen, StreamCapture]: + proc = subprocess.Popen( + [sys.executable, "-m", "tui_gateway.entry"], cwd=ph.project, env=ph.env(), + stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, bufsize=1, + ) + return proc, StreamCapture().start(proc) + + +def drive_tui_gateway(ph: ParityHome, srv: FakeLLMServer, prompt: str) -> DriveResult: + proc, cap = spawn_tui_gateway(ph) + + def send(line: str) -> None: + assert proc.stdin is not None + proc.stdin.write(line + "\n") + proc.stdin.flush() + + graceful = False + try: + rpc = RpcClient(send, cap.stdout_lines) + rpc.wait_event("gateway.ready", timeout=READY_TIMEOUT) + text = run_turn(rpc, prompt, {}) + finally: + # Normal stop for a stdio gateway: the client closes the pipe. + if proc.stdin is not None and not proc.stdin.closed: + proc.stdin.close() + try: + proc.wait(timeout=60) + graceful = True + except subprocess.TimeoutExpired: + terminate(proc) + return DriveResult(final_text=text, toolset="hermes-cli", graceful_exit=graceful, + extra={"exit_code": proc.returncode, "stderr_tail": cap.stderr[-2000:]}) + + +# hermes serve (Desktop backend) ------------------------------------------------------ + + +@dataclass +class ServeProcess: + proc: subprocess.Popen + cap: StreamCapture + token: str + + +def spawn_serve(ph: ParityHome) -> ServeProcess: + """Spawn ``hermes serve`` with the Desktop's argv/env/stdio shape (electron/main.ts).""" + token = uuid.uuid4().hex + env = ph.env({ + "TERMINAL_CWD": str(ph.project), + "HERMES_DASHBOARD_SESSION_TOKEN": token, + "HERMES_DESKTOP": "1", + }) + proc = subprocess.Popen( + hermes_argv("serve", "--host", "127.0.0.1", "--port", "0"), cwd=ph.home, env=env, + stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, bufsize=1, + ) + return ServeProcess(proc=proc, cap=StreamCapture().start(proc), token=token) + + +def first_stdout_line(sp: ServeProcess, timeout: float = READY_TIMEOUT) -> str: + """The first stdout line, failing fast if the sentinel shows up on stderr instead.""" + deadline = time.monotonic() + timeout + while True: + try: + line = sp.cap.stdout_lines.get(timeout=0.2) + except queue.Empty: + line = "" + if line is None: + raise AssertionError(f"serve exited {sp.proc.poll()} before any stdout line\n{sp.cap.stderr[-3000:]}") + if line: + return line + if "HERMES_BACKEND_READY" in sp.cap.stderr: + raise AssertionError( + "READY sentinel was written to STDERR; the Desktop only watches stdout " + f"(backend-ready.ts):\n{sp.cap.stderr[-1500:]}") + if time.monotonic() >= deadline: + raise AssertionError(f"no stdout line from serve within {timeout}s\n{sp.cap.stderr[-3000:]}") + + +def desktop_port(sp: ServeProcess, timeout: float = READY_TIMEOUT) -> int: + """The port the Desktop would learn from this backend's stdout (junk-tolerant, like + backend-ready.ts). The strict "sentinel is the first line" rule is test_boot_contract's.""" + from tests.e2e.core.parity._boot_contract import desktop_parse + + deadline = time.monotonic() + timeout + seen = "" + while True: + remaining = deadline - time.monotonic() + line = first_stdout_line(sp, timeout=max(remaining, 0.1)) + seen += line + verdict = desktop_parse(seen) + if "port" in verdict: + return int(verdict["port"]) + + +def ws_rpc(port: int, token: str) -> tuple[RpcClient, Callable[[], None]]: + from websockets.sync.client import connect + + ws = connect(f"ws://127.0.0.1:{port}/api/ws?token={token}", open_timeout=30, max_size=None) + frames: queue.Queue[str | None] = queue.Queue() + + def pump() -> None: + try: + for msg in ws: + frames.put(msg if isinstance(msg, str) else msg.decode()) + except Exception: + pass + frames.put(None) + + threading.Thread(target=pump, daemon=True, name="parity-ws").start() + return RpcClient(ws.send, frames), ws.close + + +def drive_serve(ph: ParityHome, srv: FakeLLMServer, prompt: str) -> DriveResult: + sp = spawn_serve(ph) + graceful = False + try: + port = desktop_port(sp) + rpc, close = ws_rpc(port, sp.token) + try: + rpc.wait_event("gateway.ready", timeout=READY_TIMEOUT) + text = run_turn(rpc, prompt, {"cwd": str(ph.project)}) + finally: + close() + finally: + # Desktop quit path: SIGTERM, bounded wait. + code = terminate(sp.proc, timeout=60) + graceful = code is not None and code in (0, -15, 143) + return DriveResult(final_text=text, toolset="hermes-cli", cwd_channel="TERMINAL_CWD + session.create cwd", + graceful_exit=graceful, + extra={"exit_code": sp.proc.returncode, "stderr_tail": sp.cap.stderr[-2000:]}) diff --git a/tests/e2e/core/parity/_gateway_child.py b/tests/e2e/core/parity/_gateway_child.py new file mode 100644 index 0000000000..8413b72d4e --- /dev/null +++ b/tests/e2e/core/parity/_gateway_child.py @@ -0,0 +1,118 @@ +"""Lane-private child process for ``_drive_gateway.drive_gateway``. + +Runs the REAL messaging gateway entrypoint (``gateway.run.main`` — what +``python -m gateway.run`` / the ``hermes gateway run`` service executes: host +attach decision, PID claim, MCP discovery, ``GatewayRunner.start()``, signal +handlers, the graceful shutdown tail) with exactly one substitution: the +Telegram adapter instance is a recording fake. Everything behind the adapter +(authorization, session store, SessionDB, agent cache, AIAgent, hooks, plugins, +MCP) is production code. + +The fake adapter injects ONE inbound DM through the adapter's normal +``handle_message`` path once the runner is running, prints every outbound +``send()`` and the ``on_processing_complete`` outcome as ``PARITY-GW `` +lines on stdout, and otherwise waits for the parent's normal stop +(planned-stop marker + SIGTERM, exactly like ``hermes gateway stop``). +""" + +from __future__ import annotations + +import asyncio +import json +import os +import sys +from typing import Any, Dict, Optional + +from gateway.config import Platform +from gateway.platforms.base import BasePlatformAdapter, MessageEvent, SendResult + +import gateway.run as gateway_run + +PARITY_USER_ID = os.environ.get("PARITY_GATEWAY_USER", "424242") +PARITY_CHAT_ID = PARITY_USER_ID # Telegram private chats use the user id as chat id + + +# The gateway rebinds sys.stdout during startup; report on a private dup of the original pipe. +_REPORT = os.fdopen(os.dup(1), "w", buffering=1, encoding="utf-8") + + +def emit(kind: str, **payload: Any) -> None: + _REPORT.write("PARITY-GW " + json.dumps({"kind": kind, **payload}) + "\n") + _REPORT.flush() + + +class ParityTelegramAdapter(BasePlatformAdapter): + """Telegram-shaped adapter: no network, records outbound text, injects one DM.""" + + def __init__(self, config: Any, platform: Platform) -> None: + super().__init__(config, platform) + self._sent = 0 + self._inject_task: Optional[asyncio.Task] = None + + async def connect(self, *, is_reconnect: bool = False) -> bool: + self._mark_connected() + if self._inject_task is None: + self._inject_task = asyncio.create_task(self._inject_once()) + return True + + async def _inject_once(self) -> None: + # Inbound arriving while startup restore runs is queued and replayed later (the first + # handle_message then "completes" without a turn); inject once the gateway is serving. + runner = self.gateway_runner + while not getattr(runner, "_running", False) or getattr(runner, "_startup_restore_in_progress", True): + await asyncio.sleep(0.05) + emit("ready") + source = self.build_source( + chat_id=PARITY_CHAT_ID, chat_name="parity", chat_type="dm", + user_id=PARITY_USER_ID, user_name="parity-user", message_id="1", + ) + event = MessageEvent( + text=os.environ["PARITY_GATEWAY_PROMPT"], source=source, message_id="1", + user_id=PARITY_USER_ID, user_name="parity-user", + ) + await self.handle_message(event) + emit("accepted", accepted=bool(getattr(event, "_gateway_accepted", False))) + + async def disconnect(self) -> None: + if self._inject_task is not None and not self._inject_task.done(): + self._inject_task.cancel() + self._mark_disconnected() + + async def send(self, chat_id: str, content: str, reply_to: Optional[str] = None, + metadata: Optional[Dict[str, Any]] = None) -> SendResult: + self._sent += 1 + message_id = f"parity-{self._sent}" + emit("send", chat_id=str(chat_id), message_id=message_id, content=content) + return SendResult(success=True, message_id=message_id) + + async def edit_message(self, chat_id: str, message_id: str, content: str, *, + finalize: bool = False) -> SendResult: + emit("edit", chat_id=str(chat_id), message_id=message_id, content=content) + return SendResult(success=True, message_id=message_id) + + async def get_chat_info(self, chat_id: str) -> Dict[str, Any]: + return {"name": "parity", "type": "dm", "chat_id": chat_id} + + async def on_processing_complete(self, event: MessageEvent, outcome: Any) -> None: + emit("complete", outcome=str(getattr(outcome, "value", outcome))) + + +_real_instantiate = gateway_run.GatewayRunner._instantiate_adapter + + +def _instantiate_adapter(self, platform: Platform, config: Any): + if platform == Platform.TELEGRAM: + return ParityTelegramAdapter(config, platform) + return _real_instantiate(self, platform, config) + + +def main() -> None: + # The one seam: adapter construction (``GatewayRunner._create_adapter`` still wires runner + + # handlers onto whatever this returns, exactly as for the real Telegram adapter). + gateway_run.GatewayRunner._instantiate_adapter = _instantiate_adapter + sys.argv = ["gateway.run"] + gateway_run.main() + + +if __name__ == "__main__": + main() diff --git a/tests/e2e/core/parity/_helpers.py b/tests/e2e/core/parity/_helpers.py new file mode 100644 index 0000000000..91ef212747 --- /dev/null +++ b/tests/e2e/core/parity/_helpers.py @@ -0,0 +1,453 @@ +"""Shared fixture HERMES_HOME + invariant checks for the entrypoint-parity suite. + +One fixture home carries every feature that has historically been wired into +some entrypoints and forgotten in others (issue class C19): a shell hook and a +Python-plugin hook on ``pre_llm_call`` (both write a marker AND inject a +context canary), an ``AGENTS.md`` context file in the working directory, a +skill, a memory entry, a stdio MCP server (issue class C15) whose tool returns a +canary, the custom provider pointed at the recording fake, and a toolset +restriction (``agent.disabled_toolsets``). + +Every entrypoint driver runs ONE turn against the scripted fake provider +(turn 1: call the MCP tool; turn 2: answer) and hands the recorded requests to +:func:`collect_observation`; the test then asserts the same invariants for +every entrypoint, so any red cell is a wiring-parity bug. +""" + +from __future__ import annotations + +import json +import os +import re +import signal +import subprocess +import sys +import time +import uuid +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Callable, Iterable + +import yaml + +from tests.fakes.fake_llm_provider import FakeLLMServer, Text, ToolCall, write_hermes_home + +REPO_ROOT = Path(__file__).resolve().parents[4] +FIXTURE_MCP_SERVER = Path(__file__).with_name("fixture_mcp_server.py") + +MCP_SERVER_NAME = "parity" +MCP_TOOL_NAME = "mcp__parity__parity_canary" +FINAL_ANSWER = "PARITY-TURN-COMPLETE" +PLUGIN_NAME = "parity-plugin" +# Always-available (no credential/check_fn gate) toolset the fixture disables. +DISABLED_TOOLSET = "todo" + +_SECRET_ENV_SUFFIXES = ("_API_KEY", "_TOKEN", "_SECRET", "_ACCESS_KEY") +_PASSTHROUGH_ENV = frozenset({ + "PATH", "LANG", "LANGUAGE", "USER", "LOGNAME", "SHELL", "TMPDIR", "TZ", + "SYSTEMROOT", "SystemRoot", "COMSPEC", "PATHEXT", "WINDIR", "TEMP", "TMP", +}) + + +TURN_TIMEOUT = 240.0 + + +@dataclass +class DriveResult: + """What an entrypoint driver reports back (see ``_drive_cli`` for the contract).""" + + final_text: str | None + toolset: str # the platform toolset this surface documents (a toolsets.py key) + cwd_channel: str = "launch dir" + graceful_exit: bool = True + extra: dict[str, Any] = field(default_factory=dict) + + +@dataclass +class ParityHome: + """Filesystem layout + canaries of one fixture home.""" + + root: Path + home: Path + hermes_home: Path + project: Path + markers: Path + pid_log: Path + tag: str + canaries: dict[str, str] = field(default_factory=dict) + + def env(self, extra: dict[str, str] | None = None) -> dict[str, str]: + """Hermetic env for a subprocess Hermes: fake HOME, no real credentials.""" + import pwd # POSIX-only; the suite is Linux-gated + + real_root = Path(pwd.getpwuid(os.getuid()).pw_dir, ".hermes").resolve() + assert real_root not in (self.hermes_home.resolve(), *self.hermes_home.resolve().parents), ( + f"fixture HERMES_HOME {self.hermes_home} is inside the real {real_root}") + # Allowlist, not denylist: the runner may itself be a Hermes process whose + # TERMINAL_CWD / HERMES_* / credential env would silently reroute the child. + env = { + k: v for k, v in os.environ.items() + if (k in _PASSTHROUGH_ENV or k.startswith("LC_")) and not k.endswith(_SECRET_ENV_SUFFIXES) + } + env.update({ + "HOME": str(self.home), + "HERMES_HOME": str(self.hermes_home), + "PYTHONPATH": str(REPO_ROOT), + "PYTHONUNBUFFERED": "1", + "NO_COLOR": "1", + "TERM": "dumb", + # Orphan-scan tag: every process in the spawned tree inherits it. + "PARITY_TREE_TAG": self.tag, + # The child's HOME is the fixture home, so its ``~/.hermes/state.db`` IS + # the tmp HERMES_HOME's db; under a pytest ancestor the live-DB guard + # (hermes_state_guard) would refuse it. This is the guard's documented + # child-process escape hatch; the path is tmp_path by construction. + "HERMES_STATE_DB_GUARD_BYPASS": "1", + }) + env.update(extra or {}) + return env + + def update_config(self, mutate: Callable[[dict], None]) -> None: + """Apply ``mutate`` to config.yaml (for surfaces with a documented config-only channel).""" + path = self.hermes_home / "config.yaml" + cfg = yaml.safe_load(path.read_text(encoding="utf-8")) + mutate(cfg) + path.write_text(yaml.safe_dump(cfg, sort_keys=False), encoding="utf-8") + + def pin_terminal_cwd(self) -> None: + """Documented cwd channel for daemon surfaces (gateway, api_server): ``terminal.cwd``.""" + self.update_config(lambda cfg: cfg.setdefault("terminal", {}).__setitem__("cwd", str(self.project))) + + +def build_parity_home(root: Path, base_url: str, *, grandchild: bool = True) -> ParityHome: + """Write the full fixture home under ``root`` (a tmp_path).""" + home = root / "home" + hermes_home = home / ".hermes" + project = root / "project" + markers = root / "markers" + for d in (hermes_home, project, markers): + d.mkdir(parents=True, exist_ok=True) + tag = uuid.uuid4().hex + c = {name: f"{name.upper()}-{uuid.uuid4().hex[:12]}" for name in ( + "context", "skill", "memory", "shell_hook", "plugin_hook", "mcp")} + ph = ParityHome(root=root, home=home, hermes_home=hermes_home, project=project, + markers=markers, pid_log=root / "mcp_pids.log", tag=tag, canaries=c) + + write_hermes_home(hermes_home, base_url) + cfg = yaml.safe_load((hermes_home / "config.yaml").read_text(encoding="utf-8")) + cfg["agent"]["disabled_toolsets"] = [DISABLED_TOOLSET] + cfg["hooks"] = {"pre_llm_call": [{ + "command": f"{sys.executable} {hermes_home / 'agent-hooks' / 'shell_hook.py'}", + "timeout": 30, + }]} + cfg["hooks_auto_accept"] = True + cfg["plugins"] = {"enabled": [PLUGIN_NAME]} + cfg["mcp_servers"] = {MCP_SERVER_NAME: { + "command": sys.executable, + "args": [str(FIXTURE_MCP_SERVER)], + "env": { + "PARITY_MCP_CANARY": c["mcp"], + "PARITY_MCP_PID_LOG": str(ph.pid_log), + "PARITY_MCP_SPAWN_GRANDCHILD": "1" if grandchild else "0", + "PARITY_TREE_TAG": tag, + # Only the MCP server and its descendants carry this one. + "PARITY_MCP_TREE_TAG": tag, + "PYTHONPATH": str(REPO_ROOT), + }, + "connect_timeout": 60, + "timeout": 60, + }} + # Keep turns hermetic and short: no title/aux model chatter decides anything here. + cfg.setdefault("display", {})["compact"] = True + (hermes_home / "config.yaml").write_text(yaml.safe_dump(cfg, sort_keys=False), encoding="utf-8") + + hooks_dir = hermes_home / "agent-hooks" + hooks_dir.mkdir() + (hooks_dir / "shell_hook.py").write_text( + "import json, os, sys\n" + "payload = json.load(sys.stdin)\n" + f"with open({str(markers / 'shell_hook.log')!r}, 'a') as fh:\n" + " fh.write(json.dumps({'event': payload.get('hook_event_name'),\n" + " 'session_id': payload.get('session_id')}) + '\\n')\n" + f"print(json.dumps({{'context': {c['shell_hook']!r}}}))\n", + encoding="utf-8", + ) + + plugin_dir = hermes_home / "plugins" / PLUGIN_NAME + plugin_dir.mkdir(parents=True) + (plugin_dir / "plugin.yaml").write_text( + f"name: {PLUGIN_NAME}\nversion: 1.0.0\ndescription: parity suite marker plugin\n", + encoding="utf-8", + ) + (plugin_dir / "__init__.py").write_text( + "import json\n" + "def register(ctx):\n" + " def _pre_llm_call(**kwargs):\n" + f" with open({str(markers / 'plugin_hook.log')!r}, 'a') as fh:\n" + " fh.write(json.dumps({'platform': kwargs.get('platform'),\n" + " 'session_id': kwargs.get('session_id')}) + '\\n')\n" + f" return {{'context': {c['plugin_hook']!r}}}\n" + " ctx.register_hook('pre_llm_call', _pre_llm_call)\n", + encoding="utf-8", + ) + + skill_dir = hermes_home / "skills" / "parity-skill" + skill_dir.mkdir(parents=True) + (skill_dir / "SKILL.md").write_text( + "---\nname: parity-skill\n" + f"description: Use when checking entrypoint parity {c['skill']}.\n---\n\n" + "# Parity skill\n\nNothing to do.\n", + encoding="utf-8", + ) + + mem_dir = hermes_home / "memories" + mem_dir.mkdir() + (mem_dir / "MEMORY.md").write_text(f"The parity memory canary is {c['memory']}.", encoding="utf-8") + + (project / "AGENTS.md").write_text( + f"# Project rules\n\nThe parity context canary is {c['context']}.\n", encoding="utf-8") + return ph + + +# Scripted provider ----------------------------------------------------------- + + +def parity_responder(nonce: str) -> Callable[[dict[str, Any]], Any]: + """Stateless script: call the MCP tool until a tool result exists, then answer. + + Stateless so a retried request (or a second concurrent entrypoint request) + cannot desynchronise the script. + """ + + def respond(record: dict[str, Any]): + body = record["body"] + if any(m.get("role") == "tool" for m in body.get("messages") or []): + return Text(FINAL_ANSWER) + args = {"nonce": nonce} + if MCP_TOOL_NAME in _tool_names(body) or "tool_call" not in _tool_names(body): + return ToolCall(MCP_TOOL_NAME, args) + # Tool Search active: MCP tools sit in the deferred catalog behind the bridge. + return ToolCall("tool_call", {"calls": [{"name": MCP_TOOL_NAME, "arguments": args}]}) + + return respond + + +def start_provider(nonce: str) -> FakeLLMServer: + srv = FakeLLMServer(parity_responder(nonce)) + srv.start() + return srv + + +# Observation + invariants ---------------------------------------------------- + + +def _text(content: Any) -> str: + if isinstance(content, str): + return content + if isinstance(content, list): + return "\n".join(p.get("text", "") for p in content if isinstance(p, dict)) + return "" + + +def _tool_names(body: dict[str, Any]) -> set[str]: + return {(t.get("function") or {}).get("name") or t.get("name") for t in body.get("tools") or []} + + +_CATALOG_LINE = re.compile(r"^- ([A-Za-z0-9_.:-]+): ", re.M) + + +def offered_tool_names(body: dict[str, Any]) -> set[str]: + """Direct tool schemas plus the Tool Search deferred catalog (names the model may invoke).""" + names = _tool_names(body) + for t in body.get("tools") or []: + fn = t.get("function") or {} + if fn.get("name") == "tool_search": + names |= set(_CATALOG_LINE.findall(fn.get("description") or "")) + return names + + +@dataclass +class Observation: + entrypoint: str + system: str + first_user: str + tool_names: set[str] + tool_results: list[str] + main_requests: int + shell_hook_events: list[dict] + plugin_hook_events: list[dict] + final_text: str | None = None + extra: dict[str, Any] = field(default_factory=dict) + + +def _read_jsonl(path: Path) -> list[dict]: + if not path.exists(): + return [] + return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines() if line.strip()] + + +def collect_observation(entrypoint: str, ph: ParityHome, srv: FakeLLMServer, + final_text: str | None = None) -> Observation: + mains = srv.main_requests() + assert mains, f"{entrypoint}: no main-turn request reached the provider" + first = mains[0] + msgs = first.get("messages") or [] + system = "\n".join(_text(m.get("content")) for m in msgs if m.get("role") == "system") + users = [_text(m.get("content")) for m in msgs if m.get("role") == "user"] + tool_results = [ + _text(m.get("content")) for body in mains for m in body.get("messages") or [] + if m.get("role") == "tool" + ] + return Observation( + entrypoint=entrypoint, system=system, first_user=users[-1] if users else "", + tool_names=offered_tool_names(first), tool_results=tool_results, main_requests=len(mains), + shell_hook_events=_read_jsonl(ph.markers / "shell_hook.log"), + plugin_hook_events=_read_jsonl(ph.markers / "plugin_hook.log"), + final_text=final_text, + ) + + +def disabled_tool_names() -> set[str]: + from toolsets import resolve_toolset + + return set(resolve_toolset(DISABLED_TOOLSET)) + + +# Agent features every surface must carry when its documented toolset includes them. +_FEATURE_TOOLS = ("terminal", "read_file", "write_file", "patch", "search_files", "memory", + "skills_list", "skill_view") + + +def required_tool_names(toolset: str) -> set[str]: + """The core feature tools the surface's documented toolset promises, plus the MCP tool.""" + from toolsets import resolve_toolset + + promised = set(resolve_toolset(toolset)) + return {t for t in _FEATURE_TOOLS if t in promised} | {MCP_TOOL_NAME} + + +def check_invariants(obs: Observation, ph: ParityHome, *, toolset: str = "hermes-cli", + context_file: bool = True) -> dict[str, bool]: + """Evaluate every parity invariant; returns {cell: ok} (all must be True).""" + c = ph.canaries + cells = { + "context_file": c["context"] in obs.system, + "skill_index": c["skill"] in obs.system, + "memory": c["memory"] in obs.system, + "shell_hook_fired": bool(obs.shell_hook_events), + "shell_hook_injected": c["shell_hook"] in obs.first_user, + "plugin_hook_fired": bool(obs.plugin_hook_events), + "plugin_hook_injected": c["plugin_hook"] in obs.first_user, + "mcp_tool_offered": MCP_TOOL_NAME in obs.tool_names, + "mcp_call_returned_canary": any(c["mcp"] in r for r in obs.tool_results), + "toolset_restriction": not (obs.tool_names & disabled_tool_names()), + "feature_tools": required_tool_names(toolset) <= obs.tool_names, + "turn_completed": obs.main_requests >= 2, + } + if not context_file: + cells.pop("context_file") + return cells + + +# Process-tree hygiene -------------------------------------------------------- + + +def tagged_pids(tag: str, var: str = "PARITY_TREE_TAG") -> set[int]: + """Live PIDs whose environment carries ``=`` (reparented orphans included).""" + needle = f"{var}={tag}".encode() + found: set[int] = set() + me = os.getpid() + for entry in os.listdir("/proc"): + if not entry.isdigit() or int(entry) == me: + continue + try: + with open(f"/proc/{entry}/environ", "rb") as fh: + env = fh.read() + with open(f"/proc/{entry}/stat", "rb") as fh: + state = fh.read().rsplit(b")", 1)[1].split()[0] + except OSError: + continue + if state != b"Z" and needle in env.split(b"\0"): + found.add(int(entry)) + return found + + +def mcp_pids(ph: ParityHome) -> dict[str, list[int]]: + out: dict[str, list[int]] = {} + if ph.pid_log.exists(): + for line in ph.pid_log.read_text(encoding="utf-8").splitlines(): + kind, pid = line.split() + out.setdefault(kind, []).append(int(pid)) + return out + + +def wait_until(pred: Callable[[], Any], timeout: float, what: str, interval: float = 0.05): + deadline = time.monotonic() + timeout + while True: + value = pred() + if value: + return value + if time.monotonic() >= deadline: + raise AssertionError(f"timed out after {timeout}s waiting for {what}") + time.sleep(interval) + + +def wait_no_orphans(ph: ParityHome, timeout: float = 30.0, *, mcp_only: bool = True) -> set[int]: + """Poll until no MCP-tree process (or, ``mcp_only=False``, no tagged process at all) + survives; returns the survivors (empty = clean).""" + var = "PARITY_MCP_TREE_TAG" if mcp_only else "PARITY_TREE_TAG" + deadline = time.monotonic() + timeout + survivors = tagged_pids(ph.tag, var) + while survivors and time.monotonic() < deadline: + time.sleep(0.1) + survivors = tagged_pids(ph.tag, var) + return survivors + + +def describe_pids(pids: Iterable[int]) -> list[str]: + out = [] + for pid in sorted(pids): + try: + with open(f"/proc/{pid}/cmdline", "rb") as fh: + cmd = fh.read().replace(b"\0", b" ").decode(errors="replace") + out.append(f"{pid}: {cmd[:160]}") + except OSError: + out.append(f"{pid}: ") + return out + + +def kill_tagged(ph: ParityHome) -> None: + """Test cleanup: SIGKILL anything this fixture spawned that is still alive.""" + for pid in tagged_pids(ph.tag): + try: + os.kill(pid, signal.SIGKILL) + except OSError: + pass + + +def hermes_argv(*args: str) -> list[str]: + return [sys.executable, "-m", "hermes_cli.main", *args] + + +def terminate(proc: subprocess.Popen, timeout: float = 30.0) -> int | None: + """Graceful SIGTERM (the normal stop), SIGKILL only if it overstays.""" + if proc.poll() is None: + proc.send_signal(signal.SIGTERM) + try: + proc.wait(timeout=timeout) + except subprocess.TimeoutExpired: + proc.kill() + proc.wait(timeout=10) + return proc.returncode + + +def format_cells(results: Iterable[tuple[str, dict[str, bool]]]) -> str: + rows = list(results) + cols = sorted({k for _, cells in rows for k in cells}) + head = "| entrypoint | " + " | ".join(cols) + " |" + sep = "|---" * (len(cols) + 1) + "|" + body = [ + f"| {ep} | " + " | ".join(("✅" if cells.get(k) else ("—" if k not in cells else "❌")) for k in cols) + " |" + for ep, cells in rows + ] + return "\n".join([head, sep, *body]) diff --git a/tests/e2e/core/parity/desktop_ready_parse.mjs b/tests/e2e/core/parity/desktop_ready_parse.mjs new file mode 100644 index 0000000000..a2bd9f36af --- /dev/null +++ b/tests/e2e/core/parity/desktop_ready_parse.mjs @@ -0,0 +1,30 @@ +// Feed bytes (stdin) through the Desktop's REAL READY-sentinel parser. +// +// Imports apps/desktop/electron/backend-ready.ts itself (Node >= 22.18 strips the +// types natively) and drives `waitForDashboardPort` with a child-shaped emitter +// whose stdout emits exactly the bytes the Python test captured from the backend's +// stdout. Prints {"port": N} or {"error": "..."}. The Python side therefore never +// re-implements or regexes the TS contract: if the parser changes, this follows. +import { EventEmitter } from 'node:events' +import { pathToFileURL } from 'node:url' + +const [parserPath, timeoutMs] = process.argv.slice(2) +const { waitForDashboardPort } = await import(pathToFileURL(parserPath).href) + +const chunks = [] +for await (const chunk of process.stdin) { + chunks.push(chunk) +} + +const child = new EventEmitter() +child.stdout = new EventEmitter() +const pending = waitForDashboardPort(child, Number(timeoutMs) || 5000) +child.stdout.emit('data', Buffer.concat(chunks)) +// Nothing more will arrive: the captured stream ended here. +child.emit('exit', 0, null) + +try { + console.log(JSON.stringify({ port: await pending })) +} catch (err) { + console.log(JSON.stringify({ error: String(err && err.message ? err.message : err) })) +} diff --git a/tests/e2e/core/parity/fixture_mcp_server.py b/tests/e2e/core/parity/fixture_mcp_server.py new file mode 100644 index 0000000000..9ecc19efea --- /dev/null +++ b/tests/e2e/core/parity/fixture_mcp_server.py @@ -0,0 +1,60 @@ +"""Tiny stdio MCP server used by the entrypoint-parity suite. + +Exposes one tool, ``parity_canary``, whose result is the value of the +``PARITY_MCP_CANARY`` env var — so a test can prove a REAL tool call round-tripped +through the real Hermes MCP client (discovery, transport, liveness checks, result +plumbing) instead of a registered-but-dead schema. + +``PARITY_MCP_SPAWN_GRANDCHILD=1`` makes the server fork a long-lived grandchild at +startup (the shape of npx/uvx wrappers and servers with worker helpers) so +shutdown tests can prove Hermes reaps the whole process tree, not just its direct +child. Every PID the server owns is appended to ``PARITY_MCP_PID_LOG`` so the test +can check liveness of exactly the processes this fixture created. +""" + +from __future__ import annotations + +import os +import subprocess +import sys + + +def _log_pid(kind: str, pid: int) -> None: + path = os.environ.get("PARITY_MCP_PID_LOG") + if not path: + return + with open(path, "a", encoding="utf-8") as fh: + fh.write(f"{kind} {pid}\n") + + +def _spawn_grandchild() -> None: + # Inherits the environment (and so the orphan-scan tag). Reads nothing, + # writes nothing, lives until killed: an orphan if the tree is not reaped. + child = subprocess.Popen( + [sys.executable, "-c", "import time\nwhile True: time.sleep(60)"], + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + ) + _log_pid("grandchild", child.pid) + + +def main() -> None: + from mcp.server import MCPServer + + _log_pid("server", os.getpid()) + if os.environ.get("PARITY_MCP_SPAWN_GRANDCHILD") == "1": + _spawn_grandchild() + + server = MCPServer("parity") + + @server.tool() + def parity_canary(nonce: str = "") -> str: + """Return the parity fixture canary (echoing the caller's nonce).""" + return f"{os.environ.get('PARITY_MCP_CANARY', 'NO-CANARY')}:{nonce}" + + server.run(transport="stdio") + + +if __name__ == "__main__": + main() diff --git a/tests/e2e/core/parity/test_entrypoint_parity.py b/tests/e2e/core/parity/test_entrypoint_parity.py new file mode 100644 index 0000000000..79878c20c7 --- /dev/null +++ b/tests/e2e/core/parity/test_entrypoint_parity.py @@ -0,0 +1,177 @@ +"""Entrypoint parity matrix (issue classes C19 + C15). + +One fixture HERMES_HOME (shell hook + plugin hook on ``pre_llm_call``, AGENTS.md, +a skill, a memory entry, a stdio MCP server that spawns a grandchild, the custom +provider pointed at the recording fake, a disabled toolset) and ONE scripted turn +per entrypoint: the fake model calls the MCP canary tool, then answers. + +For every entrypoint the same invariants must hold on the FIRST turn after a cold +start — any red cell is a wiring-parity bug of the "works in the CLI, missing in +Desktop/gateway/ACP/cron" class: + +* the request carries the context file, the skill index and the memory entry; +* both hooks fired AND their injected context reached the model; +* the MCP tool was offered and a REAL call returned the fixture's canary (the + inverted ``_stdio_children_dead`` burst failed exactly here on every surface); +* the documented toolset's core feature tools are present, the disabled toolset + is absent; +* the surface delivered the model's answer to its client; +* after the surface's normal shutdown no MCP server / grandchild survives. + +``PARITY_TABLE_OUT=`` appends a markdown row per entrypoint (REPORT.md table). +""" + +from __future__ import annotations + +import json +import os +import sys +import time +import uuid +from concurrent.futures import ThreadPoolExecutor +from dataclasses import asdict, dataclass, field +from pathlib import Path +from typing import Callable + +import pytest + +from tests.e2e.core.parity import _drive_acp, _drive_cli, _drive_cron, _drive_gateway, _drive_rpc +from tests.e2e.core.parity._helpers import ( + FINAL_ANSWER, + DriveResult, + ParityHome, + build_parity_home, + check_invariants, + collect_observation, + describe_pids, + kill_tagged, + mcp_pids, + start_provider, + wait_no_orphans, +) + + +# Linux-only (/proc process-tree scans). The live-system guard bypass is needed +# ONLY for the orphan sweep: orphans reparented to init (the exact failure this suite +# hunts) sit outside the pytest subtree, and kill_tagged() signals nothing but +# PIDs carrying this run's unique PARITY_TREE_TAG. Children run with a tmp +# HOME/HERMES_HOME (asserted in ParityHome.env), so no real state is reachable. +pytestmark = [ + pytest.mark.skipif(not sys.platform.startswith("linux"), reason="process-tree checks use /proc"), + pytest.mark.live_system_guard_bypass, +] + +PROMPT = "Call the parity canary tool, then report." + +Driver = Callable[..., DriveResult] + +DRIVERS: dict[str, Driver] = { + "oneshot (-z)": _drive_cli.drive_oneshot, + "chat -q": _drive_cli.drive_chat_q, + "tui_gateway (stdio)": _drive_rpc.drive_tui_gateway, + "serve (Desktop WS)": _drive_rpc.drive_serve, + "gateway (fake adapter)": _drive_gateway.drive_gateway, + "api_server": _drive_gateway.drive_api_server, + "acp (stdio)": _drive_acp.drive_acp, + "cron (run-now)": _drive_cron.drive_cron, +} + +# Cells that are red on current main for a tracked, open bug. Strict: the test +# FAILS as soon as the cell turns green, so the entry is removed with the fix +# instead of silently masking a later regression of the same cell. +KNOWN_RED: dict[tuple[str, str], str] = { + # ACP builds its AIAgent without the configured agent.disabled_toolsets. + ("acp (stdio)", "toolset_restriction"): "#74582", +} + + +@dataclass +class Row: + entrypoint: str + cells: dict[str, bool] = field(default_factory=dict) + error: str | None = None + detail: str = "" + toolset: str | None = None + cwd_channel: str | None = None + other_survivors: list[str] = field(default_factory=list) + wall_s: float = 0.0 + + +def _run_row(entrypoint: str, root: Path) -> Row: + """Drive one entrypoint end to end and evaluate every cell. Never raises: the + row carries the error, and its own process tree is swept before returning.""" + row = Row(entrypoint) + started = time.monotonic() + srv = start_provider(uuid.uuid4().hex[:8]) + ph: ParityHome | None = None + try: + ph = build_parity_home(root, srv.base_url) + result = DRIVERS[entrypoint](ph, srv, PROMPT) + row.toolset, row.cwd_channel = result.toolset, result.cwd_channel + obs = collect_observation(entrypoint, ph, srv, result.final_text) + cells = check_invariants(obs, ph, toolset=result.toolset) + cells["answer_delivered"] = FINAL_ANSWER in (result.final_text or "") + cells["graceful_exit"] = result.graceful_exit + pids = mcp_pids(ph) + # Vacuity guard: the orphan check only means something if the tree existed. + cells["mcp_tree_spawned"] = bool(pids.get("server")) and bool(pids.get("grandchild")) + survivors = wait_no_orphans(ph, timeout=30.0) + cells["zero_mcp_orphans"] = not survivors + # Report-only: non-MCP descendants still alive (timing-dependent, e.g. a + # picker-prewarm `gh auth token` orphaned by a fast host exit). + row.other_survivors = describe_pids(wait_no_orphans(ph, timeout=10.0, mcp_only=False)) + row.cells = cells + stderr_tail = str(result.extra.get("stderr_tail", "")) + log = result.extra.get("stderr_log") + if log and Path(log).exists(): + stderr_tail = Path(log).read_text(encoding="utf-8", errors="replace") + row.detail = ( + f" tools offered: {sorted(obs.tool_names)}\n" + f" tool results: {[r[-300:] for r in obs.tool_results]}\n" + f" surviving MCP-tree pids: {describe_pids(survivors)} (mcp pid log {pids})\n" + f" final text: {(result.final_text or '')[-300:]!r}\n" + f" host exit code: {result.extra.get('exit_code')}\n" + f" host stderr tail: {stderr_tail[-1500:]}" + ) + except Exception as exc: # noqa: BLE001 - reported per row + row.error = f"{type(exc).__name__}: {exc}"[:4000] + finally: + if ph is not None: + kill_tagged(ph) + srv.stop() + row.wall_s = round(time.monotonic() - started, 1) + return row + + +@pytest.fixture(scope="module") +def matrix(tmp_path_factory: pytest.TempPathFactory) -> dict[str, Row]: + """All entrypoints driven CONCURRENTLY (each in its own home, provider and + process tree): the file's wall time is the slowest cold start, not the sum. + Everything, including the orphan sweep, completes during setup, while the + first test's live-guard bypass is in effect.""" + roots = {ep: tmp_path_factory.mktemp(f"parity{i}") for i, ep in enumerate(DRIVERS)} + with ThreadPoolExecutor(max_workers=len(DRIVERS), thread_name_prefix="parity") as pool: + futures = {ep: pool.submit(_run_row, ep, roots[ep]) for ep in DRIVERS} + rows = {ep: fut.result() for ep, fut in futures.items()} + out = os.environ.get("PARITY_TABLE_OUT") + if out: + with open(out, "a", encoding="utf-8") as fh: + for row in rows.values(): + fh.write(json.dumps({ + **asdict(row), "detail": None, + "known_red": {c: ref for (ep, c), ref in KNOWN_RED.items() if ep == row.entrypoint}, + }) + "\n") + return rows + + +@pytest.mark.parametrize("entrypoint", list(DRIVERS)) +def test_entrypoint_parity(entrypoint: str, matrix: dict[str, Row]) -> None: + row = matrix[entrypoint] + assert row.error is None, f"{entrypoint}: turn failed before the cells could be evaluated:\n{row.error}" + known = {cell for (ep, cell) in KNOWN_RED if ep == entrypoint} + fixed = sorted(cell for cell in known if row.cells.get(cell)) + assert not fixed, ( + f"{entrypoint}: {fixed} now green — drop the KNOWN_RED entry " + f"({[KNOWN_RED[(entrypoint, c)] for c in fixed]}) so the cell is enforced again") + failed = sorted(k for k, ok in row.cells.items() if not ok and k not in known) + assert not failed, f"{entrypoint}: parity cells red: {failed}\n{row.detail}" From ecbc87126378d855de7a0bc822e1f635ebb5496b Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:09:27 -0700 Subject: [PATCH 014/104] test: serve/tui_gateway boot handshake contract, Python side (C5) Spawns hermes serve exactly as the Desktop does and tui_gateway.entry as the Ink TUI does. serve: the first stdout line must be the READY sentinel, resolved to the live port by the Desktop's REAL parser (apps/desktop/electron/backend-ready.ts run under Node, no regex copy), and the port answers an authenticated /api/status. tui_gateway: the first stdout frame is a contract-valid gateway.ready and stdout stays pure JSON-RPC. Found setup.ready leaking onto serve stdout before READY. --- tests/e2e/core/parity/test_boot_contract.py | 140 ++++++++++++++++++++ 1 file changed, 140 insertions(+) create mode 100644 tests/e2e/core/parity/test_boot_contract.py diff --git a/tests/e2e/core/parity/test_boot_contract.py b/tests/e2e/core/parity/test_boot_contract.py new file mode 100644 index 0000000000..e4625528fc --- /dev/null +++ b/tests/e2e/core/parity/test_boot_contract.py @@ -0,0 +1,140 @@ +"""Backend boot handshake contract, Python side (issue class C5). + +The Desktop spawns ``hermes serve`` with stdout piped and learns the port ONLY from +the READY sentinel on stdout (``apps/desktop/electron/backend-ready.ts``); the Ink +TUI spawns ``tui_gateway.entry`` and treats stdout as a pure JSON-RPC stream whose +first frame is ``gateway.ready``. Both break the same way — the sentinel/frame goes +to the wrong stream, arrives after other bytes, or stops parsing — and every such +break presents to users as "backend failed to start" / a hung spinner. + +Invariants (spawned exactly as the real clients spawn them, cold, fake provider): + +* serve: the FIRST stdout line is the sentinel, the Desktop's own parser (run under + Node, not a Python copy of its regex) resolves a port from the stdout bytes seen + up to and including it, and that port is the live backend (authenticated + ``/api/status`` answers). +* tui_gateway: the FIRST stdout line is a JSON-RPC ``gateway.ready`` event whose + payload validates against the published contract, and every later stdout line + is JSON-RPC too (stray prints must land on stderr). +""" + +from __future__ import annotations + +import contextlib +import json +import sys +import time +import urllib.request +from pathlib import Path + +import pytest + +from tests.e2e.core.parity._boot_contract import desktop_parse, require_node +from tests.e2e.core.parity._drive_rpc import ( + READY_TIMEOUT, + RpcClient, + first_stdout_line, + spawn_serve, + spawn_tui_gateway, +) +from tests.e2e.core.parity._helpers import build_parity_home, kill_tagged, start_provider, terminate + +# Linux-only (/proc process-tree scans). The live-system guard bypass is needed +# ONLY for teardown: orphans reparented to init (the exact failure this suite +# hunts) sit outside the pytest subtree, and kill_tagged() signals nothing but +# PIDs carrying this run's unique PARITY_TREE_TAG. Children run with a tmp +# HOME/HERMES_HOME (asserted in ParityHome.env), so no real state is reachable. +pytestmark = [ + pytest.mark.skipif(not sys.platform.startswith("linux"), reason="process-tree cleanup uses /proc"), + pytest.mark.live_system_guard_bypass, +] + + +@contextlib.contextmanager +def boot_home(tmp_path: Path): + # A context manager used INSIDE the test body, not a fixture: the orphan sweep + # must run while the test's live-guard bypass is still in effect. + srv = start_provider("boot") + ph = build_parity_home(tmp_path, srv.base_url) + try: + yield ph + finally: + kill_tagged(ph) + srv.stop() + + +def test_serve_ready_sentinel_is_first_stdout_line_and_names_the_live_port(tmp_path: Path) -> None: + require_node() + with boot_home(tmp_path) as home: + _check_serve_ready(home) + + +def _check_serve_ready(home) -> None: + sp = spawn_serve(home) + try: + first = first_stdout_line(sp) + # The Desktop parser sees exactly the stream so far; with nothing before + # the sentinel, the first line alone must already resolve. + verdict = desktop_parse(first) + assert "port" in verdict, ( + f"first stdout line is not a READY sentinel the Desktop accepts: {first!r} -> {verdict}\n" + f"(stdout must carry nothing before READY)\nstderr tail:\n{sp.cap.stderr[-1500:]}") + port = int(verdict["port"]) + + req = urllib.request.Request(f"http://127.0.0.1:{port}/api/status", + headers={"X-Hermes-Session-Token": sp.token}) + with urllib.request.urlopen(req, timeout=30) as resp: + assert resp.status == 200, resp.status + json.loads(resp.read()) + finally: + terminate(sp.proc, timeout=60) + + +def test_tui_gateway_first_stdout_frame_is_contract_valid_gateway_ready(tmp_path: Path) -> None: + with boot_home(tmp_path) as home: + _check_tui_gateway_ready(home) + + +def _check_tui_gateway_ready(home) -> None: + from tui_gateway.contracts.registry import EVENTS + + proc, cap = spawn_tui_gateway(home) + try: + deadline = time.monotonic() + READY_TIMEOUT + while not cap.stdout_seen: + assert proc.poll() is None, f"tui_gateway exited {proc.returncode}\n{cap.stderr[-2000:]}" + assert time.monotonic() < deadline, f"no stdout frame within {READY_TIMEOUT}s\n{cap.stderr[-2000:]}" + time.sleep(0.05) + first = cap.stdout_seen[0] + try: + frame = json.loads(first) + except ValueError: + frame = None + assert isinstance(frame, dict), f"first stdout line of tui_gateway is not JSON-RPC: {first!r}" + params = frame.get("params") or {} + assert frame.get("method") == "event" and params.get("type") == "gateway.ready", frame + ready = EVENTS["gateway.ready"] + if ready.payload is not None: + ready.payload.model_validate(params.get("payload") or {}) + + # Exercise one RPC so post-ready output has a chance to leak, then check + # the whole stream stayed JSON-RPC. + def send(line: str) -> None: + assert proc.stdin is not None + proc.stdin.write(line + "\n") + proc.stdin.flush() + + rpc = RpcClient(send, cap.stdout_lines) + rpc.call("session.create", {}, timeout=READY_TIMEOUT) + for line in list(cap.stdout_seen): + try: + assert json.loads(line).get("jsonrpc") == "2.0" + except ValueError: + pytest.fail(f"non-JSON bytes on tui_gateway stdout: {line!r}") + finally: + if proc.stdin is not None and not proc.stdin.closed: + proc.stdin.close() + try: + proc.wait(timeout=60) + except Exception: + terminate(proc) From 56ce6e2204fe18a263438d76048f67496a90f3a5 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:46:13 -0700 Subject: [PATCH 015/104] test: MCP server lifecycle failure transitions on the long-lived host (C15) tui_gateway host + the real stdio MCP fixture: - fast death: the server crashes mid-call leaving a helper that inherited its stdio (npx/uvx wrapper shape). The call must surface as an outcome-uncertain tool error exactly once (never replayed onto the respawned server), the host survives, the turn completes within a bound 12x under the configured call timeout, and shutdown reaps the helper. - host crash: SIGKILL the host after a turn; the death supervisor must still reap the MCP server and its grandchild. Also pins mcp_discovery_timeout in the fixture home: interactive surfaces wait only ~1.5 s for discovery by design, which a loaded box exceeds, so the first-turn-complete invariant needs the documented knob raised. --- tests/e2e/core/parity/_helpers.py | 25 ++-- tests/e2e/core/parity/fixture_mcp_server.py | 21 ++- tests/e2e/core/parity/test_mcp_lifecycle.py | 144 ++++++++++++++++++++ 3 files changed, 181 insertions(+), 9 deletions(-) create mode 100644 tests/e2e/core/parity/test_mcp_lifecycle.py diff --git a/tests/e2e/core/parity/_helpers.py b/tests/e2e/core/parity/_helpers.py index 91ef212747..fb0c6dbc2b 100644 --- a/tests/e2e/core/parity/_helpers.py +++ b/tests/e2e/core/parity/_helpers.py @@ -119,7 +119,8 @@ class ParityHome: self.update_config(lambda cfg: cfg.setdefault("terminal", {}).__setitem__("cwd", str(self.project))) -def build_parity_home(root: Path, base_url: str, *, grandchild: bool = True) -> ParityHome: +def build_parity_home(root: Path, base_url: str, *, grandchild: bool = True, + death_tool: bool = False, mcp_timeout: int = 60) -> ParityHome: """Write the full fixture home under ``root`` (a tmp_path).""" home = root / "home" hermes_home = home / ".hermes" @@ -149,14 +150,22 @@ def build_parity_home(root: Path, base_url: str, *, grandchild: bool = True) -> "PARITY_MCP_CANARY": c["mcp"], "PARITY_MCP_PID_LOG": str(ph.pid_log), "PARITY_MCP_SPAWN_GRANDCHILD": "1" if grandchild else "0", + "PARITY_MCP_DEATH_TOOL": "1" if death_tool else "0", "PARITY_TREE_TAG": tag, # Only the MCP server and its descendants carry this one. "PARITY_MCP_TREE_TAG": tag, "PYTHONPATH": str(REPO_ROOT), }, "connect_timeout": 60, - "timeout": 60, + "timeout": mcp_timeout, }} + # Interactive surfaces wait only ~1.5 s (mcp_discovery_timeout) for MCP discovery + # before the first agent build, BY DESIGN (a slow server must not block the + # shell; late tools arrive via refresh). Under a loaded box the fixture server's + # import alone can exceed that, so pin the documented knob high: the join returns + # the instant discovery finishes, and the first turn is deterministically complete. + cfg["mcp_discovery_timeout"] = 120 + cfg["mcp_single_query_discovery_timeout"] = 120 # Keep turns hermetic and short: no title/aux model chatter decides anything here. cfg.setdefault("display", {})["compact"] = True (hermes_home / "config.yaml").write_text(yaml.safe_dump(cfg, sort_keys=False), encoding="utf-8") @@ -212,7 +221,7 @@ def build_parity_home(root: Path, base_url: str, *, grandchild: bool = True) -> # Scripted provider ----------------------------------------------------------- -def parity_responder(nonce: str) -> Callable[[dict[str, Any]], Any]: +def parity_responder(nonce: str, tool: str = MCP_TOOL_NAME) -> Callable[[dict[str, Any]], Any]: """Stateless script: call the MCP tool until a tool result exists, then answer. Stateless so a retried request (or a second concurrent entrypoint request) @@ -224,16 +233,16 @@ def parity_responder(nonce: str) -> Callable[[dict[str, Any]], Any]: if any(m.get("role") == "tool" for m in body.get("messages") or []): return Text(FINAL_ANSWER) args = {"nonce": nonce} - if MCP_TOOL_NAME in _tool_names(body) or "tool_call" not in _tool_names(body): - return ToolCall(MCP_TOOL_NAME, args) + if tool in _tool_names(body) or "tool_call" not in _tool_names(body): + return ToolCall(tool, args) # Tool Search active: MCP tools sit in the deferred catalog behind the bridge. - return ToolCall("tool_call", {"calls": [{"name": MCP_TOOL_NAME, "arguments": args}]}) + return ToolCall("tool_call", {"calls": [{"name": tool, "arguments": args}]}) return respond -def start_provider(nonce: str) -> FakeLLMServer: - srv = FakeLLMServer(parity_responder(nonce)) +def start_provider(nonce: str, tool: str = MCP_TOOL_NAME) -> FakeLLMServer: + srv = FakeLLMServer(parity_responder(nonce, tool)) srv.start() return srv diff --git a/tests/e2e/core/parity/fixture_mcp_server.py b/tests/e2e/core/parity/fixture_mcp_server.py index 9ecc19efea..0749acef23 100644 --- a/tests/e2e/core/parity/fixture_mcp_server.py +++ b/tests/e2e/core/parity/fixture_mcp_server.py @@ -8,7 +8,10 @@ plumbing) instead of a registered-but-dead schema. ``PARITY_MCP_SPAWN_GRANDCHILD=1`` makes the server fork a long-lived grandchild at startup (the shape of npx/uvx wrappers and servers with worker helpers) so shutdown tests can prove Hermes reaps the whole process tree, not just its direct -child. Every PID the server owns is appended to ``PARITY_MCP_PID_LOG`` so the test +child. ``PARITY_MCP_DEATH_TOOL=1`` adds ``parity_die``, which kills the server while its +own call is in flight, leaving a helper holding the stdio pipe open so no EOF ever +arrives (the fast-death supervisor's target shape, #81995). +Every PID the server owns is appended to ``PARITY_MCP_PID_LOG`` so the test can check liveness of exactly the processes this fixture created. """ @@ -17,6 +20,8 @@ from __future__ import annotations import os import subprocess import sys +import threading +import time def _log_pid(kind: str, pid: int) -> None: @@ -53,6 +58,20 @@ def main() -> None: """Return the parity fixture canary (echoing the caller's nonce).""" return f"{os.environ.get('PARITY_MCP_CANARY', 'NO-CANARY')}:{nonce}" + if os.environ.get("PARITY_MCP_DEATH_TOOL") == "1": + @server.tool() + def parity_die(nonce: str = "") -> str: + """Crash this MCP server mid-call: the RPC never gets a response.""" + # A helper that inherits our stdio keeps the pipe open after we die (the + # npx/uvx-wrapper shape), so the client never sees EOF: only the child + # liveness watch can end the call. + holder = subprocess.Popen([sys.executable, "-c", "import time\nwhile True: time.sleep(60)"]) + _log_pid("pipe_holder", holder.pid) + _log_pid("dying_server", os.getpid()) + threading.Timer(0.3, lambda: os._exit(3)).start() + time.sleep(3600) + return "unreachable" + server.run(transport="stdio") diff --git a/tests/e2e/core/parity/test_mcp_lifecycle.py b/tests/e2e/core/parity/test_mcp_lifecycle.py new file mode 100644 index 0000000000..408e9e7250 --- /dev/null +++ b/tests/e2e/core/parity/test_mcp_lifecycle.py @@ -0,0 +1,144 @@ +"""MCP server lifecycle on the long-lived host (issue class C15). + +Uses the tui_gateway stdio host (the Ink TUI's backend; the Desktop's WS backend +runs the same MCP runtime) with the real stdio MCP fixture: + +* fast death: the MCP server crashes WHILE its call is in flight, leaving a helper + that inherited its stdio (npx/uvx wrapper shape) (#81995). The call must fail + as a tool error EXACTLY once (outcome uncertain: never silently replayed onto + the respawned server), the host must survive, the turn complete well inside the configured MCP call + timeout (bound 300 s: >10x the ~18 s observed idle, 12x under the 3600 s timeout), and a normal host + shutdown afterwards still reaps the pipe holder. +* host crash: the host is SIGKILLed after a turn (no shutdown code runs at all). + The MCP server AND its grandchild must still be reaped (death supervisor + + process-group kill), because a crashed Desktop/TUI backend otherwise leaves a + pile of MCP processes behind on every restart. + +The parity matrix covers the normal-shutdown reap and the healthy call on every +entrypoint; this file covers the two failure transitions. +""" + +from __future__ import annotations + +import contextlib +import os +import signal +import sys +import time +import uuid +from pathlib import Path + +import pytest + +from tests.e2e.core.parity._drive_rpc import READY_TIMEOUT, RpcClient, spawn_tui_gateway +from tests.e2e.core.parity._helpers import ( + FINAL_ANSWER, + build_parity_home, + collect_observation, + describe_pids, + kill_tagged, + mcp_pids, + start_provider, + terminate, + wait_no_orphans, + wait_until, +) + +# Same rationale as test_entrypoint_parity: Linux /proc scans, and the bypass is +# for teardown of reparented orphans carrying this run's unique tag only. +pytestmark = [ + pytest.mark.skipif(not sys.platform.startswith("linux"), reason="process-tree checks use /proc"), + pytest.mark.live_system_guard_bypass, +] + +MCP_DIE_TOOL = "mcp__parity__parity_die" +MCP_CALL_TIMEOUT = 3600 +FAST_DEATH_BOUND = 300.0 +REAP_BOUND = 30.0 + + +@contextlib.contextmanager +def tui_host(tmp_path: Path, *, tool: str | None = None, **home_kwargs): + nonce = uuid.uuid4().hex[:8] + srv = start_provider(nonce, tool) if tool else start_provider(nonce) + ph = build_parity_home(tmp_path, srv.base_url, **home_kwargs) + proc, cap = spawn_tui_gateway(ph) + + def send(line: str) -> None: + assert proc.stdin is not None + proc.stdin.write(line + "\n") + proc.stdin.flush() + + try: + rpc = RpcClient(send, cap.stdout_lines) + rpc.wait_event("gateway.ready", timeout=READY_TIMEOUT) + yield ph, srv, proc, cap, rpc + finally: + if proc.poll() is None: + if proc.stdin is not None and not proc.stdin.closed: + with contextlib.suppress(OSError): + proc.stdin.close() + try: + proc.wait(timeout=60) + except Exception: + terminate(proc) + kill_tagged(ph) + srv.stop() + + +def _alive(pid: int) -> bool: + try: + with open(f"/proc/{pid}/stat", "rb") as fh: + return fh.read().rsplit(b")", 1)[1].split()[0] != b"Z" + except OSError: + return False + + +def _turn(rpc: RpcClient, timeout: float) -> str: + sid = rpc.call("session.create", {}, timeout=READY_TIMEOUT)["session_id"] + rpc.call("prompt.submit", {"session_id": sid, "text": "Use the parity tool, then report."}, timeout=READY_TIMEOUT) + done = rpc.wait_event("message.complete", lambda e: e.get("session_id") == sid, timeout=timeout) + return str((done.get("payload") or {}).get("text") or "") + + +def test_mcp_server_death_mid_call_fails_fast_and_the_turn_completes(tmp_path: Path) -> None: + with tui_host(tmp_path, tool=MCP_DIE_TOOL, grandchild=False, death_tool=True, + mcp_timeout=MCP_CALL_TIMEOUT) as (ph, srv, proc, cap, rpc): + started = time.monotonic() + text = _turn(rpc, timeout=FAST_DEATH_BOUND) + elapsed = time.monotonic() - started + + obs = collect_observation("tui_gateway", ph, srv, text) + assert FINAL_ANSWER in text, f"turn did not complete after the MCP server died: {text[-300:]!r}" + assert obs.tool_results, "the dead MCP call produced no tool result for the model" + assert not any(ph.canaries["mcp"] in r for r in obs.tool_results), obs.tool_results + assert elapsed < FAST_DEATH_BOUND, elapsed + # The server really died mid-call (not a schema/dispatch error before the RPC). + dying = mcp_pids(ph).get("dying_server") or [] + # Exactly once: a call that died mid-flight may have had side effects, so it + # must surface as outcome-uncertain, never be silently replayed. + assert len(dying) == 1, ( + f"parity_die ran {len(dying)}x (want exactly 1): {mcp_pids(ph)}; results {obs.tool_results}") + wait_until(lambda: not _alive(dying[0]), REAP_BOUND, "fixture MCP server exit") + assert proc.poll() is None, f"host died with the MCP server:\n{cap.stderr[-2000:]}" + + assert mcp_pids(ph).get("pipe_holder"), "fixture pipe holder never started" + proc.stdin.close() + proc.wait(timeout=60) + survivors = wait_no_orphans(ph, timeout=REAP_BOUND) + assert not survivors, f"dead server's pipe holder survived host shutdown: {describe_pids(survivors)}" + + +def test_host_crash_still_reaps_the_mcp_server_and_its_grandchild(tmp_path: Path) -> None: + with tui_host(tmp_path) as (ph, srv, proc, cap, rpc): + text = _turn(rpc, timeout=READY_TIMEOUT) + assert FINAL_ANSWER in text, text[-300:] + pids = mcp_pids(ph) + assert pids.get("server") and pids.get("grandchild"), f"MCP tree never spawned: {pids}" + + os.kill(proc.pid, signal.SIGKILL) # no atexit, no shutdown hook, no finally + proc.wait(timeout=30) + + survivors = wait_no_orphans(ph, timeout=REAP_BOUND) + assert not survivors, ( + f"MCP tree survived a host crash for {REAP_BOUND}s: {describe_pids(survivors)} (pid log {pids})") From 766f3983207b6138a837d128a08f9968e9b8c1df Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:12:22 -0700 Subject: [PATCH 016/104] test: chaos provider + deadline oracle for a real AIAgent turn (C8 liveness) Class C8 (agent-turn liveness): hangs, runaway retries, lost tool results, wedged agents, orphan processes. A real AIAgent + SessionDB runs in a child process with config.yaml liveness knobs (agent.api_max_retries, providers.custom request/stale timeouts, agent.max_turns) against the scripted loopback provider. 22 fault modes (hang, reasoning-model stall, 500/429/400/garbage forever, drop mid-stream, hang then recover, slow trickle, malformed/unknown tool calls, 8-way parallel, hung/stdin/huge tools, background swarms, endless tool loop, interrupts) are each followed by a probe turn on the same agent and checked against the same invariants: bounded termination, bounded provider calls, reusability with history kept, every tool_call answered (in state.db and on the wire), no surviving tagged process, integrity_check ok. --- tests/e2e/core/chaos/__init__.py | 0 tests/e2e/core/chaos/_agent_driver.py | 78 +++ tests/e2e/core/chaos/_helpers.py | 299 ++++++++++ .../core/chaos/test_agent_turn_liveness.py | 523 ++++++++++++++++++ 4 files changed, 900 insertions(+) create mode 100644 tests/e2e/core/chaos/__init__.py create mode 100644 tests/e2e/core/chaos/_agent_driver.py create mode 100644 tests/e2e/core/chaos/_helpers.py create mode 100644 tests/e2e/core/chaos/test_agent_turn_liveness.py diff --git a/tests/e2e/core/chaos/__init__.py b/tests/e2e/core/chaos/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/tests/e2e/core/chaos/_agent_driver.py b/tests/e2e/core/chaos/_agent_driver.py new file mode 100644 index 0000000000..839de818c0 --- /dev/null +++ b/tests/e2e/core/chaos/_agent_driver.py @@ -0,0 +1,78 @@ +"""Child-process driver: one real AIAgent + real SessionDB, several turns, stdin control. + +Run as ``python -m tests.e2e.core.chaos._agent_driver `` with a hermetic +env (see ``_helpers.hermetic_env``). The agent is built the way the CLI builds it: +config.yaml under HERMES_HOME supplies the provider, retry and timeout knobs, and +``agent.max_turns`` becomes the iteration budget. Each turn feeds the previous +turn's messages back as ``conversation_history`` (the CLI's multi-turn contract). + +Protocol (one JSON object per stdout line, prefixed ``CHAOS ``; everything else on +stdout/stderr is the agent's own output and is ignored by the parent): + {"ev": "ready"} agent built + {"ev": "turn_start", "i": n} + {"ev": "turn_end", "i": n, "failed": .., "interrupted": .., "completed": .., + "final": "...", "exit_reason": "..."} + {"ev": "closed"} agent.close() returned +stdin lines: ``interrupt`` -> ``agent.interrupt()`` (the CLI's Ctrl+C / Esc path). +""" + +from __future__ import annotations + +import json +import sys +import threading + + +def _emit(**payload: object) -> None: + sys.__stdout__.write("CHAOS " + json.dumps(payload, default=str) + "\n") + sys.__stdout__.flush() + + +def main(spec_path: str) -> None: + spec = json.loads(open(spec_path, encoding="utf-8").read()) + + from hermes_cli.config import load_config + from hermes_state import SessionDB + from run_agent import AIAgent + + cfg = load_config() + model_cfg = cfg.get("model") or {} + agent = AIAgent( + provider=model_cfg.get("provider"), + base_url=model_cfg.get("base_url"), + api_key="sk-fake-chaos", + model=model_cfg.get("default"), + max_iterations=int((cfg.get("agent") or {}).get("max_turns") or 90), + session_db=SessionDB(), + session_id=spec["session_id"], + quiet_mode=True, + platform="cli", + ) + + def _control() -> None: + for line in sys.stdin: + if line.strip() == "interrupt": + agent.interrupt("chaos: user interrupt") + + threading.Thread(target=_control, name="chaos-control", daemon=True).start() + _emit(ev="ready") + + history: list = [] + for i, message in enumerate(spec["turns"]): + _emit(ev="turn_start", i=i) + result = agent.run_conversation(message, conversation_history=history) + history = result.get("messages") or history + _emit( + ev="turn_end", i=i, + failed=bool(result.get("failed")), + interrupted=bool(result.get("interrupted")), + completed=bool(result.get("completed")), + final=(result.get("final_response") or "")[:2000], + exit_reason=result.get("turn_exit_reason"), + ) + agent.close() + _emit(ev="closed") + + +if __name__ == "__main__": + main(sys.argv[1]) diff --git a/tests/e2e/core/chaos/_helpers.py b/tests/e2e/core/chaos/_helpers.py new file mode 100644 index 0000000000..e21f7f76a2 --- /dev/null +++ b/tests/e2e/core/chaos/_helpers.py @@ -0,0 +1,299 @@ +"""Shared oracle helpers for the chaos (agent-turn liveness) E2E lane. + +The chaos suites drive a REAL Hermes surface (AIAgent child process, GatewayRunner, +tui_gateway JSON-RPC subprocess) against ``tests/fakes/fake_llm_provider`` in a fault +mode and then check the same liveness invariants everywhere: + +* bounded termination — the turn ends (answer or surfaced error) within the + configured deadline plus a generous epsilon, never "whenever the fault clears"; +* bounded provider calls — no retry storm; +* reusability — the same session accepts and answers a new message afterwards; +* history integrity — every assistant ``tool_call`` has a matching ``tool`` row, + both in state.db and in what is sent back to the model; +* process hygiene — no process carrying the scenario's tag survives the session. + +Everything here is surface-agnostic; surface drivers live next to their test file. +""" + +from __future__ import annotations + +import json +import os +import signal +import sqlite3 +import sys +import time +import uuid +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Iterable + +REPO_ROOT = Path(__file__).resolve().parents[4] + +# One scenario's liveness budget. The fake faults last an HOUR; the configured +# deadlines below are seconds, so a turn that is still running at DEADLINE is +# waiting on the fault, not merely slow — margins are >10x the healthy runtime. +STALE_TIMEOUT_S = 3 +REQUEST_TIMEOUT_S = 5 +API_MAX_RETRIES = 2 +MAX_TURNS = 8 +TOOL_TIMEOUT_S = 3 +TURN_DEADLINE_S = 90.0 +# A retry storm is dozens-to-thousands of calls (#92450: ~64/s). Healthy fault +# handling with API_MAX_RETRIES=2 stays in single digits (stream retries x API retries). +MAX_PROVIDER_CALLS_PER_FAILED_TURN = 16 +INTERRUPT_DEADLINE_S = 15.0 +# Interrupt scenarios run with deadlines far beyond INTERRUPT_DEADLINE_S so only a +# working interrupt can end the turn in time. +LONG_TIMEOUT_S = 600 +PROCESS_REAP_S = 15.0 + +PLAIN_MODEL = "fake-model" +# A slug with a reasoning stale-timeout floor: an explicit +# providers..stale_timeout_seconds must still win over the floor (#115024). +REASONING_MODEL = "deepseek-r1" + + +def new_tag() -> str: + return f"chaos{uuid.uuid4().hex[:12]}" + + +def chaos_config( + base_url: str, + *, + model: str = PLAIN_MODEL, + stale_timeout: float = STALE_TIMEOUT_S, + request_timeout: float = REQUEST_TIMEOUT_S, + api_max_retries: int = API_MAX_RETRIES, + max_turns: int = MAX_TURNS, + extra: str = "", +) -> str: + """config.yaml text with every liveness knob short and explicit (real keys: + ``agent.api_max_retries``, ``agent.auto_recovery_cycles``, ``agent.max_turns``, + ``providers..request_timeout_seconds`` / ``stale_timeout_seconds``).""" + return ( + "model:\n" + " provider: custom\n" + f" base_url: {base_url}\n" + f" default: {model}\n" + " context_length: 128000\n" + "agent:\n" + f" api_max_retries: {api_max_retries}\n" + " auto_recovery_cycles: 0\n" + f" max_turns: {max_turns}\n" + "providers:\n" + " custom:\n" + f" request_timeout_seconds: {request_timeout}\n" + f" stale_timeout_seconds: {stale_timeout}\n" + "compression:\n" + " enabled: false\n" + "memory:\n" + " memory_enabled: false\n" + " user_profile_enabled: false\n" + + extra + ) + + +def write_chaos_home(root: Path, base_url: str, **cfg: Any) -> tuple[Path, Path]: + """Create ``root/home`` (fake $HOME) and ``root/home/.hermes`` (HERMES_HOME).""" + home = root / "home" + hermes_home = home / ".hermes" + hermes_home.mkdir(parents=True, exist_ok=True) + (hermes_home / "config.yaml").write_text(chaos_config(base_url, **cfg), encoding="utf-8") + (hermes_home / ".env").write_text("OPENAI_API_KEY=sk-fake-chaos\n", encoding="utf-8") + return home, hermes_home + + +def hermetic_env(home: Path, hermes_home: Path, tag: str) -> dict[str, str]: + """Child env: fake HOME/HERMES_HOME, no real credentials, repo importable, and a + tag every descendant inherits so the orphan scan can find it after reparenting.""" + env = { + k: v for k, v in os.environ.items() + if not ( + k.endswith(("_API_KEY", "_TOKEN", "_SECRET")) + or k.startswith(("HERMES_", "OPENROUTER", "ANTHROPIC", "OPENAI", "NOUS_")) + or k in {"PYTEST_CURRENT_TEST"} + ) + } + env.update({ + "HOME": str(home), + "HERMES_HOME": str(hermes_home), + "OPENAI_API_KEY": "sk-fake-chaos", + "PYTHONPATH": str(REPO_ROOT), + "PYTHONUNBUFFERED": "1", + "PYTHONFAULTHANDLER": "1", + "CHAOS_TAG": tag, + # HOME *is* the tmp root, so the live-DB guard (pytest ancestry) would read the + # tmp state.db as "production"; the documented child opt-out is safe here. + "HERMES_STATE_DB_GUARD_BYPASS": "1", + "TZ": "UTC", + "NO_COLOR": "1", + }) + return env + + +# ── process hygiene ───────────────────────────────────────────────────────── + + +def tagged_pids(tag: str, *, exclude: Iterable[int] = ()) -> list[int]: + """PIDs whose environment carries ``CHAOS_TAG=`` (survives reparenting to init).""" + needle = f"CHAOS_TAG={tag}".encode() + skip = {os.getpid(), *exclude} + found: list[int] = [] + for entry in os.listdir("/proc"): + if not entry.isdigit() or int(entry) in skip: + continue + try: + with open(f"/proc/{entry}/environ", "rb") as fh: + data = fh.read() + with open(f"/proc/{entry}/stat", "rb") as fh: + state = fh.read().rsplit(b")", 1)[1].split()[0] + except OSError: + continue + if state == b"Z": + continue # zombie: already dead, only waiting on its parent's reap + if needle in data.split(b"\0"): + found.append(int(entry)) + return found + + +def describe_pids(pids: Iterable[int]) -> list[str]: + out = [] + for pid in pids: + try: + cmd = Path(f"/proc/{pid}/cmdline").read_bytes().replace(b"\0", b" ").decode(errors="replace") + except OSError: + cmd = "?" + out.append(f"{pid}: {cmd[:160]}") + return out + + +def wait_no_tagged(tag: str, timeout: float = PROCESS_REAP_S) -> list[int]: + """Poll until no tagged process remains; return survivors at the deadline.""" + deadline = time.monotonic() + timeout + while True: + left = tagged_pids(tag) + if not left or time.monotonic() >= deadline: + return left + time.sleep(0.1) + + +def kill_tagged(tag: str) -> None: + """Test cleanup only: SIGKILL anything still carrying our tag (never pkill -f). + + The conftest live-system guard refuses os.kill on PIDs reparented out of the test's + subtree — exactly the orphans a red run leaves behind. The tag proves they are ours, + so fall back to a pidfd signal (also immune to PID reuse) instead of leaking them.""" + for pid in tagged_pids(tag): + try: + os.kill(pid, signal.SIGKILL) + except OSError: + pass + except RuntimeError: + try: + fd = os.pidfd_open(pid) + except OSError: + continue + try: + signal.pidfd_send_signal(fd, signal.SIGKILL) + except OSError: + pass + finally: + os.close(fd) + + +# ── history integrity ─────────────────────────────────────────────────────── + + +def _call_ids(msg: dict[str, Any]) -> list[str]: + calls = msg.get("tool_calls") or [] + if isinstance(calls, str): + try: + calls = json.loads(calls) + except json.JSONDecodeError: + return [""] + return [c.get("id") or c.get("call_id") or "" for c in calls if isinstance(c, dict)] + + +def unanswered_tool_calls(messages: list[dict[str, Any]]) -> list[str]: + """tool_call ids with no ``tool`` message in the block right after their assistant. + + Wire rule (every provider): an assistant ``tool_calls`` message must be followed by + one ``tool`` result per id before the next assistant/user message. + """ + missing: list[str] = [] + for i, msg in enumerate(messages): + if msg.get("role") != "assistant": + continue + ids = _call_ids(msg) + if not ids: + continue + answered = set() + for follow in messages[i + 1:]: + if follow.get("role") != "tool": + break + answered.add(follow.get("tool_call_id")) + missing.extend(tid for tid in ids if tid not in answered) + return missing + + +def tool_results_by_id(messages: list[dict[str, Any]]) -> dict[str, str]: + out: dict[str, str] = {} + for msg in messages: + if msg.get("role") == "tool": + content = msg.get("content") + if not isinstance(content, str): + content = json.dumps(content) + out[msg.get("tool_call_id") or ""] = content + return out + + +def persisted_messages(state_db: Path, session_id: str | None = None) -> list[dict[str, Any]]: + """Active message rows from a real state.db, oldest first, OpenAI-shaped. + + With ``session_id=None`` the newest session that has messages is used.""" + conn = sqlite3.connect(f"file:{state_db}?mode=ro", uri=True, timeout=10) + try: + conn.row_factory = sqlite3.Row + if session_id is None: + row = conn.execute( + "SELECT session_id FROM messages ORDER BY id DESC LIMIT 1").fetchone() + if row is None: + return [] + session_id = row["session_id"] + rows = conn.execute( + "SELECT role, content, tool_call_id, tool_calls FROM messages " + "WHERE session_id = ? AND active = 1 ORDER BY id", (session_id,)).fetchall() + finally: + conn.close() + msgs = [] + for r in rows: + m: dict[str, Any] = {"role": r["role"], "content": r["content"]} + if r["tool_call_id"]: + m["tool_call_id"] = r["tool_call_id"] + if r["tool_calls"]: + m["tool_calls"] = r["tool_calls"] + msgs.append(m) + return msgs + + +def integrity_ok(state_db: Path) -> str: + conn = sqlite3.connect(f"file:{state_db}?mode=ro", uri=True, timeout=10) + try: + return conn.execute("PRAGMA integrity_check").fetchone()[0] + finally: + conn.close() + + +@dataclass +class Timing: + started: float + ended: float | None = None + + @property + def elapsed(self) -> float: + return (self.ended or time.monotonic()) - self.started + + +def python_exe() -> str: + return sys.executable diff --git a/tests/e2e/core/chaos/test_agent_turn_liveness.py b/tests/e2e/core/chaos/test_agent_turn_liveness.py new file mode 100644 index 0000000000..ad03410008 --- /dev/null +++ b/tests/e2e/core/chaos/test_agent_turn_liveness.py @@ -0,0 +1,523 @@ +"""C8 chaos: agent-turn liveness of a REAL AIAgent (child process) under provider and tool faults. + +Each scenario starts ``_agent_driver`` in its own process with a hermetic HOME, a real +SessionDB on disk, config.yaml carrying the real liveness knobs (``agent.api_max_retries``, +``agent.auto_recovery_cycles``, ``agent.max_turns``, ``providers.custom.request_timeout_seconds`` +/ ``stale_timeout_seconds``) and the scripted loopback provider as the only fake. The driver +runs the faulted turn, then a PROBE turn on the same agent with the faulted turn as history, +then ``agent.close()``, then exits. Scenarios run concurrently and are asserted per test. + +Invariants after EVERY fault (the class, not one bug): + +* bounded termination: the faulted turn returns within TURN_DEADLINE_S (the faults last an + hour) or within INTERRUPT_DEADLINE_S of ``agent.interrupt()`` when every configured timeout + is LONG_TIMEOUT_S, with a surfaced answer or error; +* bounded provider calls: no retry storm (#92450); a 400 is not retried; a healthy slow + stream is not retried at all; +* reusability: the same agent answers the PROBE turn (no stuck interrupt flag, wedged client, + or tripped breaker), and the PROBE request still carries the faulted turn; +* history integrity: every assistant tool_call has its own tool result in state.db and in + what is sent back to the model; in an 8-way parallel batch each id carries ITS output (#93251); +* process hygiene: the driver exits on its own after close() and no process carrying the + scenario tag survives (hung tools, background swarms, setsid escapees); +* state.db integrity_check == ok. +""" + +from __future__ import annotations + +import json +import queue +import re +import subprocess +import sys +import threading +import time +import uuid +from concurrent.futures import Future, ThreadPoolExecutor +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Callable + +import pytest + +from tests.e2e.core.chaos._helpers import ( + API_MAX_RETRIES, + INTERRUPT_DEADLINE_S, + LONG_TIMEOUT_S, + MAX_PROVIDER_CALLS_PER_FAILED_TURN, + PROCESS_REAP_S, + REASONING_MODEL, + TOOL_TIMEOUT_S, + TURN_DEADLINE_S, + describe_pids, + hermetic_env, + integrity_ok, + kill_tagged, + new_tag, + persisted_messages, + python_exe, + tagged_pids, + tool_results_by_id, + unanswered_tool_calls, + wait_no_tagged, + write_chaos_home, +) +from tests.fakes.fake_llm_provider import ( + DropMidStream, + Error, + FakeLLMServer, + Hang, + Raw, + Response, + StallMidStream, + Text, + ToolCall, +) + +pytestmark = [pytest.mark.skipif(sys.platform != "linux", reason="orphan scan reads /proc")] + +BOOT_DEADLINE_S = 120.0 # cold import of the agent stack on a loaded runner +PROBE_DEADLINE_S = 60.0 +EXIT_DEADLINE_S = 30.0 +SETTLE_DEADLINE_S = 60.0 +SCENARIO_WORKERS = 6 +PARALLEL_CALLS = 8 +HUGE_OUTPUT_BYTES = 30_000_000 +# What may be persisted / re-sent for a 30 MB tool output (it is truncated or spilled). +MAX_PERSISTED_TOOL_CHARS = 1_000_000 +MAX_REQUEST_BYTES = 2_000_000 +TRICKLE_CHUNKS = 40 +TRICKLE_DELAY_S = 0.25 # 12x inside the 3 s stale timeout; total ~10 s > 3x the timeout +LOOP_BUDGET = 4 + +_PROBE_RE = re.compile(r"\[\[probe:([0-9a-f]+)\]\]") +_FAULT_RE = re.compile(r"\[\[chaos:([a-z0-9_]+)\]\]") + +LONG = {"request_timeout": LONG_TIMEOUT_S, "stale_timeout": LONG_TIMEOUT_S} + + +# ── scenario table ─────────────────────────────────────────────────────────── + + +@dataclass +class Ctx: + """What a fault responder may look at: the request, its index in the faulted turn.""" + + nonce: str + n: int + messages: list[dict[str, Any]] + + +def _after_tools(first: Callable[[Ctx], Response], then: str = "tools settled") -> Callable[[Ctx], Response]: + def respond(ctx: Ctx) -> Response: + if ctx.messages and ctx.messages[-1].get("role") == "tool": + return Text(f"{then} {ctx.nonce}") + return first(ctx) + return respond + + +def _token(nonce: str, i: int) -> str: + return f"tok{nonce}n{i}" + + +def _parallel(ctx: Ctx) -> ToolCall: + calls = [("terminal", {"command": f"echo {_token(ctx.nonce, i)}"}) for i in range(PARALLEL_CALLS)] + return ToolCall(calls[0][0], calls[0][1], parallel=calls[1:]) + + +def _trickle_text(nonce: str) -> str: + body = f"slow-{nonce}-" + return (body * (TRICKLE_CHUNKS * 2 // len(body) + 1))[: TRICKLE_CHUNKS * 2] + + +@dataclass +class Scenario: + id: str + fault: Callable[[Ctx], Response] + cfg: dict[str, Any] = field(default_factory=dict) + # "error" | "answer" (final contains ``answer`` % nonce) | "ended" | "interrupted" + expect: str = "error" + answer: str = "" + call_bound: int = MAX_PROVIDER_CALLS_PER_FAILED_TURN + exact_calls: int | None = None + # Poll condition (run, ctx-free) that, once true, sends ``interrupt`` to the driver. + interrupt_when: Callable[["Run"], bool] | None = None + tool_tokens: bool = False # parallel batch: each result must carry its own token + # "answer": the PROBE gets the scripted reply; "breaker": the stale breaker refuses it + # immediately with a surfaced error and no provider call. + probe: str = "answer" + huge: bool = False + background: bool = False # the tool must really have started a tracked background job + # Background scenarios: the provider holds its post-tool reply until this many tagged + # processes whose cmdline contains ``settle_on`` exist, so the kill at close() always + # meets the job in the intended state no matter how slowly its login shell starts. + settle_on: str = "" + settle_count: int = 0 + + +def _hung(timeout: float) -> Callable[[Ctx], Response]: + return _after_tools(lambda c: ToolCall("terminal", {"command": "sleep 3600", "timeout": timeout})) + + +SCENARIOS: list[Scenario] = [ + # provider faults that last forever: the turn must give up on its own + Scenario("provider_hang", lambda c: Hang()), + # request_timeout LONG: only the explicit stale timeout can end it, and it must beat the + # reasoning-model floor (#115024). The stale streak it leaves behind trips the cross-turn + # stale breaker (#58962), so the PROBE must be refused at once, surfaced, and unbilled. + Scenario("provider_stall_reasoning_model", lambda c: StallMidStream(), + cfg={"model": REASONING_MODEL, "request_timeout": LONG_TIMEOUT_S}, probe="breaker"), + Scenario("provider_500_forever", lambda c: Error(500)), + Scenario("provider_429_forever", lambda c: Error(429, retry_after=1)), + Scenario("provider_400_forever", lambda c: Error(400, "invalid request"), call_bound=2), + Scenario("provider_garbage_body_forever", lambda c: Raw("{not json at all")), + Scenario("provider_drop_mid_stream_forever", lambda c: DropMidStream(), expect="ended"), + # transient faults: a retry must get through + Scenario("provider_hang_then_recover", + lambda c: Hang() if c.n == 0 else Text(f"recovered {c.nonce}"), + expect="answer", answer="recovered {nonce}", call_bound=3), + Scenario("provider_slow_trickle_is_not_stale", + lambda c: Text(_trickle_text(c.nonce), chunk_chars=2, delay_per_chunk=TRICKLE_DELAY_S), + expect="answer", answer="{trickle}", exact_calls=1), + # malformed model output + # (undecodable arguments are treated as a cut-off reply: surfaced, never executed) + Scenario("tool_args_truncated_json", + _after_tools(lambda c: ToolCall("terminal", '{"command": "echo trunc' + c.nonce)), + expect="ended", call_bound=6), + Scenario("tool_args_garbage", + _after_tools(lambda c: ToolCall("terminal", "{{{ not json")), + expect="ended", call_bound=6), + Scenario("tool_name_unknown", + _after_tools(lambda c: ToolCall("no_such_tool_chaos", {"x": 1})), + expect="answer", answer="tools settled {nonce}", call_bound=6), + Scenario("parallel_8_terminal_calls", _after_tools(_parallel), + expect="answer", answer="tools settled {nonce}", call_bound=4, tool_tokens=True), + # tool faults + Scenario("tool_hang_past_timeout", _hung(TOOL_TIMEOUT_S), + expect="answer", answer="tools settled {nonce}", call_bound=4), + # stdin must be closed for tools: `cat` returns at EOF; a pipe held open would wedge the + # turn past its deadline (the tool timeout is LONG). + Scenario("tool_reads_stdin", + _after_tools(lambda c: ToolCall("terminal", { + "command": f"cat; echo stdin-closed-{c.nonce}", "timeout": LONG_TIMEOUT_S})), + expect="answer", answer="tools settled {nonce}", call_bound=4), + Scenario("tool_background_swarm", + _after_tools(lambda c: ToolCall("terminal", { + "command": "for i in 1 2 3 4 5 6 7 8; do sleep 3600 & done; setsid sleep 3600 & wait", + "background": True})), + expect="answer", answer="tools settled {nonce}", call_bound=4, background=True, + settle_on="sleep 3600", settle_count=9), + # a job still spawning when it is killed must not leave its late children behind; the + # TERM trap makes "spawns after the kill snapshot" deterministic (interactive bash keeps + # running through the grace window just like this); the loop's `sleep 0.2` proves the + # trap is installed before close() arrives + Scenario("tool_background_spawns_while_killed", + _after_tools(lambda c: ToolCall("terminal", { + "command": "trap 'for i in 1 2 3 4; do sleep 3600 & done' TERM; while :; do sleep 0.2; done", + "background": True})), + expect="answer", answer="tools settled {nonce}", call_bound=4, background=True, + settle_on="sleep 0.2", settle_count=1), + Scenario("tool_huge_output", + _after_tools(lambda c: ToolCall("terminal", { + "command": f"head -c {HUGE_OUTPUT_BYTES} /dev/zero | tr '\\0' A", "timeout": 120})), + expect="answer", answer="tools settled {nonce}", call_bound=4, huge=True), + # a model that never stops calling tools is bounded by agent.max_turns + Scenario("tool_loop_hits_turn_budget", + lambda c: ToolCall("terminal", {"command": f"echo loop-{c.n}"}), + cfg={"max_turns": LOOP_BUDGET}, expect="ended", call_bound=LOOP_BUDGET + 2), + # interrupts: every timeout LONG, so only agent.interrupt() can end the turn in time + Scenario("interrupt_provider_hang", lambda c: Hang(), cfg=LONG, expect="interrupted", + interrupt_when=lambda run: run.fault_calls() >= 1), + Scenario("interrupt_hung_tool", _hung(LONG_TIMEOUT_S), cfg=LONG, expect="interrupted", + interrupt_when=lambda run: run.hung_tool_running()), + Scenario("interrupt_endless_tool_loop", + lambda c: ToolCall("terminal", {"command": f"echo loop-{c.n}"}), + cfg={**LONG, "max_turns": 1000}, expect="interrupted", + interrupt_when=lambda run: run.fault_calls() >= 3), +] +BY_ID = {s.id: s for s in SCENARIOS} + + +# ── one scenario run ───────────────────────────────────────────────────────── + + +class Run: + def __init__(self, sc: Scenario, root: Path) -> None: + self.sc = sc + self.root = root + self.tag = new_tag() + self.nonce = uuid.uuid4().hex[:10] + self.session_id = f"chaos-{self.nonce}" + self.events: "queue.Queue[tuple[float, dict[str, Any]]]" = queue.Queue() + self.seen: list[tuple[float, dict[str, Any]]] = [] + self.srv = FakeLLMServer(self._respond) + self.fault_msg = f"[[chaos:{sc.id}]] please help ({self.nonce})" + self.probe_msg = f"[[probe:{self.nonce}]] are you still there?" + self.report: dict[str, Any] = {"id": sc.id, "nonce": self.nonce} + + # provider side + def _is_probe(self, messages: list[dict[str, Any]]) -> bool: + for msg in reversed(messages): + if msg.get("role") != "user": + continue + text = msg.get("content") if isinstance(msg.get("content"), str) else json.dumps(msg.get("content")) + if _PROBE_RE.search(text or ""): + return True + if _FAULT_RE.search(text or ""): + return False + return False + + def _respond(self, record: dict[str, Any]) -> Response: + messages = record["body"].get("messages") or [] + record["probe"] = self._is_probe(messages) + if record["probe"]: + return Text(f"alive {self.nonce}") + n = sum(1 for r in self.srv.requests if r["kind"] == "main" and not r.get("probe")) - 1 + if self.sc.settle_on and messages and messages[-1].get("role") == "tool": + deadline = time.monotonic() + SETTLE_DEADLINE_S + while time.monotonic() < deadline and sum( + self.sc.settle_on in d for d in describe_pids(tagged_pids(self.tag))) < self.sc.settle_count: + time.sleep(0.05) + return self.sc.fault(Ctx(self.nonce, n, messages)) + + def fault_calls(self) -> int: + return sum(1 for r in list(self.srv.requests) if r["kind"] == "main" and not r.get("probe")) + + def hung_tool_running(self) -> bool: + """The faulted turn's ``sleep 3600`` is up (other tagged helpers may start earlier).""" + return self.fault_calls() >= 1 and any("sleep 3600" in d for d in describe_pids(tagged_pids(self.tag))) + + def probe_requests(self) -> list[dict[str, Any]]: + return [r["body"] for r in list(self.srv.requests) if r["kind"] == "main" and r.get("probe")] + + # driver side + def _reader(self, stream) -> None: + for raw in stream: + line = raw.decode("utf-8", "replace").strip() + if line.startswith("CHAOS "): + try: + self.events.put((time.monotonic(), json.loads(line[6:]))) + except json.JSONDecodeError: + pass + + def _wait(self, pred: Callable[[dict[str, Any]], bool], deadline: float) -> tuple[float, dict[str, Any]] | None: + for t, ev in self.seen: + if pred(ev): + return t, ev + while True: + left = deadline - time.monotonic() + if left <= 0: + return None + try: + t, ev = self.events.get(timeout=min(left, 0.2)) + except queue.Empty: + if self.proc.poll() is not None and self.events.empty(): + return None + continue + self.seen.append((t, ev)) + if pred(ev): + return t, ev + + def execute(self) -> dict[str, Any]: + rep = self.report + self.srv.start() + try: + self._execute(rep) + finally: + if getattr(self, "proc", None) is not None and self.proc.poll() is None: + self.proc.kill() + self.proc.wait(timeout=30) + kill_tagged(self.tag) + self.srv.stop() + return rep + + def _execute(self, rep: dict[str, Any]) -> None: + sc = self.sc + home, hermes_home = write_chaos_home(self.root, self.srv.base_url, **sc.cfg) + work = self.root / "work" + work.mkdir() + spec = self.root / "spec.json" + spec.write_text(json.dumps({ + "session_id": self.session_id, "turns": [self.fault_msg, self.probe_msg]}), encoding="utf-8") + env = hermetic_env(home, hermes_home, self.tag) + env["TERMINAL_CWD"] = str(work) + stderr = open(self.root / "driver.stderr", "wb") + self.proc = subprocess.Popen( + [python_exe(), "-m", "tests.e2e.core.chaos._agent_driver", str(spec)], + cwd=str(work), env=env, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=stderr, + ) + threading.Thread(target=self._reader, args=(self.proc.stdout,), daemon=True).start() + rep["stderr_path"] = str(self.root / "driver.stderr") + + if self._wait(lambda e: e["ev"] == "turn_start" and e["i"] == 0, + time.monotonic() + BOOT_DEADLINE_S) is None: + rep["boot_failed"] = True + return + t0 = time.monotonic() + deadline = t0 + TURN_DEADLINE_S + if sc.interrupt_when is not None: + trigger_deadline = t0 + TURN_DEADLINE_S + while time.monotonic() < trigger_deadline and not sc.interrupt_when(self): + if self._wait(lambda e: e["ev"] == "turn_end", time.monotonic() + 0.05): + break + if sc.interrupt_when(self): + rep["interrupt_at"] = time.monotonic() - t0 + rep["calls_at_interrupt"] = self.fault_calls() + self.proc.stdin.write(b"interrupt\n") + self.proc.stdin.flush() + deadline = time.monotonic() + INTERRUPT_DEADLINE_S + end = self._wait(lambda e: e["ev"] == "turn_end" and e["i"] == 0, deadline) + rep["fault_calls"] = self.fault_calls() + if end is None: + rep["turn0"] = None + rep["turn0_open_after"] = time.monotonic() - t0 + return + rep["turn0"] = end[1] + rep["turn0_elapsed"] = end[0] - t0 + if "interrupt_at" in rep: + rep["after_interrupt"] = end[0] - t0 - rep["interrupt_at"] + # the hung tool of the faulted turn must be gone before the next turn starts + rep["tool_survivors_after_turn"] = describe_pids(wait_no_tagged(self.tag, PROCESS_REAP_S)) \ + if sc.id in {"tool_hang_past_timeout", "interrupt_hung_tool"} else [] + probe = self._wait(lambda e: e["ev"] == "turn_end" and e["i"] == 1, + time.monotonic() + PROBE_DEADLINE_S) + rep["fault_calls_final"] = self.fault_calls() + rep["turn1"] = probe[1] if probe else None + closed = self._wait(lambda e: e["ev"] == "closed", time.monotonic() + EXIT_DEADLINE_S) + rep["closed"] = closed is not None + try: + rep["exit_code"] = self.proc.wait(timeout=EXIT_DEADLINE_S) + except subprocess.TimeoutExpired: + rep["exit_code"] = None + rep["orphans"] = describe_pids(wait_no_tagged(self.tag, PROCESS_REAP_S)) + log = hermes_home / "logs" / "agent.log" + if rep["orphans"] and log.exists(): + rep["agent_log_tail"] = log.read_text(errors="replace")[-6000:] + db = hermes_home / "state.db" + rep["persisted"] = persisted_messages(db, self.session_id) if db.exists() else None + rep["integrity"] = integrity_ok(db) if db.exists() else "missing" + probes = self.probe_requests() + rep["probe_request"] = probes[-1] if probes else None + mains = [r["body"] for r in list(self.srv.requests) if r["kind"] == "main"] + rep["last_request"] = mains[-1] if mains else None + rep["max_request_bytes"] = max((len(json.dumps(b)) for b in mains), default=0) + + +# ── module fixture: run the selected scenarios concurrently ────────────────── + + +@pytest.fixture(scope="module") +def runs(request, tmp_path_factory) -> dict[str, Future]: + selected = [ + item.callspec.params["scenario_id"] for item in request.session.items + if item.module is request.module and hasattr(item, "callspec") + and "scenario_id" in item.callspec.params + ] + pool = ThreadPoolExecutor(max_workers=SCENARIO_WORKERS, thread_name_prefix="chaos") + futures = { + sid: pool.submit(Run(BY_ID[sid], tmp_path_factory.mktemp(sid)).execute) + for sid in dict.fromkeys(selected) + } + yield futures + pool.shutdown(wait=True, cancel_futures=True) + + +def _text(content: Any) -> str: + return content if isinstance(content, str) else json.dumps(content) + + +@pytest.mark.parametrize("scenario_id", list(BY_ID)) +def test_agent_turn_liveness(scenario_id: str, runs: dict[str, Future]) -> None: + sc = BY_ID[scenario_id] + rep = runs[scenario_id].result(timeout=BOOT_DEADLINE_S + 3 * TURN_DEADLINE_S + 120) + tail = Path(rep["stderr_path"]).read_text(errors="replace")[-3000:] if rep.get("stderr_path") else "" + where = f"[{scenario_id}] stderr tail:\n{tail}" + + assert not rep.get("boot_failed"), f"driver never started the turn. {where}" + + # 1. bounded termination + turn0 = rep["turn0"] + if "interrupt_at" in rep: + assert turn0 is not None, ( + f"turn still running {INTERRUPT_DEADLINE_S}s after agent.interrupt() " + f"(calls={rep['fault_calls']}). {where}") + else: + assert sc.interrupt_when is None, f"interrupt trigger never became true. {where}" + assert turn0 is not None, ( + f"faulted turn still running after {TURN_DEADLINE_S}s (calls={rep['fault_calls']}). {where}") + final = turn0["final"] + if sc.expect == "interrupted": + # the CLI renders its own interruption notice; the turn must say it was interrupted + assert turn0["interrupted"], f"interrupt not reflected in the result: {turn0}" + else: + assert final.strip(), f"turn ended with nothing surfaced to the user: {turn0}" + if sc.expect == "answer": + want = sc.answer.format(nonce=rep_nonce(rep), trickle=_trickle_text(rep_nonce(rep))) + assert want in final, f"expected the answer {want!r}, got {final[:300]!r}" + + # 2. bounded provider calls + calls = rep["fault_calls"] + if sc.exact_calls is not None: + assert calls == sc.exact_calls, f"healthy stream was retried: {calls} calls" + assert 1 <= calls <= sc.call_bound, f"{calls} provider calls for one turn (bound {sc.call_bound})" + if "interrupt_at" in rep: + extra = rep["fault_calls_final"] - rep["calls_at_interrupt"] + assert extra <= 2, f"the loop kept calling the provider after interrupt: +{extra}" + + # 3. reusability: the same agent answers the next message, with the fault in its history + assert rep["turn1"] is not None, f"PROBE turn did not finish within {PROBE_DEADLINE_S}s. {where}" + if sc.probe == "breaker": + assert rep["turn1"]["failed"] and rep["turn1"]["final"].strip(), f"breaker refusal not surfaced: {rep['turn1']}" + assert rep["probe_request"] is None, "a tripped stale breaker still billed the provider" + sent = rep["last_request"]["messages"] + else: + assert f"alive {rep_nonce(rep)}" in rep["turn1"]["final"], f"PROBE not answered: {rep['turn1']}" + assert rep["probe_request"] is not None, "PROBE never reached the provider" + sent = rep["probe_request"]["messages"] + assert any(f"[[chaos:{scenario_id}]]" in _text(m.get("content")) for m in sent if m.get("role") == "user"), \ + "the faulted user turn vanished from the history sent with the next message" + + # 4. history integrity, persisted and sent + persisted = rep["persisted"] + assert persisted, "nothing persisted to state.db" + assert unanswered_tool_calls(persisted) == [], "state.db has tool_calls without results" + assert unanswered_tool_calls(sent) == [], "request sent to the model has tool_calls without results" + assert rep["integrity"] == "ok" + if sc.tool_tokens: + for label, msgs in (("state.db", persisted), ("request", sent)): + results = tool_results_by_id(msgs) + calls_by_id = { + c["id"]: c["function"]["arguments"] + for m in msgs if m.get("role") == "assistant" + for c in (json.loads(m["tool_calls"]) if isinstance(m.get("tool_calls"), str) + else m.get("tool_calls") or []) + } + assert len(calls_by_id) == PARALLEL_CALLS, f"{label}: {len(calls_by_id)} calls" + for cid, args in calls_by_id.items(): + token = re.search(r"tok[0-9a-f]+n\d", args) + token = token.group(0) if token else args + assert token in results.get(cid, ""), f"{label}: {cid} ({token}) got {results.get(cid, '')[:160]!r}" + if sc.background: + started = [m.get("content") or "" for m in persisted if m.get("role") == "tool"] + assert any('"pid"' in c and "proc_" in c for c in started), f"background job never started: {started}" + if sc.huge: + biggest = max((len(m.get("content") or "") for m in persisted if m.get("role") == "tool"), default=0) + assert biggest <= MAX_PERSISTED_TOOL_CHARS, f"{biggest} chars of tool output persisted" + assert rep["max_request_bytes"] <= MAX_REQUEST_BYTES, f"{rep['max_request_bytes']} byte request" + + # 5. process hygiene + assert rep["tool_survivors_after_turn"] == [], f"hung tool outlived its turn: {rep['tool_survivors_after_turn']}" + assert rep["closed"], f"agent.close() did not return. {where}" + assert rep["exit_code"] is not None, f"driver did not exit {EXIT_DEADLINE_S}s after close(). {where}" + assert rep["orphans"] == [], ( + f"processes outlived the session: {rep['orphans']}\nagent.log tail:\n{rep.get('agent_log_tail', '')}") + + +def rep_nonce(rep: dict[str, Any]) -> str: + return rep["nonce"] + + +def test_retry_bound_matches_config() -> None: + """Guard the oracle itself: the per-turn call bound must stay a small multiple of the + configured retries, or a storm could hide under it.""" + assert MAX_PROVIDER_CALLS_PER_FAILED_TURN <= 8 * API_MAX_RETRIES From adc4cff5bd382dccc6226fcfd7474af2472d2ebb Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:12:23 -0700 Subject: [PATCH 017/104] test: C8 liveness chaos matrix through the real messaging gateway python -m gateway.run with a fake platform adapter: each fault mode must end the turn within its deadline (or promptly on /stop / SIGTERM), keep provider calls bounded, release the turn lease so the session answers the next message (#104303 class), keep the event loop responsive (heartbeat), answer every tool_call (#93251 class) and leave no orphan process. --- .../e2e/core/chaos/_gateway_fake_platform.py | 163 ++++++ tests/e2e/core/chaos/_gateway_harness.py | 240 +++++++++ .../core/chaos/test_gateway_turn_liveness.py | 500 ++++++++++++++++++ 3 files changed, 903 insertions(+) create mode 100644 tests/e2e/core/chaos/_gateway_fake_platform.py create mode 100644 tests/e2e/core/chaos/_gateway_harness.py create mode 100644 tests/e2e/core/chaos/test_gateway_turn_liveness.py diff --git a/tests/e2e/core/chaos/_gateway_fake_platform.py b/tests/e2e/core/chaos/_gateway_fake_platform.py new file mode 100644 index 0000000000..0c59bb2896 --- /dev/null +++ b/tests/e2e/core/chaos/_gateway_fake_platform.py @@ -0,0 +1,163 @@ +"""In-memory messaging platform for the gateway chaos suite, loaded as a REAL user plugin. + +The chaos test writes ``$HERMES_HOME/plugins/chaos-fake/`` (``kind: platform``) whose +``register`` is this module's, enables it in config.yaml, and runs the real +``python -m gateway.run`` entry point. The gateway then discovers the plugin, builds this +adapter through the platform registry, connects it and installs its own message handler +— the same path every third-party adapter takes. Nothing in the runner is patched. + +Wire to the test process: one loopback TCP connection (port in ``CHAOS_GW_PORT``) +carrying JSON lines. + + test -> adapter {"op": "msg", "text": ..., "chat": ..., "id": ...} + adapter -> test {"ev": "ready"} once connected + {"ev": "send", "chat": ..., "text": ...} for every outbound send/edit + {"ev": "hb", "max_gap": s, "ticks": n} every ~0.5 s: the largest gap + between 100 ms event-loop ticks since the previous report (a sync call + blocking the gateway loop shows up here, not in the test's own timing) + {"ev": "disconnect"} from the real shutdown path +""" + +from __future__ import annotations + +import asyncio +import json +import os +import time +import uuid +from typing import Any, Dict, Optional + +from gateway.config import Platform, PlatformConfig +from gateway.platforms.base import BasePlatformAdapter, SendResult +from gateway.platforms.event import MessageEvent, MessageType + +PLATFORM_NAME = "chaosfake" +PLUGIN_DIR_NAME = "chaos-fake" +ALLOW_ALL_ENV = "CHAOS_FAKE_ALLOW_ALL_USERS" +PORT_ENV = "CHAOS_GW_PORT" +HEARTBEAT_TICK_S = 0.1 +HEARTBEAT_REPORT_S = 0.5 + + +class ChaosFakeAdapter(BasePlatformAdapter): + def __init__(self, config: PlatformConfig): + super().__init__(config=config, platform=Platform(PLATFORM_NAME)) + self._reader: Optional[asyncio.StreamReader] = None + self._writer: Optional[asyncio.StreamWriter] = None + self._tasks: list[asyncio.Task] = [] + self._seq = 0 + + # -- wire ---------------------------------------------------------------- + + def _emit(self, payload: Dict[str, Any]) -> None: + writer = self._writer + if writer is None or writer.is_closing(): + return + payload["t"] = time.time() + writer.write((json.dumps(payload) + "\n").encode()) + + async def _heartbeat(self) -> None: + last = time.monotonic() + window_max = 0.0 + last_report = last + ticks = 0 + while True: + await asyncio.sleep(HEARTBEAT_TICK_S) + now = time.monotonic() + window_max = max(window_max, now - last) + last = now + ticks += 1 + if now - last_report >= HEARTBEAT_REPORT_S: + self._emit({"ev": "hb", "max_gap": round(window_max, 3), "ticks": ticks, + "busy": sorted(self._active_sessions)}) + window_max = 0.0 + last_report = now + + async def _read_commands(self) -> None: + assert self._reader is not None + while True: + line = await self._reader.readline() + if not line: + return + try: + cmd = json.loads(line) + except json.JSONDecodeError: + continue + if cmd.get("op") == "msg": + await self._inbound(cmd) + + async def _inbound(self, cmd: Dict[str, Any]) -> None: + chat = str(cmd.get("chat") or "chaos-chat") + msg_id = str(cmd.get("id") or uuid.uuid4().hex) + source = self.build_source( + chat_id=chat, chat_name=chat, chat_type="dm", + user_id="chaos-user", user_name="chaos-user", message_id=msg_id) + event = MessageEvent( + text=str(cmd.get("text") or ""), message_type=MessageType.TEXT, + source=source, message_id=msg_id) + await self.handle_message(event) + + # -- BasePlatformAdapter ------------------------------------------------- + + async def connect(self, *, is_reconnect: bool = False) -> bool: + port = int(os.environ[PORT_ENV]) + self._reader, self._writer = await asyncio.open_connection("127.0.0.1", port) + self._tasks = [ + asyncio.create_task(self._heartbeat(), name="chaos-heartbeat"), + asyncio.create_task(self._read_commands(), name="chaos-inbound"), + ] + self._mark_connected() + self._emit({"ev": "ready", "pid": os.getpid()}) + return True + + async def disconnect(self) -> None: + self._emit({"ev": "disconnect"}) + self._mark_disconnected() + for task in self._tasks: + task.cancel() + for task in self._tasks: + try: + await task + except (asyncio.CancelledError, Exception): + pass + self._tasks = [] + if self._writer is not None: + try: + await self._writer.drain() + self._writer.close() + except Exception: + pass + self._writer = None + + async def send(self, chat_id: str, content: str, reply_to: Optional[str] = None, + metadata: Optional[Dict[str, Any]] = None) -> SendResult: + self._seq += 1 + message_id = f"chaos-out-{self._seq}" + self._emit({"ev": "send", "chat": str(chat_id), "text": content, "id": message_id}) + return SendResult(success=True, message_id=message_id) + + async def edit_message(self, chat_id: str, message_id: str, content: str, **_kw: Any) -> SendResult: + self._emit({"ev": "send", "chat": str(chat_id), "text": content, "id": message_id, "edit": True}) + return SendResult(success=True, message_id=message_id) + + async def on_processing_start(self, event: MessageEvent) -> None: + self._emit({"ev": "start", "id": event.message_id}) + + async def on_processing_complete(self, event: MessageEvent, outcome: Any) -> None: + self._emit({"ev": "complete", "id": event.message_id, "outcome": str(outcome)}) + + async def get_chat_info(self, chat_id: str) -> Dict[str, Any]: + return {"name": chat_id, "type": "dm"} + + +def register(ctx) -> None: + """Plugin entry point (the chaos test writes a plugin dir importing this).""" + ctx.register_platform( + name=PLATFORM_NAME, + label="Chaos Fake", + adapter_factory=ChaosFakeAdapter, + check_fn=lambda: True, + validate_config=lambda _cfg: True, + is_connected=lambda _cfg: True, + allow_all_env=ALLOW_ALL_ENV, + ) diff --git a/tests/e2e/core/chaos/_gateway_harness.py b/tests/e2e/core/chaos/_gateway_harness.py new file mode 100644 index 0000000000..a67197b8be --- /dev/null +++ b/tests/e2e/core/chaos/_gateway_harness.py @@ -0,0 +1,240 @@ +"""Test-side driver for a real ``python -m gateway.run`` process with the chaos fake platform. + +The gateway child runs the production entry point (plugin discovery, platform registry, +adapter connect, message handler install, SIGTERM drain) in a hermetic home; this module +owns the loopback socket the fake adapter dials, records every adapter event, and gives +the scenarios deadline-bounded waits. It never sleeps as synchronization: every wait is a +condition-variable poll against a deadline. +""" + +from __future__ import annotations + +import json +import os +import signal +import socket +import subprocess +import threading +import time +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Callable, Optional + +from tests.e2e.core.chaos import _gateway_fake_platform as fake_platform +from tests.e2e.core.chaos._helpers import hermetic_env, python_exe, write_chaos_home + +BOOT_DEADLINE_S = 180.0 +SHUTDOWN_DEADLINE_S = 60.0 + +PLUGIN_MANIFEST = ( + f"name: {fake_platform.PLUGIN_DIR_NAME}\n" + "label: Chaos Fake\n" + "kind: platform\n" + "version: 0.0.1\n" + "description: in-memory platform for the gateway chaos suite\n" +) +PLUGIN_INIT = "from tests.e2e.core.chaos._gateway_fake_platform import register # noqa: F401\n" + + +def gateway_extra_config() -> str: + """config.yaml additions: enable the user plugin + its platform, terminal-only toolset.""" + name = fake_platform.PLATFORM_NAME + return ( + "plugins:\n" + f" enabled: [{fake_platform.PLUGIN_DIR_NAME}]\n" + "platforms:\n" + f" {name}:\n" + " enabled: true\n" + "platform_toolsets:\n" + f" {name}: [terminal]\n" + "terminal:\n" + " backend: local\n" + # /new is a scenario step (stale-breaker recovery), not an interactive confirmation. + "approvals:\n" + " destructive_slash_confirm: false\n" + ) + + +@dataclass +class Event: + ev: str + data: dict[str, Any] + mono: float + idx: int + + +@dataclass +class GatewayProc: + root: Path + base_url: str + tag: str + cfg: dict[str, Any] = field(default_factory=dict) + extra_config: str = "" + + def __post_init__(self) -> None: + self.home, self.hermes_home = write_chaos_home( + self.root, self.base_url, extra=gateway_extra_config() + self.extra_config, **self.cfg) + plugin_dir = self.hermes_home / "plugins" / fake_platform.PLUGIN_DIR_NAME + plugin_dir.mkdir(parents=True, exist_ok=True) + (plugin_dir / "plugin.yaml").write_text(PLUGIN_MANIFEST, encoding="utf-8") + (plugin_dir / "__init__.py").write_text(PLUGIN_INIT, encoding="utf-8") + self.state_db = self.hermes_home / "state.db" + self.log_path = self.root / "gateway.log" + self.events: list[Event] = [] + self._cond = threading.Condition() + self._conn: Optional[socket.socket] = None + self._listener: Optional[socket.socket] = None + self.proc: Optional[subprocess.Popen] = None + self.exit_code: Optional[int] = None + self.shutdown_s: Optional[float] = None + self._msg_seq = 0 + + # -- lifecycle ------------------------------------------------------------- + + def start(self) -> None: + listener = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + listener.bind(("127.0.0.1", 0)) + listener.listen(1) + listener.settimeout(1.0) + self._listener = listener + env = hermetic_env(self.home, self.hermes_home, self.tag) + for key in ("NOTIFY_SOCKET", "INVOCATION_ID", "WATCHDOG_USEC", "WATCHDOG_PID", "XDG_STATE_HOME", + "XDG_CONFIG_HOME", "XDG_DATA_HOME", "XDG_CACHE_HOME"): + env.pop(key, None) + env.update({ + fake_platform.PORT_ENV: str(listener.getsockname()[1]), + fake_platform.ALLOW_ALL_ENV: "true", + # Host rendezvous (one-gateway-per-user lock + record) must never see the live gateway. + "HERMES_GATEWAY_LOCK_DIR": str(self.root / "locks"), + "HERMES_GATEWAY_MAX_STARTS": "0", + # The child's HOME is the tmp root, so the live-DB guard (pytest ancestry) would take + # its tmp state.db for "production"; the documented child opt-out is safe here. + "HERMES_STATE_DB_GUARD_BYPASS": "1", + }) + log = open(self.log_path, "wb") + self.proc = subprocess.Popen( + [python_exe(), "-m", "gateway.run"], cwd=str(self.home), env=env, + stdin=subprocess.DEVNULL, stdout=log, stderr=subprocess.STDOUT, start_new_session=True) + log.close() + deadline = time.monotonic() + BOOT_DEADLINE_S + while True: + try: + conn, _ = listener.accept() + break + except socket.timeout: + if self.proc.poll() is not None: + raise AssertionError(f"gateway exited during boot rc={self.proc.returncode}\n{self.log_tail()}") + if time.monotonic() >= deadline: + raise AssertionError(f"gateway did not connect the fake platform in {BOOT_DEADLINE_S}s\n" + f"{self.log_tail()}") + conn.settimeout(None) + self._conn = conn + threading.Thread(target=self._pump, name=f"chaos-gw-{self.tag}", daemon=True).start() + self.wait_for(lambda e: e.ev == "ready", BOOT_DEADLINE_S, "adapter ready") + + def _pump(self) -> None: + assert self._conn is not None + buf = b"" + while True: + try: + chunk = self._conn.recv(65536) + except OSError: + chunk = b"" + if not chunk: + with self._cond: + self.events.append(Event("eof", {}, time.monotonic(), len(self.events))) + self._cond.notify_all() + return + buf += chunk + while b"\n" in buf: + line, buf = buf.split(b"\n", 1) + try: + data = json.loads(line) + except json.JSONDecodeError: + continue + with self._cond: + self.events.append(Event(str(data.get("ev")), data, time.monotonic(), len(self.events))) + self._cond.notify_all() + + def stop(self) -> None: + """Real SIGTERM shutdown; records wall time and exit code. Escalates to SIGKILL only + after SHUTDOWN_DEADLINE_S (the test then fails on ``shutdown_s``/``exit_code``).""" + if self.proc is None or self.exit_code is not None: + return + t0 = time.monotonic() + if self.proc.poll() is None: + self.proc.send_signal(signal.SIGTERM) + try: + self.exit_code = self.proc.wait(timeout=SHUTDOWN_DEADLINE_S) + self.shutdown_s = time.monotonic() - t0 + except subprocess.TimeoutExpired: + self.shutdown_s = None + with _suppress_oserror(): + os.killpg(self.proc.pid, signal.SIGKILL) + self.exit_code = self.proc.wait(timeout=30) + finally: + for sock in (self._conn, self._listener): + if sock is not None: + with _suppress_oserror(): + sock.close() + + # -- driving ----------------------------------------------------------------- + + def send_user(self, text: str, chat: str = "chaos-chat") -> str: + assert self._conn is not None + self._msg_seq += 1 + msg_id = f"in-{self._msg_seq}" + self._conn.sendall((json.dumps({"op": "msg", "text": text, "chat": chat, "id": msg_id}) + "\n").encode()) + return msg_id + + def mark(self) -> int: + with self._cond: + return len(self.events) + + def wait_for(self, pred: Callable[[Event], bool], timeout: float, what: str, + since: int = 0) -> Optional[Event]: + """First event at index >= ``since`` matching ``pred``; None at the deadline.""" + deadline = time.monotonic() + timeout + with self._cond: + idx = since + while True: + while idx < len(self.events): + ev = self.events[idx] + idx += 1 + if pred(ev): + return ev + left = deadline - time.monotonic() + if left <= 0: + return None + self._cond.wait(min(left, 0.5)) + + def events_between(self, start: int, end: int | None = None) -> list[Event]: + with self._cond: + return list(self.events[start:end]) + + def sends_since(self, since: int) -> list[str]: + with self._cond: + return [e.data.get("text") or "" for e in self.events[since:] if e.ev == "send"] + + def max_heartbeat_gap(self) -> float: + with self._cond: + return max((float(e.data.get("max_gap") or 0) for e in self.events if e.ev == "hb"), default=0.0) + + def heartbeat_count(self) -> int: + with self._cond: + return sum(1 for e in self.events if e.ev == "hb") + + def log_tail(self, n: int = 80) -> str: + try: + lines = self.log_path.read_text(encoding="utf-8", errors="replace").splitlines() + except OSError: + return "" + return "\n".join(lines[-n:]) + + +class _suppress_oserror: + def __enter__(self) -> None: + return None + + def __exit__(self, exc_type, *_: object) -> bool: + return exc_type is not None and issubclass(exc_type, (OSError, ProcessLookupError)) diff --git a/tests/e2e/core/chaos/test_gateway_turn_liveness.py b/tests/e2e/core/chaos/test_gateway_turn_liveness.py new file mode 100644 index 0000000000..80d06a5521 --- /dev/null +++ b/tests/e2e/core/chaos/test_gateway_turn_liveness.py @@ -0,0 +1,500 @@ +"""C8 chaos: agent-turn liveness through the REAL messaging gateway (``python -m gateway.run``). + +The gateway process is the shipped entry point: it discovers a user platform plugin +(``_gateway_fake_platform``, an in-memory adapter dialled over loopback), connects it, +installs the runner's own message handler, and runs every inbound message through the real +``GatewayRunner._handle_message`` -> ``_handle_message_with_agent`` -> ``AIAgent`` turn with +the real turn lease, session store, busy-input policy, /stop path, state.db and SIGTERM +drain. Only the LLM provider (``tests/fakes/fake_llm_provider``) and the chat platform are +fake. Each scenario gets its own gateway so process hygiene is judged per fault; all +scenarios run concurrently against one scripted provider and are asserted per test. + +After every fault the same invariants hold: + +* the faulted turn ends (the adapter's processing-complete hook fires for that message) + within TURN_DEADLINE_S — or within INTERRUPT_DEADLINE_S of /stop, a busy follow-up or + SIGTERM when every configured timeout is LONG_TIMEOUT_S — and the user was sent + something (answer or surfaced error) before it ended; +* provider main calls for the faulted turn stay <= MAX_PROVIDER_CALLS_PER_FAILED_TURN; +* the gateway event loop keeps ticking (100 ms heartbeat inside the gateway process); +* the SAME chat then accepts a new message and gets exactly one scripted answer, with the + faulted turn in its history (turn-lease leak oracle, #104303); +* every tool_call has its own result in state.db and in what is sent back to the model + (8-way parallel batch: each id carries ITS token, #93251); +* hung tool processes are reaped; SIGTERM makes the gateway exit by itself; no process + carrying the scenario tag survives; state.db passes integrity_check. +""" + +from __future__ import annotations + +import json +import re +import sqlite3 +import sys +import time +import uuid +from concurrent.futures import Future, ThreadPoolExecutor +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Callable + +import pytest + +from tests.e2e.core.chaos._gateway_harness import SHUTDOWN_DEADLINE_S, Event, GatewayProc +from tests.e2e.core.chaos._helpers import ( + INTERRUPT_DEADLINE_S, + LONG_TIMEOUT_S, + MAX_PROVIDER_CALLS_PER_FAILED_TURN, + PLAIN_MODEL, + PROCESS_REAP_S, + REASONING_MODEL, + TOOL_TIMEOUT_S, + TURN_DEADLINE_S, + describe_pids, + integrity_ok, + kill_tagged, + new_tag, + tagged_pids, + tool_results_by_id, + unanswered_tool_calls, + wait_no_tagged, +) +from tests.fakes.fake_llm_provider import ( + DropMidStream, + Error, + FakeLLMServer, + Hang, + StallMidStream, + Text, + ToolCall, +) + +pytestmark = [ + pytest.mark.skipif(sys.platform != "linux", reason="orphan scan reads /proc; POSIX gateway signals"), +] + +# The adapter ticks every 100 ms on the gateway loop. Healthy gaps are ~0.11 s, worst seen +# 0.6 s at load average ~100 (10x margin -> 6 s). A turn run synchronously on the loop +# freezes it for the whole faulted turn (>= 22 s for a provider hang with the chaos timeouts, +# LONG_TIMEOUT_S in the interrupt scenarios), so the bound still catches it. +HEARTBEAT_MAX_GAP_S = 6.0 +PARALLEL_CALLS = 8 +HUNG_COMMAND = "sleep 3600" +SCENARIO_WORKERS = 6 +CHAT = "chaos-chat" +OTHER_CHAT = "chaos-chat-b" + +_MARK = re.compile(r"\[\[chaos:(?P[a-z0-9_]+):(?P[0-9a-f]+)\]\]") +_PROBE = re.compile(r"\[\[probe:(?P[0-9a-f]+):(?P\d+)\]\]") + + +def _text_of(content: Any) -> str: + if isinstance(content, str): + return content + if isinstance(content, list): + return " ".join(str(p.get("text", "")) if isinstance(p, dict) else str(p) for p in content) + return str(content or "") + + +def _last_user_text(messages: list[dict[str, Any]]) -> str: + for msg in reversed(messages): + if msg.get("role") == "user": + return _text_of(msg.get("content")) + return "" + + +def _token(nonce: str, i: int) -> str: + return f"tok{nonce}n{i}" + + +def _alive(nonce: str, n: int) -> str: + return f"alive {nonce} {n}" + + +# ── scripted provider: one module-wide server, faults keyed by the prompt marker ──────── + + +def _tools_then_text(first: Callable[[str], ToolCall]) -> Callable[[list, str], Any]: + def respond(messages: list[dict[str, Any]], nonce: str): + if messages and messages[-1].get("role") == "tool": + return Text(f"tools settled {nonce}") + return first(nonce) + return respond + + +def _parallel_batch(nonce: str) -> ToolCall: + calls = [("terminal", {"command": f"echo {_token(nonce, i)}"}) for i in range(PARALLEL_CALLS)] + return ToolCall(calls[0][0], calls[0][1], parallel=calls[1:]) + + +FAULTS: dict[str, Callable[[list, str], Any]] = { + "hang": lambda _m, _n: Hang(), + "stall": lambda _m, _n: StallMidStream(), + "err500": lambda _m, _n: Error(500), + "err429": lambda _m, _n: Error(429), + "drop": lambda _m, _n: DropMidStream(), + "parallel": _tools_then_text(_parallel_batch), + "toolhang": _tools_then_text( + lambda _n: ToolCall("terminal", {"command": HUNG_COMMAND, "timeout": TOOL_TIMEOUT_S})), + "toolhang_long": _tools_then_text( + lambda _n: ToolCall("terminal", {"command": HUNG_COMMAND, "timeout": LONG_TIMEOUT_S})), +} + + +def responder(record: dict[str, Any]): + messages = record["body"].get("messages") or [] + last_user = _last_user_text(messages) + if probe := _PROBE.search(last_user): + return Text(_alive(probe["nonce"], int(probe["n"]))) + if mark := _MARK.search(last_user): + return FAULTS[mark["fault"]](messages, mark["nonce"]) + return Text("unscripted") + + +def _main_requests_for(srv: FakeLLMServer, needle: str) -> list[dict[str, Any]]: + return [ + r["body"] for r in list(srv.requests) + if r["kind"] == "main" and needle in _last_user_text(r["body"].get("messages") or []) + ] + + +# ── scenario matrix ────────────────────────────────────────────────────────────────────── + + +@dataclass(frozen=True) +class Scenario: + id: str + fault: str + model: str = PLAIN_MODEL + long_timeouts: bool = False + # How the faulted turn is ended: None = by the gateway's own timeouts/retry budget; + # "stop" = /stop in the same chat; "busy" = a second user message (busy-input path); + # "sigterm" = the gateway process is terminated mid-turn. + ender: str | None = None + # When the ender fires: "provider" = the provider holds the request; "tool" = the hung + # tool process is running. + when: str = "provider" + busy_mode: str | None = None + reaps_hung_tool: bool = False + # A stalled turn makes >= HERMES_STREAM_STALE_GIVEUP stale attempts, and the cross-turn + # breaker (#58962) then refuses the session's next turn WITHOUT calling the provider + # until /new or a provider swap. Allowed only here, and only as a bounded, surfaced + # refusal with zero provider calls; /new must then make the chat answer again. + may_trip_stale_breaker: bool = False + + +SCENARIOS = [ + Scenario("provider_hang", "hang"), + # Reasoning slug: the explicit stale_timeout_seconds must beat its reasoning floor. + Scenario("provider_stall_reasoning_model", "stall", model=REASONING_MODEL, may_trip_stale_breaker=True), + Scenario("provider_500_forever", "err500"), + Scenario("provider_429_forever", "err429"), + Scenario("provider_drop_mid_stream", "drop"), + Scenario("parallel_8_terminal_calls", "parallel"), + Scenario("tool_hang_past_timeout", "toolhang", reaps_hung_tool=True), + Scenario("stop_during_provider_hang", "hang", long_timeouts=True, ender="stop"), + Scenario("stop_during_hung_tool", "toolhang_long", long_timeouts=True, ender="stop", when="tool", + reaps_hung_tool=True), + # busy_input_mode interrupt: the follow-up must preempt a turn that would hang 600 s. + Scenario("busy_interrupt_during_provider_hang", "hang", long_timeouts=True, ender="busy", + busy_mode="interrupt"), + # busy_input_mode queue: the follow-up waits for the faulted turn to fail, then is answered. + Scenario("busy_queue_during_provider_hang", "hang", ender="busy", busy_mode="queue"), + Scenario("sigterm_during_provider_hang", "hang", long_timeouts=True, ender="sigterm"), +] + + +def _hung_tool_pids(tag: str) -> list[int]: + out = [] + for pid in tagged_pids(tag): + try: + cmd = Path(f"/proc/{pid}/cmdline").read_bytes().replace(b"\0", b" ").decode(errors="replace") + except OSError: + continue + if HUNG_COMMAND in cmd: + out.append(pid) + return out + + +def _wait_until(pred: Callable[[], object], timeout: float) -> bool: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + if pred(): + return True + time.sleep(0.05) + return bool(pred()) + + +def _session_with(state_db: Path, needle: str) -> str | None: + conn = sqlite3.connect(f"file:{state_db}?mode=ro", uri=True, timeout=10) + try: + row = conn.execute( + "SELECT session_id FROM messages WHERE role = 'user' AND content LIKE ? ORDER BY id DESC LIMIT 1", + (f"%{needle}%",)).fetchone() + finally: + conn.close() + return row[0] if row else None + + +def _persisted(state_db: Path, session_id: str) -> list[dict[str, Any]]: + conn = sqlite3.connect(f"file:{state_db}?mode=ro", uri=True, timeout=10) + try: + conn.row_factory = sqlite3.Row + rows = conn.execute( + "SELECT role, content, tool_call_id, tool_calls FROM messages " + "WHERE session_id = ? AND active = 1 AND role IN ('user', 'assistant', 'tool') ORDER BY id", + (session_id,)).fetchall() + finally: + conn.close() + out = [] + for r in rows: + m: dict[str, Any] = {"role": r["role"], "content": r["content"]} + if r["tool_call_id"]: + m["tool_call_id"] = r["tool_call_id"] + if r["tool_calls"]: + m["tool_calls"] = r["tool_calls"] + out.append(m) + return out + + +def _assert_parallel_results(messages: list[dict[str, Any]], nonce: str, where: str) -> None: + results = tool_results_by_id(messages) + seen = 0 + for msg in messages: + calls = msg.get("tool_calls") + if msg.get("role") != "assistant" or not calls: + continue + for call in json.loads(calls) if isinstance(calls, str) else calls: + args = json.loads(call["function"]["arguments"]) + token = args["command"].split()[-1] + assert token.startswith(f"tok{nonce}"), (where, args) + body = results.get(call["id"], "") + assert token in body and "Result unavailable" not in body, ( + f"{where}: tool_call {call['id']} ({token}) got {body[:200]!r}") + seen += 1 + assert seen == PARALLEL_CALLS, f"{where}: {seen} tool calls, expected {PARALLEL_CALLS}" + + +def _is_send(chat: str, needle: str | None = None) -> Callable[[Event], bool]: + return lambda e: e.ev == "send" and e.data.get("chat") == chat and ( + needle is None or needle in (e.data.get("text") or "")) + + +def _is_complete(msg_id: str) -> Callable[[Event], bool]: + return lambda e: e.ev == "complete" and e.data.get("id") == msg_id + + +def _settle(gw: GatewayProc, msg_id: str, since: int, what: str) -> Event: + """The message's processing finished AND the adapter's session guard is released. + + ``since`` must be at/after the message's answer: a message that arrives while startup + restore is still running is parked (an early processing-complete) and replayed later.""" + done = gw.wait_for(_is_complete(msg_id), TURN_DEADLINE_S, what, since=since) + assert done is not None, f"{what}: processing never completed{gw.log_tail()}" + idle = gw.wait_for(lambda e: e.ev == "hb" and not e.data.get("busy"), TURN_DEADLINE_S, what, + since=done.idx) + assert idle is not None, f"{what}: session guard never released{gw.log_tail()}" + return done + + +def _probe(gw: GatewayProc, nonce: str, n: int, chat: str, deadline: float, what: str) -> tuple[str, int, float]: + start = gw.mark() + t0 = time.monotonic() + msg_id = gw.send_user(f"are you there [[probe:{nonce}:{n}]]", chat=chat) + got = gw.wait_for(_is_send(chat, _alive(nonce, n)), deadline, what, since=start) + assert got is not None, ( + f"{what}: no answer within {deadline}s; sends={gw.sends_since(start)}{gw.log_tail()}") + return msg_id, got.idx, time.monotonic() - t0 + + +def _follow_up_through_breaker(gw: GatewayProc, srv: FakeLLMServer, scn: Scenario, nonce: str, + stats: dict[str, Any]) -> tuple[int, bool]: + """Follow-up for a scenario that may trip the stale breaker. Returns (probe n that was + answered, whether that probe ran in the faulted session).""" + start = gw.mark() + t0 = time.monotonic() + msg_id = gw.send_user(f"are you there [[probe:{nonce}:2]]", chat=CHAT) + done = gw.wait_for(_is_complete(msg_id), TURN_DEADLINE_S, "follow-up", since=start) + assert done is not None, f"{scn.id}: follow-up never completed{gw.log_tail()}" + stats["probe_s"] = round(time.monotonic() - t0, 2) + _settle(gw, msg_id, start, f"{scn.id}: follow-up") + if gw.wait_for(_is_send(CHAT, _alive(nonce, 2)), 0.1, "answer", since=start): + return 2, True + # Refused: it must be a surfaced, provider-free refusal (not a silent drop, not a storm). + assert gw.sends_since(start), f"{scn.id}: follow-up refused silently" + assert not _main_requests_for(srv, f"[[probe:{nonce}:2]]"), ( + f"{scn.id}: follow-up neither answered nor short-circuited") + stats["stale_breaker_tripped"] = True + reset_start = gw.mark() + reset_id = gw.send_user("/new", chat=CHAT) + _settle(gw, reset_id, reset_start, f"{scn.id}: /new") + probe_id, probe_answer, probe_s = _probe(gw, nonce, 3, CHAT, TURN_DEADLINE_S, f"{scn.id}: probe after /new") + stats["probe_after_new_s"] = round(probe_s, 2) + _settle(gw, probe_id, probe_answer, f"{scn.id}: probe after /new") + return 3, False + + +def run_scenario(scn: Scenario, srv: FakeLLMServer, root: Path) -> dict[str, Any]: + nonce = uuid.uuid4().hex[:10] + marker = f"[[chaos:{scn.fault}:{nonce}]]" + cfg: dict[str, Any] = {"model": scn.model} + if scn.long_timeouts: + cfg.update(request_timeout=LONG_TIMEOUT_S, stale_timeout=LONG_TIMEOUT_S) + extra = f"display:\n busy_input_mode: {scn.busy_mode}\n" if scn.busy_mode else "" + tag = new_tag() + gw = GatewayProc(root, srv.base_url, tag, cfg=cfg, extra_config=extra) + stats: dict[str, Any] = {"scenario": scn.id} + try: + t_boot = time.monotonic() + gw.start() + # Warm-up turn: the first inbound after boot waits for startup restore and builds the + # agent cold; the scenario starts only once that turn is answered and settled. + warm_mark = gw.mark() + warm_id, warm_answer, _ = _probe(gw, nonce, 0, CHAT, TURN_DEADLINE_S, f"{scn.id}: warm-up") + _settle(gw, warm_id, warm_answer, f"{scn.id}: warm-up") + stats["boot_s"] = round(time.monotonic() - t_boot, 1) + + # ── the faulted turn ── + start = gw.mark() + hb_from = start + t0 = time.monotonic() + fault_id = gw.send_user(f"please work {marker}", chat=CHAT) + deadline = TURN_DEADLINE_S + if scn.ender: + if scn.when == "tool": + assert _wait_until(lambda: _hung_tool_pids(tag), TURN_DEADLINE_S), ( + f"{scn.id}: hung tool process never spawned{gw.log_tail()}") + else: + assert _wait_until(lambda: _main_requests_for(srv, marker), TURN_DEADLINE_S), ( + f"{scn.id}: provider never saw the turn{gw.log_tail()}") + # Let the fault sit so the heartbeat samples the wedged gateway. + gw.wait_for(lambda e: e.ev == "hb", TURN_DEADLINE_S, "hb", since=gw.mark()) + gw.wait_for(lambda e: e.ev == "hb", TURN_DEADLINE_S, "hb", since=gw.mark()) + if scn.ender == "stop": + # A wedged chat must not wedge the gateway: another chat is served meanwhile. + _, _, other_s = _probe(gw, nonce, 7, OTHER_CHAT, TURN_DEADLINE_S, + f"{scn.id}: other chat while wedged") + stats["other_chat_s"] = round(other_s, 2) + t0, deadline = time.monotonic(), INTERRUPT_DEADLINE_S + gw.send_user("/stop", chat=CHAT) + elif scn.ender == "busy": + busy_deadline = INTERRUPT_DEADLINE_S if scn.long_timeouts else TURN_DEADLINE_S + t_busy = time.monotonic() + _, _, busy_s = _probe(gw, nonce, 1, CHAT, busy_deadline, + f"{scn.id}: follow-up sent while the turn hangs") + stats["busy_answer_s"] = round(busy_s, 2) + t0 = t_busy + deadline = busy_deadline + elif scn.ender == "sigterm": + gw.stop() + stats["shutdown_s"] = gw.shutdown_s and round(gw.shutdown_s, 2) + assert gw.shutdown_s is not None, ( + f"{scn.id}: gateway ignored SIGTERM for {SHUTDOWN_DEADLINE_S}s during a hung turn" + f"{gw.log_tail()}") + assert gw.exit_code is not None and gw.exit_code >= 0, ( + f"{scn.id}: gateway died by signal {gw.exit_code}{gw.log_tail()}") + + if scn.ender != "sigterm": + done = gw.wait_for(_is_complete(fault_id), max(0.0, deadline - (time.monotonic() - t0)), + "fault turn", since=start) + turn_s = time.monotonic() - t0 + assert done is not None, ( + f"{scn.id}: faulted turn still running after {deadline}s; " + f"sends={gw.sends_since(start)}{gw.log_tail()}") + stats["turn_s"] = round(turn_s, 2) + surfaced = [e for e in gw.events_between(start, done.idx) if _is_send(CHAT)(e)] + assert surfaced, f"{scn.id}: the turn ended without telling the user anything{gw.log_tail()}" + _settle(gw, fault_id, start, f"{scn.id}: faulted turn") + if scn.fault in ("parallel", "toolhang"): + assert gw.wait_for(_is_send(CHAT, f"tools settled {nonce}"), 0.1, "answer", since=start), ( + f"{scn.id}: final answer after the tool batch never delivered: {gw.sends_since(start)}") + + calls = len(_main_requests_for(srv, marker)) + stats["provider_calls"] = calls + assert 1 <= calls <= MAX_PROVIDER_CALLS_PER_FAILED_TURN, f"{scn.id}: {calls} provider calls" + + if scn.reaps_hung_tool: + assert _wait_until(lambda: not _hung_tool_pids(tag), PROCESS_REAP_S), ( + f"{scn.id}: hung tool survived its turn: {describe_pids(_hung_tool_pids(tag))}") + + state_db = gw.state_db + if scn.ender != "sigterm": + # ── the same chat must take and answer a new message, exactly once ── + n = 2 + history_kept = True + if scn.may_trip_stale_breaker: + n, history_kept = _follow_up_through_breaker(gw, srv, scn, nonce, stats) + else: + probe_id, probe_answer, probe_s = _probe(gw, nonce, n, CHAT, TURN_DEADLINE_S, + f"{scn.id}: follow-up after the fault") + stats["probe_s"] = round(probe_s, 2) + _settle(gw, probe_id, probe_answer, f"{scn.id}: follow-up") + answers = [t for t in gw.sends_since(warm_mark) if _alive(nonce, n) in t] + assert len(answers) == 1, f"{scn.id}: follow-up answered {len(answers)} times" + + gap = max((float(e.data.get("max_gap") or 0) for e in gw.events_between(hb_from) if e.ev == "hb"), + default=0.0) + stats["heartbeat_max_gap_s"] = gap + assert gap < HEARTBEAT_MAX_GAP_S, f"{scn.id}: gateway event loop froze for {gap}s" + + probe_reqs = _main_requests_for(srv, f"[[probe:{nonce}:{n}]]") + assert probe_reqs, f"{scn.id}: follow-up never reached the provider" + sent = probe_reqs[-1]["messages"] + if history_kept: + assert any(m.get("role") == "user" and marker in _text_of(m.get("content")) for m in sent), ( + f"{scn.id}: follow-up ran without the faulted turn in its history (not the same session)") + assert unanswered_tool_calls(sent) == [], f"{scn.id}: model was sent unanswered tool_calls" + + session_id = _session_with(state_db, marker) + assert session_id, f"{scn.id}: faulted user message never persisted" + persisted = _persisted(state_db, session_id) + assert unanswered_tool_calls(persisted) == [], f"{scn.id}: state.db has unanswered tool_calls" + if scn.fault == "parallel": + _assert_parallel_results(persisted, nonce, "state.db") + _assert_parallel_results(sent, nonce, "provider request") + + # ── SIGTERM: the gateway leaves on its own and takes every child with it ── + gw.stop() + stats["shutdown_s"] = gw.shutdown_s and round(gw.shutdown_s, 2) + assert gw.shutdown_s is not None, f"{scn.id}: gateway ignored SIGTERM{gw.log_tail()}" + assert gw.exit_code is not None and gw.exit_code >= 0, ( + f"{scn.id}: gateway died by signal {gw.exit_code}{gw.log_tail()}") + else: + session_id = _session_with(state_db, marker) + if session_id: + assert unanswered_tool_calls(_persisted(state_db, session_id)) == [] + + survivors = wait_no_tagged(tag) + assert survivors == [], f"{scn.id}: orphans after exit: {describe_pids(survivors)}" + assert integrity_ok(state_db) == "ok" + return stats + finally: + gw.stop() + kill_tagged(tag) + + +# ── pytest wiring: all scenarios run concurrently (one gateway each), asserted per test ── + + +@pytest.fixture(scope="module") +def scenario_futures(request: pytest.FixtureRequest, tmp_path_factory: pytest.TempPathFactory): + # Only the scenarios selected for this session (-k) are started. + selected = { + item.callspec.params["scn"].id for item in request.session.items + if getattr(item, "callspec", None) is not None and "scn" in item.callspec.params + } + with FakeLLMServer(responder) as srv, ThreadPoolExecutor( + max_workers=SCENARIO_WORKERS, thread_name_prefix="chaos-gw") as pool: + futures: dict[str, Future] = { + scn.id: pool.submit(run_scenario, scn, srv, tmp_path_factory.mktemp(scn.id)) + for scn in SCENARIOS if scn.id in selected + } + yield futures + for fut in futures.values(): + fut.cancel() + + +@pytest.mark.parametrize("scn", SCENARIOS, ids=[s.id for s in SCENARIOS]) +def test_gateway_turn_stays_live_under_fault(scn: Scenario, scenario_futures) -> None: + stats = scenario_futures[scn.id].result(timeout=len(SCENARIOS) * (TURN_DEADLINE_S * 2 + 60)) + print(json.dumps(stats)) From cf320e61217949d4f190a41f4c0326b40650b694 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:12:24 -0700 Subject: [PATCH 018/104] test: C8 liveness chaos matrix through the real tui_gateway JSON-RPC server python -m tui_gateway.entry over stdio: every fault mode ends with a terminal message.complete within its deadline (interrupt within seconds), the dispatcher keeps answering during a wedged turn, the busy flag is released so the same session streams the next prompt, tool results are complete, and exit on EOF/SIGTERM mid-turn leaves no orphans (the foreground-tool orphan on exit is a known base bug, strict xfail). --- tests/e2e/core/chaos/_tui_rpc.py | 236 +++++++++ .../chaos/test_tui_gateway_turn_liveness.py | 464 ++++++++++++++++++ 2 files changed, 700 insertions(+) create mode 100644 tests/e2e/core/chaos/_tui_rpc.py create mode 100644 tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py diff --git a/tests/e2e/core/chaos/_tui_rpc.py b/tests/e2e/core/chaos/_tui_rpc.py new file mode 100644 index 0000000000..5ed78c0851 --- /dev/null +++ b/tests/e2e/core/chaos/_tui_rpc.py @@ -0,0 +1,236 @@ +"""Stdio JSON-RPC driver for a REAL ``python -m tui_gateway.entry`` subprocess. + +This is the process the Ink TUI and Desktop talk to: newline-delimited JSON-RPC 2.0 on +stdin/stdout, responses carry the request ``id``, server-pushed events are +``{"method": "event", "params": {"type", "session_id", "payload"}}``. The driver keeps +one reader thread that routes responses to waiters and appends every event to a log, +so tests can poll with deadlines instead of sleeping. +""" + +from __future__ import annotations + +import itertools +import json +import os +import subprocess +import threading +import time +from pathlib import Path +from typing import Any, Callable + +from tests.e2e.core.chaos._helpers import REPO_ROOT, python_exe + + +class RpcError(AssertionError): + pass + + +class TuiGatewayProcess: + def __init__(self, env: dict[str, str], cwd: Path, stderr_path: Path) -> None: + self._stderr_fh = open(stderr_path, "wb") # noqa: SIM115 - closed in close() + self.stderr_path = stderr_path + self.proc = subprocess.Popen( + [python_exe(), "-X", "faulthandler", "-m", "tui_gateway.entry"], + cwd=str(cwd), + env=env, + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=self._stderr_fh, + bufsize=0, + ) + self._ids = itertools.count(1) + self._lock = threading.Condition() + self._responses: dict[Any, dict[str, Any]] = {} + self.events: list[dict[str, Any]] = [] + self.malformed: list[bytes] = [] + self._write_lock = threading.Lock() + self._reader = threading.Thread(target=self._read_loop, name="tui-rpc-reader", daemon=True) + self._reader.start() + + # ── wire ─────────────────────────────────────────────────────────────── + def _read_loop(self) -> None: + assert self.proc.stdout is not None + for raw in iter(self.proc.stdout.readline, b""): + try: + frame = json.loads(raw) + except json.JSONDecodeError: + self.malformed.append(raw[:400]) + continue + with self._lock: + if frame.get("method") == "event": + params = dict(frame.get("params") or {}) + params["_t"] = time.monotonic() + params["_i"] = len(self.events) + self.events.append(params) + elif "id" in frame and ("result" in frame or "error" in frame): + self._responses[frame["id"]] = frame + self._lock.notify_all() + with self._lock: + self._lock.notify_all() + + def send(self, method: str, params: dict[str, Any] | None = None) -> int: + rid = next(self._ids) + line = json.dumps({"jsonrpc": "2.0", "id": rid, "method": method, "params": params or {}}) + assert self.proc.stdin is not None + with self._write_lock: + self.proc.stdin.write(line.encode() + b"\n") + self.proc.stdin.flush() + return rid + + def wait_response(self, rid: int, timeout: float) -> dict[str, Any]: + deadline = time.monotonic() + timeout + with self._lock: + while rid not in self._responses: + left = deadline - time.monotonic() + if left <= 0: + raise RpcError(f"no response to rpc id={rid} within {timeout}s{self.tail()}") + if self.proc.poll() is not None and not self._reader.is_alive(): + raise RpcError(f"gateway exited rc={self.proc.returncode} before rpc id={rid}{self.tail()}") + self._lock.wait(min(left, 0.5)) + return self._responses.pop(rid) + + def call(self, method: str, params: dict[str, Any] | None = None, *, timeout: float = 60.0) -> dict[str, Any]: + frame = self.wait_response(self.send(method, params), timeout) + if "error" in frame: + raise RpcError(f"{method} -> {frame['error']}") + return frame["result"] + + def timed_call(self, method: str, params: dict[str, Any] | None = None, *, timeout: float) -> float: + t0 = time.monotonic() + frame = self.wait_response(self.send(method, params), timeout) + if "error" in frame: + raise RpcError(f"{method} -> {frame['error']}") + return time.monotonic() - t0 + + # ── events ───────────────────────────────────────────────────────────── + def event_count(self) -> int: + with self._lock: + return len(self.events) + + def wait_event( + self, pred: Callable[[dict[str, Any]], bool], *, timeout: float, start: int = 0, + ) -> dict[str, Any] | None: + """First event at index >= ``start`` matching ``pred``; None at the deadline.""" + deadline = time.monotonic() + timeout + seen = start + with self._lock: + while True: + while seen < len(self.events): + ev = self.events[seen] + seen += 1 + if pred(ev): + return ev + left = deadline - time.monotonic() + if left <= 0 or (self.proc.poll() is not None and not self._reader.is_alive()): + return None + self._lock.wait(min(left, 0.5)) + + def events_since(self, start: int, sid: str | None = None) -> list[dict[str, Any]]: + with self._lock: + evs = list(self.events[start:]) + return [e for e in evs if sid is None or e.get("session_id") == sid] + + # ── lifecycle ────────────────────────────────────────────────────────── + def close_stdin_and_wait(self, timeout: float) -> int | None: + """EOF on stdin is how the TUI/Desktop parent says goodbye; the gateway must exit + on its own. Returns the exit code, or None if it is still alive at the deadline.""" + try: + assert self.proc.stdin is not None + self.proc.stdin.close() + except OSError: + pass + try: + return self.proc.wait(timeout=timeout) + except subprocess.TimeoutExpired: + return None + + def kill(self) -> None: + if self.proc.poll() is None: + self.proc.kill() + try: + self.proc.wait(timeout=10) + except subprocess.TimeoutExpired: + pass + self._stderr_fh.close() + + def tail(self, n: int = 4000) -> str: + try: + if not self._stderr_fh.closed: + self._stderr_fh.flush() + data = self.stderr_path.read_bytes()[-n:].decode(errors="replace") + except OSError: + data = "" + return f"\n--- gateway stderr tail ---\n{data}" if data else "" + + +class Heartbeat: + """Issue cheap existing RPCs (round-robin) every ``interval`` s on their own thread and + record each round-trip latency — the dispatcher must keep answering while a turn is wedged.""" + + def __init__(self, gw: TuiGatewayProcess, calls: list[tuple[str, dict[str, Any]]], *, + interval: float = 0.15, per_call_timeout: float = 30.0) -> None: + self.gw, self.calls = gw, calls + self.interval, self.per_call_timeout = interval, per_call_timeout + self.latencies: list[float] = [] + self.failures: list[str] = [] + self._stop = threading.Event() + self._thread = threading.Thread(target=self._run, name="tui-heartbeat", daemon=True) + + def _run(self) -> None: + n = 0 + while not self._stop.is_set(): + method, params = self.calls[n % len(self.calls)] + n += 1 + self._inflight_since = time.monotonic() + try: + self.latencies.append(self.gw.timed_call(method, params, timeout=self.per_call_timeout)) + except Exception as exc: # recorded and asserted by the test + self.failures.append(repr(exc)[:300]) + if len(self.failures) > 3: + return + finally: + self._inflight_since = None + self._stop.wait(self.interval) + + def __enter__(self) -> "Heartbeat": + self._inflight_since: float | None = None + self._thread.start() + return self + + def __exit__(self, *_exc: object) -> None: + self._stop.set() + self._thread.join(5.0) + # A call still unanswered when the window closes counts with its age so far: a + # dispatcher wedged for the whole turn must not look healthy by answering nothing. + since = self._inflight_since + if self._thread.is_alive() and since is not None: + self.latencies.append(time.monotonic() - since) + + +def gateway_cwd(root: Path) -> Path: + work = root / "work" + work.mkdir(parents=True, exist_ok=True) + return work + + +# System dirs only: user-level CLIs (``gh``, ``node``…) that the gateway probes in the +# background would otherwise make process hygiene depend on the developer's machine. +_SYSTEM_PATH = ("/usr/local/bin", "/usr/bin", "/bin") + + +def env_for_gateway(base_env: dict[str, str], work: Path) -> dict[str, str]: + env = dict(base_env) + env["TERMINAL_ENV"] = "local" + env["HERMES_YOLO_MODE"] = "1" # the terminal tool must never wait on an approval prompt + # The child's HOME *is* the tmp root, so the live-DB guard (pytest ancestry) would + # take our tmp state.db for "production"; the documented child opt-out is safe here. + env["HERMES_STATE_DB_GUARD_BYPASS"] = "1" + env["PWD"] = str(work) + tmp = work.parent / "tmp" + tmp.mkdir(exist_ok=True) + env["TMPDIR"] = str(tmp) # tool snapshots/spill files stay inside the scenario root + env["TERMINAL_CWD"] = str(work) + env["PATH"] = os.pathsep.join((str(Path(python_exe()).parent), *_SYSTEM_PATH)) + for key in ("DISPLAY", "WAYLAND_DISPLAY", "DBUS_SESSION_BUS_ADDRESS", "SSH_AUTH_SOCK"): + env.pop(key, None) + return env diff --git a/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py b/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py new file mode 100644 index 0000000000..9a36f6895b --- /dev/null +++ b/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py @@ -0,0 +1,464 @@ +"""C8 chaos: agent-turn liveness through the REAL tui_gateway JSON-RPC subprocess. + +``python -m tui_gateway.entry`` is the exact process the Ink TUI and Desktop drive over +stdio. Only the LLM provider is faked (``tests/fakes/fake_llm_provider``); the agent, +tools, SessionDB and the RPC dispatcher are the shipped code. Each scenario starts its +own gateway (so process hygiene is judged per fault), opens a session, submits a turn +whose provider or tool misbehaves FOREVER, and then checks the same invariants: + +* the turn ends with a terminal ``message.complete`` within TURN_DEADLINE_S (an + interrupt ends it within INTERRUPT_DEADLINE_S even though every configured timeout is + LONG_TIMEOUT_S) and the session then settles (``session.info`` running=false); +* provider main calls stay <= MAX_PROVIDER_CALLS_PER_FAILED_TURN (no retry storm); +* the dispatcher keeps answering cheap RPCs DURING the wedged turn (heartbeat); +* the same session accepts a new prompt (``status: streaming``, not queued behind a + leaked busy flag) and answers it, with the faulted turn still in its history; +* every tool_call has its own result, in state.db and in what is sent back to the model + (8-way parallel batch: each id carries ITS token, #93251); +* hung tool processes are reaped, EOF on stdin makes the gateway exit by itself, no + process carrying the scenario tag survives, and state.db passes integrity_check. + +Exit scenarios drop the client (stdin EOF) or SIGTERM the gateway WHILE the turn is wedged: +it must exit within EXIT_TIMEOUT_S, take the hung request/tool tree with it, and leave no +unanswered tool_call in state.db. +""" + +from __future__ import annotations + +import contextlib +import json +import os +import re +import signal +import subprocess +import sys +import time +import uuid +from concurrent.futures import Future, ThreadPoolExecutor +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Callable + +import pytest + +from tests.e2e.core.chaos._helpers import ( + INTERRUPT_DEADLINE_S, + LONG_TIMEOUT_S, + MAX_PROVIDER_CALLS_PER_FAILED_TURN, + PLAIN_MODEL, + PROCESS_REAP_S, + REASONING_MODEL, + TOOL_TIMEOUT_S, + TURN_DEADLINE_S, + describe_pids, + hermetic_env, + integrity_ok, + kill_tagged, + new_tag, + persisted_messages, + tagged_pids, + tool_results_by_id, + unanswered_tool_calls, + wait_no_tagged, + write_chaos_home, +) +from tests.e2e.core.chaos._tui_rpc import ( + Heartbeat, + TuiGatewayProcess, + env_for_gateway, + gateway_cwd, +) +from tests.fakes.fake_llm_provider import ( + DropMidStream, + Error, + FakeLLMServer, + Hang, + StallMidStream, + Text, + ToolCall, +) + +pytestmark = [ + pytest.mark.skipif(sys.platform != "linux", reason="orphan scan reads /proc; POSIX stdio gateway"), +] + +# The healthy dispatcher answers ping/session.status in ~1 ms; a dispatcher blocked by the +# turn answers only when the fault ends (>= 10 s here, 600 s for interrupt scenarios). +HEARTBEAT_MAX_S = 3.0 +HEARTBEAT_INTERVAL_S = 0.15 +MIN_HEARTBEATS = 5 +READY_TIMEOUT_S = 60.0 +SETTLE_TIMEOUT_S = 30.0 +EXIT_TIMEOUT_S = 30.0 +PARALLEL_CALLS = 8 +HUNG_COMMAND = "sleep 3600" +# The first turn pays the cold agent build (imports, tool registry): setup, not the +# invariant under test, so it gets a budget sized for a box that is loaded 3x over. +WARMUP_DEADLINE_S = 300.0 + +_MARK = re.compile(r"\[\[chaos:(?P[a-z0-9_]+):(?P[0-9a-f]+)\]\]") +_PROBE = re.compile(r"\[\[probe:(?P[0-9a-f]+)\]\]") + + +def _text_of(content: Any) -> str: + if isinstance(content, str): + return content + if isinstance(content, list): + return " ".join(str(p.get("text", "")) if isinstance(p, dict) else str(p) for p in content) + return str(content or "") + + +def _deciding_user_text(messages: list[dict[str, Any]]) -> str: + """The newest user message carrying a scenario/probe marker. Synthetic user turns the + agent adds itself (stream-continuation nudges after a partial reply) carry neither, so a + fault keeps applying to the whole turn — "forever" means forever.""" + for msg in reversed(messages): + if msg.get("role") == "user": + text = _text_of(msg.get("content")) + if _PROBE.search(text) or _MARK.search(text): + return text + return "" + + +def _token(nonce: str, i: int) -> str: + return f"tok{nonce}n{i}" + + +# ── scripted provider: one module-wide server, faults keyed by the prompt marker ──────── + + +def _tools_then_text(first: Callable[[str], ToolCall]) -> Callable[[list, str], Any]: + def respond(messages: list[dict[str, Any]], nonce: str): + if messages and messages[-1].get("role") == "tool": + return Text(f"tools settled {nonce}") + return first(nonce) + return respond + + +def _parallel_batch(nonce: str) -> ToolCall: + calls = [("terminal", {"command": f"echo {_token(nonce, i)}"}) for i in range(PARALLEL_CALLS)] + return ToolCall(calls[0][0], calls[0][1], parallel=calls[1:]) + + +FAULTS: dict[str, Callable[[list, str], Any]] = { + "hang": lambda _m, _n: Hang(), + "stall": lambda _m, _n: StallMidStream(), + "err500": lambda _m, _n: Error(500), + "err429": lambda _m, _n: Error(429), + "drop": lambda _m, _n: DropMidStream(), + "parallel": _tools_then_text(_parallel_batch), + "toolhang": _tools_then_text( + lambda _n: ToolCall("terminal", {"command": HUNG_COMMAND, "timeout": TOOL_TIMEOUT_S})), + "toolhang_long": _tools_then_text( + lambda _n: ToolCall("terminal", {"command": HUNG_COMMAND, "timeout": LONG_TIMEOUT_S})), +} + + +def responder(record: dict[str, Any]): + messages = record["body"].get("messages") or [] + deciding = _deciding_user_text(messages) + if probe := _PROBE.search(deciding): + return Text(f"alive {probe['nonce']}") + if mark := _MARK.search(deciding): + return FAULTS[mark["fault"]](messages, mark["nonce"]) + return Text("unscripted") + + +def _main_requests_for(srv: FakeLLMServer, needle: str) -> list[dict[str, Any]]: + return [ + r["body"] for r in list(srv.requests) + if r["kind"] == "main" and needle in _deciding_user_text(r["body"].get("messages") or []) + ] + + +# ── scenario matrix ────────────────────────────────────────────────────────────────────── + + +@dataclass(frozen=True) +class Scenario: + id: str + fault: str + model: str = PLAIN_MODEL + long_timeouts: bool = False + # Only the stale-stream detector may end the turn (#115024): the transport timeout is long. + long_request_timeout: bool = False + wait_for: str | None = None # act once the fault is live: "provider" | "tool" + action: str | None = None # "interrupt" | "eof" | "sigterm" + reaps_hung_tool: bool = False + + +SCENARIOS = [ + Scenario("provider_hang", "hang"), + # Reasoning slug + long transport timeout: only the explicit stale_timeout_seconds can end + # the stalled stream, and it must win over the reasoning-model floor. + Scenario("provider_stall_reasoning_model", "stall", model=REASONING_MODEL, long_request_timeout=True), + Scenario("provider_500_forever", "err500"), + Scenario("provider_429_forever", "err429"), + Scenario("provider_drop_mid_stream", "drop"), + Scenario("parallel_8_terminal_calls", "parallel"), + Scenario("tool_hang_past_timeout", "toolhang", reaps_hung_tool=True), + Scenario("interrupt_during_provider_hang", "hang", long_timeouts=True, wait_for="provider", action="interrupt"), + Scenario("interrupt_during_hung_tool", "toolhang_long", long_timeouts=True, + wait_for="tool", action="interrupt", reaps_hung_tool=True), + # The parent (TUI/Desktop) goes away while the turn is wedged: the gateway must still + # leave on its own and take the hung request / tool process with it. + Scenario("stdin_eof_during_provider_hang", "hang", long_timeouts=True, wait_for="provider", action="eof"), + Scenario("stdin_eof_during_hung_tool", "toolhang_long", long_timeouts=True, wait_for="tool", action="eof"), + Scenario("sigterm_during_hung_tool", "toolhang_long", long_timeouts=True, wait_for="tool", action="sigterm"), + Scenario("sigterm_during_provider_hang", "hang", long_timeouts=True, wait_for="provider", action="sigterm"), +] +EXIT_ACTIONS = ("eof", "sigterm") + + +def _is(ev: dict[str, Any], kind: str, sid: str) -> bool: + return ev.get("type") == kind and ev.get("session_id") == sid + + +def _hung_tool_pids(tag: str) -> list[int]: + out = [] + for pid in tagged_pids(tag): + try: + cmd = Path(f"/proc/{pid}/cmdline").read_bytes().replace(b"\0", b" ").decode(errors="replace") + except OSError: + continue + if HUNG_COMMAND in cmd: + out.append(pid) + return out + + +def _wait_until(pred: Callable[[], object], timeout: float) -> bool: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + if pred(): + return True + time.sleep(0.05) + return bool(pred()) + + +def _assert_parallel_results(messages: list[dict[str, Any]], nonce: str, where: str) -> None: + results = tool_results_by_id(messages) + seen = 0 + for msg in messages: + calls = msg.get("tool_calls") + if msg.get("role") != "assistant" or not calls: + continue + for call in json.loads(calls) if isinstance(calls, str) else calls: + args = json.loads(call["function"]["arguments"]) + token = args["command"].split()[-1] + assert token.startswith(f"tok{nonce}"), (where, args) + body = results.get(call["id"], "") + assert token in body and "Result unavailable" not in body, ( + f"{where}: tool_call {call['id']} ({token}) got {body[:200]!r}") + seen += 1 + assert seen == PARALLEL_CALLS, f"{where}: {seen} tool calls, expected {PARALLEL_CALLS}" + + +def _wait_settled(gw: TuiGatewayProcess, sid: str, done: dict[str, Any], what: str) -> None: + """``message.complete`` is emitted BEFORE the worker clears ``running``; the settled + ``session.info`` (running=false) that follows is the idle signal a client waits for.""" + settled = gw.wait_event( + lambda e: _is(e, "session.info", sid) and (e.get("payload") or {}).get("running") is False, + timeout=SETTLE_TIMEOUT_S, start=done["_i"] + 1) + assert settled is not None, f"{what}: session never settled (running stays true){gw.tail()}" + + +def _answered_turn(gw: TuiGatewayProcess, sid: str, nonce: str, what: str, + deadline: float = TURN_DEADLINE_S) -> float: + """Submit a probe prompt to an idle session; it must start at once and be answered.""" + start = gw.event_count() + t0 = time.monotonic() + reply = gw.call("prompt.submit", {"session_id": sid, "text": f"are you there [[probe:{nonce}]]"}) + assert reply.get("status") == "streaming", f"{what}: session busy instead of starting the turn: {reply}" + done = gw.wait_event(lambda e: _is(e, "message.complete", sid), timeout=deadline, start=start) + assert done is not None, f"{what}: prompt never completed within {deadline}s{gw.tail()}" + payload = done.get("payload") or {} + assert payload.get("status") == "complete" and f"alive {nonce}" in _text_of(payload.get("text")), ( + f"{what}: {payload}") + _wait_settled(gw, sid, done, what) + return time.monotonic() - t0 + + +def _assert_heartbeat(scn: Scenario, hb: Heartbeat, stats: dict[str, Any], turn_s: float) -> None: + assert not hb.failures, f"{scn.id}: heartbeat RPC failed during the turn: {hb.failures}" + stats["heartbeat_max_s"] = round(max(hb.latencies, default=0.0), 3) + stats["heartbeats"] = len(hb.latencies) + assert stats["heartbeat_max_s"] < HEARTBEAT_MAX_S, ( + f"{scn.id}: dispatcher blocked {stats['heartbeat_max_s']}s during the wedged turn") + assert len(hb.latencies) >= MIN_HEARTBEATS or turn_s < 2.0, ( + f"{scn.id}: only {len(hb.latencies)} heartbeats answered in a {turn_s:.1f}s turn") + + +def _exit_mid_turn(scn: Scenario, gw: TuiGatewayProcess, hb: Heartbeat, tag: str, + state_db: Path, stored: str, stats: dict[str, Any]) -> dict[str, Any]: + """The client vanishes (stdin EOF) or the supervisor stops us (SIGTERM) mid-wedge.""" + _assert_heartbeat(scn, hb, stats, turn_s=LONG_TIMEOUT_S) + t0 = time.monotonic() + if scn.action == "sigterm": + gw.proc.send_signal(signal.SIGTERM) + try: + rc = gw.proc.wait(timeout=EXIT_TIMEOUT_S) + except subprocess.TimeoutExpired: + rc = None + else: + rc = gw.close_stdin_and_wait(EXIT_TIMEOUT_S) + assert rc is not None, f"{scn.id}: gateway still alive {EXIT_TIMEOUT_S}s after {scn.action} mid-turn{gw.tail()}" + stats["exit_s"] = round(time.monotonic() - t0, 2) + survivors = wait_no_tagged(tag) + assert survivors == [], f"{scn.id}: orphans after {scn.action} mid-turn: {describe_pids(survivors)}" + assert integrity_ok(state_db) == "ok" + assert unanswered_tool_calls(persisted_messages(state_db, stored)) == [], ( + f"{scn.id}: state.db keeps a tool_call with no result after {scn.action} (next resume sends it)") + return stats + + +def _reap(gw: TuiGatewayProcess, tag: str) -> None: + """Failure-path cleanup. Kill tagged descendants BEFORE the gateway: once it dies they + reparent out of the test's process subtree, where the conftest kill guard refuses them.""" + for pid in tagged_pids(tag, exclude=(gw.proc.pid,)): + with contextlib.suppress(OSError, RuntimeError): + os.kill(pid, signal.SIGKILL) + gw.kill() + with contextlib.suppress(OSError, RuntimeError): + kill_tagged(tag) + + +def run_scenario(scn: Scenario, srv: FakeLLMServer, root: Path) -> dict[str, Any]: + nonce = uuid.uuid4().hex[:10] + marker = f"[[chaos:{scn.fault}:{nonce}]]" + probe_marker = f"[[probe:{nonce}]]" + cfg: dict[str, Any] = {"model": scn.model} + if scn.long_timeouts: + cfg.update(request_timeout=LONG_TIMEOUT_S, stale_timeout=LONG_TIMEOUT_S) + elif scn.long_request_timeout: + cfg.update(request_timeout=LONG_TIMEOUT_S) + home, hermes_home = write_chaos_home(root, srv.base_url, **cfg) + tag = new_tag() + work = gateway_cwd(root) + env = env_for_gateway(hermetic_env(home, hermes_home, tag), work) + gw = TuiGatewayProcess(env, work, root / "gateway-stderr.log") + stats: dict[str, Any] = {"scenario": scn.id} + try: + assert gw.wait_event(lambda e: e.get("type") == "gateway.ready", timeout=READY_TIMEOUT_S), ( + f"gateway never became ready{gw.tail()}") + created = gw.call("session.create", {"cols": 100}) + sid, stored = created["session_id"], created["stored_session_id"] + # A healthy first turn builds the agent, so the fault lands mid-session and the + # heartbeat measures the dispatcher during the fault, not the cold agent build. + stats["warm_s"] = round(_answered_turn( + gw, sid, uuid.uuid4().hex[:10], f"{scn.id} warm-up", WARMUP_DEADLINE_S), 2) + + # ── the faulted turn ── + start = gw.event_count() + with Heartbeat(gw, [("ping", {}), ("session.status", {"session_id": sid})], + interval=HEARTBEAT_INTERVAL_S, per_call_timeout=LONG_TIMEOUT_S) as hb: + t0 = time.monotonic() + submitted = gw.call("prompt.submit", {"session_id": sid, "text": f"please work {marker}"}) + assert submitted.get("status") == "streaming", submitted + if scn.wait_for == "provider": + assert _wait_until(lambda: _main_requests_for(srv, marker), TURN_DEADLINE_S), ( + f"provider never saw the turn{gw.tail()}") + elif scn.wait_for == "tool": + assert gw.wait_event(lambda e: _is(e, "tool.start", sid), timeout=TURN_DEADLINE_S, start=start), ( + f"hung tool never started{gw.tail()}") + assert _wait_until(lambda: _hung_tool_pids(tag), TURN_DEADLINE_S), "hung tool process never spawned" + deadline = TURN_DEADLINE_S + if scn.action: + # Let the fault sit a moment so the heartbeat samples the wedged turn. + _wait_until(lambda: len(hb.latencies) >= MIN_HEARTBEATS, TURN_DEADLINE_S) + if scn.action == "interrupt": + t_int = time.monotonic() + assert gw.call("session.interrupt", {"session_id": sid}, timeout=INTERRUPT_DEADLINE_S)[ + "status"] == "interrupted" + t0, deadline = t_int, INTERRUPT_DEADLINE_S + done = None if scn.action in EXIT_ACTIONS else gw.wait_event( + lambda e: _is(e, "message.complete", sid), timeout=deadline, start=start) + turn_s = time.monotonic() - t0 + if scn.action in EXIT_ACTIONS: + return _exit_mid_turn(scn, gw, hb, tag, hermes_home / "state.db", stored, stats) + seen_types = sorted({str(e.get("type")) for e in gw.events_since(start, sid)}) + assert done is not None, ( + f"{scn.id}: no terminal message.complete within {deadline}s " + f"(events: {seen_types}){gw.tail()}") + stats["turn_s"] = round(turn_s, 2) + stats["status"] = (done.get("payload") or {}).get("status") + if scn.action == "interrupt": + assert stats["status"] == "interrupted", done + _wait_settled(gw, sid, done, scn.id) + + calls = len(_main_requests_for(srv, marker)) + stats["provider_calls"] = calls + assert 1 <= calls <= MAX_PROVIDER_CALLS_PER_FAILED_TURN, f"{scn.id}: {calls} provider calls" + + _assert_heartbeat(scn, hb, stats, turn_s) + + if scn.reaps_hung_tool: + assert _wait_until(lambda: not _hung_tool_pids(tag), PROCESS_REAP_S), ( + f"{scn.id}: hung tool survived its turn: {describe_pids(_hung_tool_pids(tag))}") + + # ── the same session must take and answer a new prompt ── + stats["probe_s"] = round(_answered_turn(gw, sid, nonce, f"{scn.id} after the fault"), 2) + + probe_reqs = _main_requests_for(srv, probe_marker) + assert probe_reqs, f"{scn.id}: follow-up never reached the provider" + sent = probe_reqs[-1]["messages"] + assert any(m.get("role") == "user" and marker in _text_of(m.get("content")) for m in sent), ( + f"{scn.id}: follow-up ran without the faulted turn in its history (not the same session)") + assert unanswered_tool_calls(sent) == [], f"{scn.id}: model was sent unanswered tool_calls" + + state_db = hermes_home / "state.db" + persisted = persisted_messages(state_db, stored) + assert unanswered_tool_calls(persisted) == [], f"{scn.id}: state.db has unanswered tool_calls" + if scn.fault == "parallel": + _assert_parallel_results(persisted, nonce, "state.db") + _assert_parallel_results(sent, nonce, "provider request") + + # ── EOF on stdin: the gateway leaves on its own and takes every child with it ── + rc = gw.close_stdin_and_wait(EXIT_TIMEOUT_S) + assert rc is not None, f"{scn.id}: gateway ignored stdin EOF for {EXIT_TIMEOUT_S}s{gw.tail()}" + survivors = wait_no_tagged(tag) + assert survivors == [], f"{scn.id}: orphans after exit: {describe_pids(survivors)}" + assert integrity_ok(state_db) == "ok" + return stats + finally: + _reap(gw, tag) + + +# ── pytest wiring: all scenarios run concurrently (one gateway each), asserted per test ── + + +@pytest.fixture(scope="module") +def scenario_futures(request: pytest.FixtureRequest, tmp_path_factory: pytest.TempPathFactory): + selected = { + item.callspec.params["scn"].id for item in request.session.items + if isinstance(getattr(getattr(item, "callspec", None), "params", {}).get("scn"), Scenario) + } + with FakeLLMServer(responder) as srv, ThreadPoolExecutor( + max_workers=max(1, len(selected)), thread_name_prefix="chaos-tui") as pool: + futures: dict[str, Future] = { + scn.id: pool.submit(run_scenario, scn, srv, tmp_path_factory.mktemp(scn.id)) + for scn in SCENARIOS if scn.id in selected + } + try: + yield futures + finally: + for fut in futures.values(): + fut.cancel() + + +# Real production bug on base (reported, not fixed here): when the gateway leaves mid-tool — +# client closes stdin or supervisor SIGTERMs — _shutdown_sessions() closes the agents but the +# in-flight foreground terminal command (its own process group) is never killed, so the +# `bash -c ...` + `sleep 3600` tree survives, reparented to init. strict: flips red once fixed. +_ORPHANED_FOREGROUND_TOOL = pytest.mark.xfail( + strict=True, raises=AssertionError, + reason="tui_gateway exit (EOF/SIGTERM) orphans the running foreground terminal tool's process tree") +KNOWN_BUGS = {"stdin_eof_during_hung_tool": _ORPHANED_FOREGROUND_TOOL, + "sigterm_during_hung_tool": _ORPHANED_FOREGROUND_TOOL} + + +@pytest.mark.parametrize("scn", [ + pytest.param(s, id=s.id, marks=[KNOWN_BUGS[s.id]] if s.id in KNOWN_BUGS else []) + for s in SCENARIOS]) +def test_tui_gateway_turn_stays_live_under_fault(scn: Scenario, scenario_futures) -> None: + stats = scenario_futures[scn.id].result(timeout=WARMUP_DEADLINE_S + 3 * TURN_DEADLINE_S + 120) + print(json.dumps(stats)) From a0d2907f273c65ded2f8b521803f07812cf8d076 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:52:17 -0700 Subject: [PATCH 019/104] ci(e2e): run the core suites on a 32-core runner with Node and a 900 s per-file budget --- .github/workflows/tests.yml | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 9e82ba34c0..b61f71eb57 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -151,12 +151,21 @@ jobs: key: test-durations-${{ github.run_id }} e2e: - runs-on: ubuntu-latest + # The core suites spawn real serve / gateway / tui_gateway / MCP / SQLite + # writer processes per test; a 4-vCPU runner serialises them into timeouts. + runs-on: ubuntu-latest-32-core timeout-minutes: 30 steps: - name: Checkout code uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + - name: Set up Node + # The boot-contract suite feeds serve's stdout through Desktop's own + # backend-ready.ts (native TS stripping needs Node >= 22.18). + uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 + with: + node-version: 26 + - name: Install ripgrep (prebuilt binary) run: | set -euo pipefail @@ -219,6 +228,9 @@ jobs: source .venv/bin/activate scripts/run_tests.sh --include-integration tests/e2e env: + # Multi-process episodes (torture chamber, compaction kill -9, + # gateway liveness) legitimately run past the 300 s default. + HERMES_TEST_FILE_TIMEOUT: "900" OPENROUTER_API_KEY: "" OPENAI_API_KEY: "" NOUS_API_KEY: "" From fc9b4aed1f7147f7848779e80099f967bc041716 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 05:09:11 -0700 Subject: [PATCH 020/104] test: trim trailing blank line --- tests/e2e/core/parity/_boot_contract.py | 1 - 1 file changed, 1 deletion(-) diff --git a/tests/e2e/core/parity/_boot_contract.py b/tests/e2e/core/parity/_boot_contract.py index f693e7973e..3dc922a96e 100644 --- a/tests/e2e/core/parity/_boot_contract.py +++ b/tests/e2e/core/parity/_boot_contract.py @@ -51,4 +51,3 @@ def desktop_parse(stdout_bytes: str) -> dict: ) assert proc.returncode == 0, f"desktop parser bridge crashed: {proc.stderr[-2000:]}" return json.loads(proc.stdout.strip().splitlines()[-1]) - From 239671760ab7c0d8198f5a73dae4200805dc1089 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 05:18:34 -0700 Subject: [PATCH 021/104] ci(e2e): run on a WAL-capable SQLite and fail if it isn't; tolerate Pythons without os.pidfd_open CI's pinned uv only knows CPython 3.11.14, whose bundled SQLite has the WAL-reset bug, so Hermes ran state.db in DELETE mode and every torture-chamber episode skipped (green over zero coverage). The chaos cleanup also called os.pidfd_open, which that build lacks, turning two strict xfails into errors. --- .github/workflows/tests.yml | 16 +++++++++++++--- tests/e2e/core/chaos/_helpers.py | 5 ++++- 2 files changed, 17 insertions(+), 4 deletions(-) diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index b61f71eb57..da8b950cce 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -187,7 +187,10 @@ jobs: # fetching a manifest from raw.githubusercontent.com on EVERY job — # a transient fetch failure fails the whole job (2026-07-28 slice-5 # incident). Pinned, the binary downloads directly; no manifest hop. - version: "0.9.28" + # Newer than the unit job's pin on purpose: 0.9.28 only knows CPython + # 3.11.14, whose bundled SQLite has the WAL-reset bug, so Hermes runs + # state.db in DELETE mode and the WAL torture chamber would skip. + version: "0.12.13" # Persist uv's download/wheel cache (~/.cache/uv) across runs. # Keyed on the dependency manifests, so the cache is reused until # pyproject.toml or uv.lock changes. `uv sync` still runs every @@ -199,7 +202,7 @@ jobs: uv.lock - name: Set up Python 3.11 - run: uv python install 3.11 + run: uv python install 3.11.15 - name: Install dependencies # `uv sync --locked` installs the exact pinned set from uv.lock (and @@ -213,13 +216,20 @@ jobs: # in the venv up front. uses: ./.github/actions/retry with: - command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra parallel-web + command: uv sync --locked --python 3.11.15 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra parallel-web - name: Minimize uv cache # Optimized for CI: prunes pre-built wheels that are cheap to # re-download, keeping the persisted cache small and fast to restore. run: uv cache prune --ci + - name: Require a WAL-capable SQLite + # The state.db suites skip on a WAL-reset-vulnerable SQLite; fail + # instead of reporting green over zero coverage. + run: | + source .venv/bin/activate + python -c "import sqlite3, hermes_state_wal as w; print('sqlite', sqlite3.sqlite_version); assert not w.is_sqlite_wal_reset_vulnerable()" + - name: Run e2e tests # One subprocess per file, in parallel: the core suites spawn real # processes (serve, gateway, tui_gateway, MCP servers, SQLite writers) diff --git a/tests/e2e/core/chaos/_helpers.py b/tests/e2e/core/chaos/_helpers.py index e21f7f76a2..9ab7f37980 100644 --- a/tests/e2e/core/chaos/_helpers.py +++ b/tests/e2e/core/chaos/_helpers.py @@ -190,8 +190,11 @@ def kill_tagged(tag: str) -> None: except OSError: pass except RuntimeError: + pidfd_open = getattr(os, "pidfd_open", None) # absent on some Python builds + if pidfd_open is None: + continue try: - fd = os.pidfd_open(pid) + fd = pidfd_open(pid) except OSError: continue try: From 84a4b3383438dd1fe8fda0a16ca323d834e01d78 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 07:33:24 -0700 Subject: [PATCH 022/104] test(e2e): run the state.db torture chamber and compaction suites in DELETE journal mode too Both suites skipped outright on a WAL-reset-vulnerable SQLite, so the DELETE-mode population (every user whose Python bundles SQLite 3.7.0-3.51.2, e.g. uv CPython 3.11.14's 3.50.4, and the network/FUSE homes Hermes also falls back to DELETE on) had no multi-process integrity coverage. - Parametrize both module fixtures over journal mode [wal, delete]. The delete arm goes through the production decision path: each child (_roles.py, including the `hermes` CLI run as __main__) pins the version hermes_state_wal.is_sqlite_wal_reset_vulnerable() reports to 3.50.4, and apply_wal_with_fallback picks DELETE itself; the harness never issues a journal-mode pragma. The wal arm skips only where Hermes would not run WAL; the delete arm runs everywhere. - Every invariant that holds in both modes stays on in both: integrity_check, acknowledged writes exactly once, monotonic counts, repair never lowers rows, FTS parity, fd bounds, persisted == replayed. New in both: the store is still in the arm's mode and every SessionDB role reported the matching _wal_active (vacuity guard for the seam). WAL-only: the deleted -wal/-shm fd scan (the deleted main-file scan runs in both). - DELETE mode blocks readers on writes and has no writer fairness: in that arm a SQLITE_BUSY refusal of a read/open/FTS pass is waited out and counted (reported in failure context), and plain writers are paced by 20 ms; in the wal arm busy still fails the role. - chmod_flip runs 3-5 flips instead of 6-10 (same fault, process start-up dominated the time). Red-proof (delete arm red, wal arm green): repair without live-writer checks/exclusion -> "acked ... stored 0x"; a DELETE-mode commit misreported busy so the retry re-runs the insert -> "acked ... stored 2x" (torture and compaction); default config enabling WAL on a vulnerable SQLite -> "journal_mode is 'wal' (arm is delete)" + "_wal_active=True in the delete arm". --- tests/e2e/core/sqlite/_helpers.py | 91 +++++++++++-- tests/e2e/core/sqlite/_roles.py | 126 ++++++++++++++---- .../core/sqlite/test_compaction_contention.py | 29 ++-- tests/e2e/core/sqlite/test_torture_chamber.py | 50 +++---- 4 files changed, 221 insertions(+), 75 deletions(-) diff --git a/tests/e2e/core/sqlite/_helpers.py b/tests/e2e/core/sqlite/_helpers.py index 649d7d09c1..a752ab2e80 100644 --- a/tests/e2e/core/sqlite/_helpers.py +++ b/tests/e2e/core/sqlite/_helpers.py @@ -1,11 +1,19 @@ """Harness for the multi-process SQLite torture chamber (issue class C1: state.db integrity). -One real ``state.db`` in WAL mode under ``tmp_path``; every role is a separate OS process running -``_roles.py`` against it (see that module for the journal/report protocol). This module owns: +One real ``state.db`` under ``tmp_path``, in either journal mode Hermes deploys (see ``JOURNAL_MODES``); every +role is a separate OS process running ``_roles.py`` against it (see that module for the journal/report +protocol). This module owns: + +* journal-mode selection through the PRODUCTION decision path: the ``delete`` arm pins the version the + children's ``hermes_state_wal.is_sqlite_wal_reset_vulnerable()`` probe sees to a WAL-reset-vulnerable + SQLite (the one CI's uv CPython 3.11.14 bundles), so ``apply_wal_with_fallback`` itself picks DELETE + exactly as it does for every user on such a build — the same DELETE store its network/FUSE fallbacks end + in; nothing in the harness issues a journal-mode pragma; * process lifecycle (spawn / ready / stop / SIGTERM / SIGKILL, PIDs recorded, nothing else touched); -* a background ``/proc//fd`` monitor over OUR children that records any ``(deleted)`` ``-wal``/``-shm`` - (or main-file) descriptor — the kernel-level signature of a WAL generation unlinked under a live holder; +* a background ``/proc//fd`` monitor over OUR children that records any ``(deleted)`` main-file descriptor + (a store swapped under a live holder) and, in WAL mode, any ``(deleted)`` ``-wal``/``-shm`` — the + kernel-level signature of a WAL generation unlinked under a live holder; * the invariant checks, done in the test process with short-lived bare ``sqlite3`` connections only (the test process never imports ``hermes_state``, so it never becomes a foreign holder of the file). @@ -26,12 +34,34 @@ import time import zlib from pathlib import Path +from hermes_cli.sqlite_runtime import is_sqlite_wal_reset_vulnerable from tests.conformance.persistence._harness import REPO_ROOT, kill9_and_reap, wait_for ROLES = Path(__file__).with_name("_roles.py") DEFAULT_SEED = 20260923 SEED_ENV = "HERMES_SQLITE_TORTURE_SEED" +JOURNAL_MODES = ("wal", "delete") +# Read ONLY by _roles.py in each child: the SQLite version its production version probe reports. +SQLITE_PIN_ENV = "HERMES_E2E_SQLITE_VERSION_PIN" +VULNERABLE_SQLITE = "3.50.4" # bundled by uv's CPython 3.11.14 (the unit CI job): Hermes runs DELETE there +# DELETE mode holds an EXCLUSIVE lock for every commit's journal+db fsyncs and has no writer fairness: an unpaced +# append loop starves every other writer and reader. Gateway/TUI writers are paced by turns in the field. +DELETE_WRITER_PACE = 0.02 + + +def linked_sqlite_is_wal_capable() -> bool: + """The production predicate on the SQLite this interpreter (and so every child) links.""" + return not is_sqlite_wal_reset_vulnerable(sqlite3.sqlite_version_info) + + +def skip_unless_deployable(journal: str) -> None: + """WAL is not what Hermes runs on a vulnerable SQLite, so that arm is not deployable there; DELETE always is.""" + if journal == "wal" and not linked_sqlite_is_wal_capable(): + import pytest + + pytest.skip(f"linked SQLite {sqlite3.sqlite_version} runs Hermes in DELETE mode; the delete arm covers it") + def base_seed() -> int: return int(os.environ.get(SEED_ENV) or DEFAULT_SEED) @@ -59,17 +89,22 @@ def child_env(home: Path, hermes_home: Path) -> dict: class Chamber: """A private HERMES_HOME + state.db plus the processes playing roles against it.""" - def __init__(self, root: Path, *, journal_mode: str = "wal"): + def __init__(self, root: Path, *, journal: str = "wal"): + assert journal in JOURNAL_MODES, journal self.root = root + self.mode = journal self.home = root / "home" self.hermes_home = self.home / ".hermes" self.hermes_home.mkdir(parents=True, exist_ok=True) - (self.hermes_home / "config.yaml").write_text( - f"database:\n journal_mode: {journal_mode}\n", encoding="utf-8") + # The config default in BOTH arms: the delete arm is a default-config user on a vulnerable SQLite. + (self.hermes_home / "config.yaml").write_text("database:\n journal_mode: wal\n", encoding="utf-8") self.db = self.hermes_home / "state.db" self.work = root / "work" self.work.mkdir(exist_ok=True) self.env = child_env(self.home, self.hermes_home) + if journal == "delete" and linked_sqlite_is_wal_capable(): + self.env[SQLITE_PIN_ENV] = VULNERABLE_SQLITE + self.writer_pace = DELETE_WRITER_PACE if journal == "delete" else 0.0 self.procs: dict[str, subprocess.Popen] = {} self.writer_runs: list[str] = [] self.reader_seq = 0 @@ -84,8 +119,10 @@ class Chamber: # -- lifecycle --------------------------------------------------------------------------------- def spawn(self, role: str, name: str, *, env: dict | None = None, **args) -> subprocess.Popen: assert name not in self.procs, f"duplicate role name {name}" + if role == "writer": + args.setdefault("pace", self.writer_pace) payload = {"workdir": str(self.work), "name": name, "db": str(self.db), - "stop": str(self.work / f"{name}.stop"), **args} + "stop": str(self.work / f"{name}.stop"), "busy_ok": self.mode == "delete", **args} stderr = open(self.work / f"{name}.stderr", "wb") # noqa: SIM115 - closed in reap() proc = subprocess.Popen( [sys.executable, str(ROLES), role, json.dumps(payload)], @@ -100,10 +137,12 @@ class Chamber: return proc def spawn_cli(self, name: str, *argv: str) -> subprocess.Popen: - """A real `hermes …` CLI subprocess against this HERMES_HOME.""" + """A real `hermes …` CLI subprocess against this HERMES_HOME (``hermes_cli.main`` run as ``__main__`` + by ``_roles.py cli`` so the journal-mode seam applies to it too).""" stderr = open(self.work / f"{name}.stderr", "wb") # noqa: SIM115 - closed in reap() + payload = {"workdir": str(self.work), "name": name, "argv": list(argv)} proc = subprocess.Popen( - [sys.executable, "-m", "hermes_cli.main", *argv], cwd=str(REPO_ROOT), env=self.env, + [sys.executable, str(ROLES), "cli", json.dumps(payload)], cwd=str(REPO_ROOT), env=self.env, stdin=subprocess.DEVNULL, stdout=open(self.work / f"{name}.stdout", "wb"), stderr=stderr, # noqa: SIM115 ) proc._stderr_file = stderr # type: ignore[attr-defined] @@ -183,7 +222,9 @@ class Chamber: # -- kernel truth: (deleted) sidecars held by our children -------------------------------------- def _scan_loop(self) -> None: - targets = {str(self.db), f"{self.db}-wal", f"{self.db}-shm"} + targets = {str(self.db)} + if self.mode == "wal": + targets |= {f"{self.db}-wal", f"{self.db}-shm"} while not self._monitor_stop.is_set(): for name, proc in self.live(): fd_dir = f"/proc/{proc.pid}/fd" @@ -326,6 +367,34 @@ def fts_problems(db: Path, sample_tokens: list[str]) -> list[str]: return problems +def journal_mode_problems(chamber: Chamber, prefix: str = "") -> list[str]: + """The store is still in the arm's journal mode, and every production SessionDB role (whose ``ready`` + event reports ``SessionDB._wal_active``) really ran in it — the seam took, the arm is not vacuous.""" + problems = [] + mode = journal_mode(chamber.db) + if mode != chamber.mode: + problems.append(f"journal_mode is {mode!r} after the episode (arm is {chamber.mode})") + want = chamber.mode == "wal" + for name in list(chamber.procs): + if not name.startswith(prefix): + continue + ready = [e for e in chamber.events(name) if e.get("event") == "ready" and "wal" in e] + if ready and ready[0]["wal"] != want: + problems.append(f"{name} opened SessionDB with _wal_active={ready[0]['wal']} in the {chamber.mode} arm") + return problems + + +def busy_summary(chamber: Chamber, prefix: str = "") -> dict[str, int]: + """DELETE arm only: SQLITE_BUSY refusals the roles waited out, per operation (availability, not integrity).""" + out: dict[str, int] = {} + for name in list(chamber.procs): + if name.startswith(prefix): + for e in chamber.events(name): + if e.get("event") == "busy": + out[e["op"]] = out.get(e["op"], 0) + 1 + return out + + def exactly_once_problems(chamber: Chamber) -> list[str]: """Every acknowledged append is stored exactly once; an unacknowledged in-flight append at most once; no stored writer row that no writer ever intended.""" diff --git a/tests/e2e/core/sqlite/_roles.py b/tests/e2e/core/sqlite/_roles.py index f9ddc59d1a..7221bf2349 100644 --- a/tests/e2e/core/sqlite/_roles.py +++ b/tests/e2e/core/sqlite/_roles.py @@ -9,6 +9,10 @@ connection) and reports through append-only files, so a ``kill -9`` loses nothin Tokens are single FTS words (``TK`` + alnum) so the test can look every acknowledged append up by content, through ``messages`` and through the FTS indexes. + +Journal-mode seam: with ``HERMES_E2E_SQLITE_VERSION_PIN`` set, the production version probe +``hermes_state_wal.is_sqlite_wal_reset_vulnerable()`` reports that SQLite version instead of the linked one; +the real range predicate and ``apply_wal_with_fallback`` then decide the journal mode as they would there. """ from __future__ import annotations @@ -52,6 +56,37 @@ class Out: os.write(self.report_fd, (json.dumps(event) + "\n").encode()) +def _apply_sqlite_version_pin() -> None: + pin = os.environ.get("HERMES_E2E_SQLITE_VERSION_PIN") + if not pin: + return + import hermes_state_wal + + pinned = tuple(int(p) for p in pin.split(".")) + probe = hermes_state_wal.is_sqlite_wal_reset_vulnerable + + def is_sqlite_wal_reset_vulnerable(version_info=None): + return probe(pinned if version_info is None else version_info) + + hermes_state_wal.is_sqlite_wal_reset_vulnerable = is_sqlite_wal_reset_vulnerable + + +def _patient(a: dict, out: Out, op: str, fn, *, deadline: float = 90.0): + """Run ``fn()``; in the DELETE arm (``busy_ok``) a SQLITE_BUSY refusal is reported as a ``busy`` event and + retried. DELETE mode is documented to block readers on writes (hermes_state_wal), so a busy read/open is an + availability event there, never an integrity one. In the WAL arm it propagates and fails the role.""" + end = time.monotonic() + deadline + while True: + try: + return fn() + except sqlite3.OperationalError as exc: + busy = any(m in str(exc).lower() for m in ("database is locked", "database is busy")) + if not (busy and a.get("busy_ok")) or time.monotonic() > end: + raise + out.report(event="busy", op=op, error=repr(exc)) + time.sleep(0.05) + + def _fd_count() -> int: return len(os.listdir("/proc/self/fd")) if os.path.isdir("/proc/self/fd") else -1 @@ -127,18 +162,22 @@ def role_reader(a: dict, out: Out) -> int: from hermes_state import SessionDB db_path, stop_file = Path(a["db"]), Path(a["stop"]) - db = SessionDB(db_path=db_path) - out.report(event="ready", fds=_fd_count()) + db = _patient(a, out, "open", lambda: SessionDB(db_path=db_path)) + out.report(event="ready", wal=bool(getattr(db, "_wal_active", False)), fds=_fd_count()) seen: dict[str, int] = {} passes = 0 + + def _pass() -> None: + for row in db.list_sessions_rich(limit=200): + sid = row["id"] + n = db.message_count(sid) + if n < seen.get(sid, 0): + out.report(event="error", error=f"count went down for {sid}: {seen[sid]} -> {n}") + seen[sid] = max(n, seen.get(sid, 0)) + try: while not _stopping(stop_file): - for row in db.list_sessions_rich(limit=200): - sid = row["id"] - n = db.message_count(sid) - if n < seen.get(sid, 0): - out.report(event="error", error=f"count went down for {sid}: {seen[sid]} -> {n}") - seen[sid] = max(n, seen.get(sid, 0)) + _patient(a, out, "read", _pass) passes += 1 if passes % 5 == 0: out.report(event="stats", passes=passes, fds=_fd_count(), total=sum(seen.values())) @@ -161,16 +200,24 @@ def role_churn(a: dict, out: Out) -> int: db_path = Path(a["db"]) start_fds = _fd_count() fds_after_warmup = None + + def _hermes_cycle() -> None: + db = SessionDB(db_path=db_path) + try: + db.message_count() + finally: + db.close() + + def _raw_cycle() -> None: + conn = sqlite3.connect(str(db_path), timeout=30.0) + try: + conn.execute("SELECT count(*) FROM messages").fetchone() + finally: + conn.close() + try: for i in range(int(a["iterations"])): - if i % 2 == 0: - db = SessionDB(db_path=db_path) - db.message_count() - db.close() - else: - conn = sqlite3.connect(str(db_path), timeout=30.0) - conn.execute("SELECT count(*) FROM messages").fetchone() - conn.close() + _patient(a, out, "churn", _raw_cycle if i % 2 else _hermes_cycle) if i == 3: fds_after_warmup = _fd_count() except BaseException as exc: @@ -184,16 +231,23 @@ def role_churn(a: dict, out: Out) -> int: def role_opener(a: dict, out: Out) -> int: """One short-lived process: open, count, close, exit (the last-close checkpoint path).""" db_path = Path(a["db"]) - try: + + def _open_count_close() -> int: if a.get("raw"): conn = sqlite3.connect(str(db_path), timeout=30.0) - n = conn.execute("SELECT count(*) FROM messages").fetchone()[0] - conn.close() - else: - from hermes_state import SessionDB - db = SessionDB(db_path=db_path) - n = db.message_count() + try: + return conn.execute("SELECT count(*) FROM messages").fetchone()[0] + finally: + conn.close() + from hermes_state import SessionDB + db = SessionDB(db_path=db_path) + try: + return db.message_count() + finally: db.close() + + try: + n = _patient(a, out, "open", _open_count_close) except BaseException as exc: # may_fail: the chmod episode opens a read-only file; a clean refusal is correct, damage is not. if a.get("may_fail"): @@ -217,11 +271,11 @@ def role_fts(a: dict, out: Out) -> int: """Maintenance pass: full FTS rebuild + optimize through SessionDB (cross-process admission).""" from hermes_state import SessionDB - db = SessionDB(db_path=Path(a["db"])) + db = _patient(a, out, "open", lambda: SessionDB(db_path=Path(a["db"]))) out.report(event="ready") try: - rebuilt = db.rebuild_fts() - optimized = db.optimize_fts() + rebuilt = _patient(a, out, "fts", db.rebuild_fts) + optimized = _patient(a, out, "fts", db.optimize_fts) except BaseException as exc: out.report(event="error", error=repr(exc), tb=traceback.format_exc()[-3000:]) return 2 @@ -265,7 +319,7 @@ def role_agent(a: dict, out: Out) -> int: session_db=db, session_id=a["session_id"], skip_context_files=True, skip_memory=True) history = db.get_messages_as_conversation(a["session_id"]) if a.get("resume") else None out.report(event="ready", micro=bool(getattr(agent.context_compressor, "_micro_compact_enabled", False)), - resumed=len(history or [])) + resumed=len(history or []), wal=bool(getattr(db, "_wal_active", False))) bases: list[str] = [] try: for i in range(int(a["turns"])): @@ -297,8 +351,20 @@ def role_agent(a: dict, out: Out) -> int: return 0 +def role_cli(a: dict, out: Out) -> int: + """``hermes ``: ``hermes_cli.main`` run as ``__main__``, i.e. ``python -m hermes_cli.main ``.""" + import runpy + + sys.argv = ["hermes", *a["argv"]] + try: + runpy.run_module("hermes_cli.main", run_name="__main__", alter_sys=True) + except SystemExit as exc: + return exc.code if isinstance(exc.code, int) else (0 if exc.code is None else 1) + return 0 + + ROLES = { - "agent": role_agent, + "agent": role_agent, "cli": role_cli, "writer": role_writer, "reader": role_reader, "churn": role_churn, "opener": role_opener, "fts": role_fts, "repair": role_repair, } @@ -306,7 +372,9 @@ ROLES = { def main() -> int: role, args = sys.argv[1], json.loads(sys.argv[2]) - signal.signal(signal.SIGTERM, _on_sigterm) + _apply_sqlite_version_pin() + if role != "cli": # the CLI keeps its own SIGTERM handling + signal.signal(signal.SIGTERM, _on_sigterm) out = Out(Path(args["workdir"]), args["name"]) return ROLES[role](args, out) diff --git a/tests/e2e/core/sqlite/test_compaction_contention.py b/tests/e2e/core/sqlite/test_compaction_contention.py index 09fd847ffb..571f2c9543 100644 --- a/tests/e2e/core/sqlite/test_compaction_contention.py +++ b/tests/e2e/core/sqlite/test_compaction_contention.py @@ -2,7 +2,9 @@ compaction"). Real ``AIAgent`` processes drive real turns (user -> terminal tool call -> answer) through the loopback fake -provider on ONE shared WAL ``state.db``, while other processes write to, read and open/close the same file: +provider on ONE shared ``state.db`` — once per journal mode Hermes deploys (WAL, and the DELETE mode the +production ``apply_wal_with_fallback`` picks on a WAL-reset-vulnerable SQLite, see ``_helpers``) — while other +processes write to, read and open/close the same file: * a gateway-like agent with rolling micro-compaction on (every turn folds the oldest exchange into a summary and commits it through ``archive_and_compact``); @@ -18,7 +20,8 @@ turns. Invariants after every episode: as compacted history (``compacted=1``) — and never live twice; * the exchanges ``/compress here N`` promised to keep are live exactly once; * canonical row counts only grow (compaction archives, it never deletes) — sampled continuously from outside - and by the reader; ``integrity_check`` ok; FTS mirrors canonical rows; no ``(deleted)`` WAL held; + and by the reader; ``integrity_check`` ok; FTS mirrors canonical rows; the store stays in the arm's journal + mode; no ``(deleted)`` store (nor, in WAL mode, ``-wal``/``-shm``) held; * the resumed process sends the model every acknowledged user turn exactly once (persisted == sent); * the plain writer's acked appends are stored exactly once. """ @@ -37,15 +40,19 @@ from dataclasses import dataclass import pytest from tests.e2e.core.sqlite._helpers import ( + JOURNAL_MODES, SEED_ENV, Chamber, base_seed, + busy_summary, compress_journal, counts, episode_seed, exactly_once_problems, fts_problems, integrity_rows, + journal_mode_problems, + skip_unless_deployable, token_flags, ) from tests.fakes.fake_llm_provider import FakeLLMServer, Text, ToolCall, write_hermes_home @@ -110,15 +117,13 @@ class Rig: tui_env: dict -@pytest.fixture(scope="module") -def rig(tmp_path_factory): - v = sqlite3.sqlite_version_info - if not (v >= (3, 51, 3) or v in ((3, 50, 7), (3, 44, 6))): - pytest.skip(f"linked SQLite {sqlite3.sqlite_version} runs Hermes in DELETE mode") +@pytest.fixture(scope="module", params=JOURNAL_MODES) +def rig(request, tmp_path_factory): + skip_unless_deployable(request.param) aux = _Aux() server = FakeLLMServer(_responder, aux=aux) server.start() - ch = Chamber(tmp_path_factory.mktemp("compaction")) + ch = Chamber(tmp_path_factory.mktemp(f"compaction-{request.param}"), journal=request.param) ctx = " context_length: 128000\n" write_hermes_home(ch.hermes_home, server.base_url, extra_config=MICRO_CONFIG + DB_CONFIG) tui_home = ch.root / "tui-home" @@ -216,7 +221,7 @@ def _compacted_rows(ch: Chamber, sid: str) -> int: def _reap_ok(ch: Chamber, name: str, deadline: float = 120.0) -> None: rc = ch.reap(name, deadline=deadline) - errors = [e.get("error") for e in ch.events(name) if e.get("event") == "error"] + errors = [f"{e.get('error')}\n{e.get('tb', '')}" for e in ch.events(name) if e.get("event") == "error"] assert rc == 0, f"{name} exited {rc}: {errors}\n{ch.stderr(name)}" @@ -225,7 +230,7 @@ def test_compaction_episode(rig, fault): ch = rig.ch seed = episode_seed(f"compaction:{fault}") rng = random.Random(seed) - ctx = f"[fault={fault} seed={seed} base={base_seed()}; replay: {SEED_ENV}={base_seed()}]" + ctx = f"[journal={ch.mode} fault={fault} seed={seed} base={base_seed()}; replay: {SEED_ENV}={base_seed()}]" gw_sid, tui_sid = f"{fault}-gw", f"{fault}-tui" gw, tui, plain, reader = f"{fault}-agent-gw", f"{fault}-agent-tui", f"{fault}-plain", f"{fault}-reader" sampler = _GrowOnly(ch) @@ -273,6 +278,7 @@ def test_compaction_episode(rig, fault): rows = integrity_rows(ch.db) if rows != ["ok"]: problems.append(f"integrity_check: {rows[:5]}") + problems += journal_mode_problems(ch, prefix=fault) problems += sampler.violations problems += _turn_problems(ch, gw, compacting=True) problems += _turn_problems(ch, resumed, compacting=True) @@ -285,4 +291,5 @@ def test_compaction_episode(rig, fault): for sid in (gw_sid, tui_sid): # vacuity guard: a compaction commit really landed in this episode if not _compacted_rows(ch, sid): problems.append(f"no compacted history rows in {sid}: compaction never committed") - assert not problems, f"{ctx} ({sampler.samples} row-count samples)\n" + "\n".join(problems) + busy = busy_summary(ch, fault) + assert not problems, f"{ctx} ({sampler.samples} row-count samples, busy waits {busy})\n" + "\n".join(problems) diff --git a/tests/e2e/core/sqlite/test_torture_chamber.py b/tests/e2e/core/sqlite/test_torture_chamber.py index f27d92fbc3..7eece48851 100644 --- a/tests/e2e/core/sqlite/test_torture_chamber.py +++ b/tests/e2e/core/sqlite/test_torture_chamber.py @@ -1,6 +1,9 @@ """SQLite torture chamber: state.db integrity under real multi-process load (issue class C1). -One WAL ``state.db``; every role is its own OS process running the production ``SessionDB``: +One ``state.db`` per journal mode Hermes deploys — WAL, and DELETE (what it runs on a WAL-reset-vulnerable +SQLite and on network/FUSE homes; selected by the production ``apply_wal_with_fallback`` through a pinned +version probe in every child, see ``_helpers``); every role is its own OS process running the production +``SessionDB``: a gateway-like writer, a TUI-like writer, a dashboard-like reader that opens at startup and lives across episodes, short-lived openers (production ``SessionDB`` and bare ``sqlite3``, plus the real ``hermes sessions list`` / ``sessions stats`` CLI), FTS rebuild/optimize maintenance, and @@ -9,14 +12,18 @@ close, POSIX lock cancellation by a stray in-process open/close, chmod flips, co repair against a live and an offline store, FTS corruption, a whole-fleet SIGKILL — and then asserts the SAME invariants: -* ``PRAGMA integrity_check`` is ``ok`` and the store is still in WAL mode; -* no child ever held a ``(deleted)`` ``state.db``/``-wal``/``-shm`` descriptor (``/proc//fd`` scan); +* ``PRAGMA integrity_check`` is ``ok``; the store is still in the arm's journal mode and every SessionDB + role really ran in it; +* no child ever held a ``(deleted)`` ``state.db`` descriptor — nor, in WAL mode, ``-wal``/``-shm`` + (``/proc//fd`` scan); * every acknowledged append is stored exactly once (per-writer intent/ack journals), an in-flight append at most once, and no row exists that no writer intended; * canonical row counts only grow (no compaction runs here), and repair never lowers them; * FTS mirrors the canonical rows (docsize == source rows, FTS5 ``integrity-check``) and session search finds acked messages exactly once; -* no role hit an error; the long-lived reader's and the open/close churner's fd counts stay bounded. +* no role hit an error (DELETE arm: readers block on writes by design, so a SQLITE_BUSY refusal of a read, + open or FTS pass is waited out and counted, never an integrity failure); the long-lived reader's and the + open/close churner's fd counts stay bounded. Randomness (ack thresholds, kill points) is seeded per episode; the seed is in every failure message and ``HERMES_SQLITE_TORTURE_SEED`` replays a run. @@ -34,17 +41,20 @@ import pytest from tests.conformance.persistence._harness import wait_for from tests.e2e.core.sqlite._helpers import ( + JOURNAL_MODES, SEED_ENV, Chamber, acked_tokens, base_seed, + busy_summary, counts, episode_seed, exactly_once_problems, fts_problems, integrity_rows, - journal_mode, + journal_mode_problems, sample, + skip_unless_deployable, ) pytestmark = [ @@ -55,18 +65,10 @@ READER_FD_SLACK = 6 CHURN_FD_SLACK = 2 -def _sqlite_wal_capable() -> bool: - # Mirror of the requires_wal gate: SQLite 3.7.0-3.51.2 (minus backports) has the WAL-reset bug, and - # Hermes deliberately falls back to DELETE there, so a WAL chamber is not deployable on that runtime. - v = sqlite3.sqlite_version_info - return v >= (3, 51, 3) or v in ((3, 50, 7), (3, 44, 6)) - - -@pytest.fixture(scope="module") -def chamber(tmp_path_factory): - if not _sqlite_wal_capable(): - pytest.skip(f"linked SQLite {sqlite3.sqlite_version} runs Hermes in DELETE mode; chamber needs WAL") - ch = Chamber(tmp_path_factory.mktemp("chamber")) +@pytest.fixture(scope="module", params=JOURNAL_MODES) +def chamber(request, tmp_path_factory): + skip_unless_deployable(request.param) + ch = Chamber(tmp_path_factory.mktemp(f"chamber-{request.param}"), journal=request.param) _ensure_reader(ch) yield ch ch.shutdown() @@ -112,7 +114,7 @@ def _reap_all(ch: Chamber, names: list[str]) -> None: proc = ch.procs[name] if proc.returncode is None: rc = ch.reap(name) - errors = [e.get("error") for e in ch.events(name) if e.get("event") == "error"] + errors = [f"{e.get('error')}\n{e.get('tb', '')}" for e in ch.events(name) if e.get("event") == "error"] assert rc == 0, f"{name} exited {rc}: {errors}\n{ch.stderr(name)}" @@ -200,7 +202,7 @@ def ep_chmod_flip(ch, ep, rng): gw, tui = _writers(ch, ep) files = [ch.db, ch.db.with_name(ch.db.name + "-wal"), ch.db.with_name(ch.db.name + "-shm")] try: - for i in range(rng.randint(6, 10)): + for i in range(rng.randint(3, 5)): # each flip is the same fault; process start-up dominates for f in files: if f.exists(): os.chmod(f, 0o644) # a permissive mode the next SessionDB open tightens @@ -338,7 +340,8 @@ def _churn_fd_problems(ch: Chamber, ep: str) -> list[str]: def test_torture_episode(chamber, episode): seed = episode_seed(episode) rng = random.Random(seed) - ctx = f"[episode={episode} seed={seed} base={base_seed()}; replay: {SEED_ENV}={base_seed()}]" + ctx = (f"[journal={chamber.mode} episode={episode} seed={seed} base={base_seed()}; " + f"replay: {SEED_ENV}={base_seed()}]") _ensure_reader(chamber) before = counts(chamber.db) if chamber.db.exists() else {"__total__": 0} started = time.monotonic() @@ -354,9 +357,7 @@ def test_torture_episode(chamber, episode): rows = integrity_rows(chamber.db) if rows != ["ok"]: problems.append(f"integrity_check: {rows[:5]}") - mode = journal_mode(chamber.db) - if mode != "wal": - problems.append(f"journal_mode is {mode!r} after the episode (was wal)") + problems += journal_mode_problems(chamber) problems += exactly_once_problems(chamber) after = counts(chamber.db) for sid, n in before.items(): @@ -368,7 +369,8 @@ def test_torture_episode(chamber, episode): problems += fts_problems(chamber.db, sample(rng, episode_acks, 12)) problems += _reader_fd_problems(chamber) problems += _churn_fd_problems(chamber, episode) - assert not problems, f"{ctx} ({time.monotonic() - started:.1f}s)\n" + "\n".join(problems) + busy = busy_summary(chamber, episode) + assert not problems, f"{ctx} ({time.monotonic() - started:.1f}s, busy waits {busy})\n" + "\n".join(problems) def test_long_lived_reader_saw_every_episode_grow_only(chamber): From 23366b44f3e97b25bb705d642d2cdc1cda440731 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 08:49:01 -0700 Subject: [PATCH 023/104] test(e2e): close the core-suite review findings (narrow xfails, surfaced handler errors, no-retry CI) Independent review of #120171 found checks that could not fail. Each is now proven red by a mutation that the old version reported as XFAIL or pass. - chaos/test_tui_gateway_turn_liveness: the orphaned-tool xfail used raises=AssertionError and RpcError subclasses it, so a gateway crash counted as the expected failure. Every invariant is now asserted normally; only the known leftovers (surviving tool tree and the tool_call it leaves without a result, both fixed by #120306) raise ToolOutlivedGateway, the only exception the xfail accepts. The DB check used to sit behind the orphan assert and never ran; running it exposed the dangling tool_call half of the same bug. - history/test_prefix_stability: surface_switch's strict xfail tripped at the first prefix break, before usage and integrity. Messages/system prompt, usage and integrity are asserted first; the tools-array drift is checked last and raises ToolsArrayDrift, the only exception the xfail accepts. - history/test_transcript_ledger: scripted steer/interrupt callables run on the fake provider's handler thread, where an assert only dropped the connection. Script records those failures and the test re-raises them after every turn; steer must land and the interrupted turn must report interrupted=True within 30 s. - fakes/fake_llm_provider: Hang drops the connection at its deadline instead of leaving a kept-alive client waiting past it. - parity: the API server port was picked, released, then bound by the child. Readiness now requires our child's pid from authenticated /health/detailed and retries on a fresh port when the child reports it in use. The fixture guard refused any HERMES_HOME under ~/.hermes, failing all parity tests whenever TMPDIR is Hermes's scratch dir; it now refuses only the live root or a real profile. - chaos/_gateway_harness: the gateway stays in pytest's process group, so the runner's kill of a timed-out file reaches it. - sqlite: a DELETE-mode open can fail with SQLITE_BUSY reported as "vtable constructor failed: messages_fts"; the delete arm's busy tolerance keys on the result code. A failed episode's roles are stopped so the shared chamber and rig no longer fail every later episode. - chaos, compaction, parity homes: updates.check=false (history already had it). The passive update check made a GitHub round-trip from every test surface, and on a blobless clone whose objects lag upstream its `git merge-base --is-ancestor HEAD` starts a lazy fetch that the 5 s timeout orphans; the orphan scans then failed on git processes. - chaos/test_agent_turn_liveness: a PROBE failure now carries the provider call counts and the agent's stale-kill log, so a cross-turn breaker trip can be told apart from a slow probe. - tests.yml e2e: HERMES_TEST_FILE_RETRIES=0 so a race detector's red is never retried into green; own uv cache entry (cache-suffix: e2e). --- .github/workflows/tests.yml | 7 +++ tests/e2e/core/chaos/_gateway_harness.py | 9 ++- tests/e2e/core/chaos/_helpers.py | 4 ++ .../core/chaos/test_agent_turn_liveness.py | 8 ++- .../chaos/test_tui_gateway_turn_liveness.py | 30 +++++++--- tests/e2e/core/compaction/_helpers.py | 2 + tests/e2e/core/history/_helpers.py | 35 +++++++++-- .../e2e/core/history/test_prefix_stability.py | 17 +++++- .../core/history/test_transcript_ledger.py | 25 ++++++-- tests/e2e/core/parity/_drive_gateway.py | 58 +++++++++++++------ tests/e2e/core/parity/_helpers.py | 11 +++- tests/e2e/core/sqlite/_roles.py | 5 +- .../core/sqlite/test_compaction_contention.py | 6 ++ tests/e2e/core/sqlite/test_torture_chamber.py | 14 +++-- tests/fakes/fake_llm_provider.py | 5 +- 15 files changed, 188 insertions(+), 48 deletions(-) diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index da8b950cce..da242d55ab 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -197,6 +197,9 @@ jobs: # time, but resolves from the warm cache instead of re-downloading # and re-building wheels. enable-cache: true + # Own cache entry: the unit job's older uv would otherwise share and + # overwrite this key with a cache this uv version did not write. + cache-suffix: e2e cache-dependency-glob: | pyproject.toml uv.lock @@ -241,6 +244,10 @@ jobs: # Multi-process episodes (torture chamber, compaction kill -9, # gateway liveness) legitimately run past the 300 s default. HERMES_TEST_FILE_TIMEOUT: "900" + # No automatic re-run of a failed file: the torture chamber and the + # exactly-once/compaction suites are race detectors, and a rare + # corruption that passes on retry is still a corruption. + HERMES_TEST_FILE_RETRIES: "0" OPENROUTER_API_KEY: "" OPENAI_API_KEY: "" NOUS_API_KEY: "" diff --git a/tests/e2e/core/chaos/_gateway_harness.py b/tests/e2e/core/chaos/_gateway_harness.py index a67197b8be..c2bf6525ae 100644 --- a/tests/e2e/core/chaos/_gateway_harness.py +++ b/tests/e2e/core/chaos/_gateway_harness.py @@ -21,7 +21,7 @@ from pathlib import Path from typing import Any, Callable, Optional from tests.e2e.core.chaos import _gateway_fake_platform as fake_platform -from tests.e2e.core.chaos._helpers import hermetic_env, python_exe, write_chaos_home +from tests.e2e.core.chaos._helpers import hermetic_env, kill_tagged, python_exe, write_chaos_home BOOT_DEADLINE_S = 180.0 SHUTDOWN_DEADLINE_S = 60.0 @@ -112,9 +112,11 @@ class GatewayProc: "HERMES_STATE_DB_GUARD_BYPASS": "1", }) log = open(self.log_path, "wb") + # Same process group as pytest (no start_new_session): when the runner kills a timed-out + # file's group, the gateway goes with it instead of outliving the run. self.proc = subprocess.Popen( [python_exe(), "-m", "gateway.run"], cwd=str(self.home), env=env, - stdin=subprocess.DEVNULL, stdout=log, stderr=subprocess.STDOUT, start_new_session=True) + stdin=subprocess.DEVNULL, stdout=log, stderr=subprocess.STDOUT) log.close() deadline = time.monotonic() + BOOT_DEADLINE_S while True: @@ -169,8 +171,9 @@ class GatewayProc: self.shutdown_s = time.monotonic() - t0 except subprocess.TimeoutExpired: self.shutdown_s = None + kill_tagged(self.tag) with _suppress_oserror(): - os.killpg(self.proc.pid, signal.SIGKILL) + self.proc.kill() self.exit_code = self.proc.wait(timeout=30) finally: for sock in (self._conn, self._listener): diff --git a/tests/e2e/core/chaos/_helpers.py b/tests/e2e/core/chaos/_helpers.py index 9ab7f37980..9b10a91b0c 100644 --- a/tests/e2e/core/chaos/_helpers.py +++ b/tests/e2e/core/chaos/_helpers.py @@ -90,6 +90,10 @@ def chaos_config( "memory:\n" " memory_enabled: false\n" " user_profile_enabled: false\n" + # Offline: the passive update check does a GitHub round-trip and, on a partial clone + # whose objects lag upstream, spawns a git lazy fetch that outlives the gateway. + "updates:\n" + " check: false\n" + extra ) diff --git a/tests/e2e/core/chaos/test_agent_turn_liveness.py b/tests/e2e/core/chaos/test_agent_turn_liveness.py index ad03410008..ef722e64c3 100644 --- a/tests/e2e/core/chaos/test_agent_turn_liveness.py +++ b/tests/e2e/core/chaos/test_agent_turn_liveness.py @@ -393,11 +393,15 @@ class Run: log = hermes_home / "logs" / "agent.log" if rep["orphans"] and log.exists(): rep["agent_log_tail"] = log.read_text(errors="replace")[-6000:] + # Stale-kill timeline: tells a PROBE refused by the cross-turn breaker apart from a slow probe. + rep["stale_log"] = [ln[:110] for ln in (log.read_text(errors="replace").splitlines() if log.exists() else []) + if "stale" in ln.lower() and ("WARNING" in ln or "ERROR" in ln)][-12:] db = hermes_home / "state.db" rep["persisted"] = persisted_messages(db, self.session_id) if db.exists() else None rep["integrity"] = integrity_ok(db) if db.exists() else "missing" probes = self.probe_requests() rep["probe_request"] = probes[-1] if probes else None + rep["probe_count"] = len(probes) mains = [r["body"] for r in list(self.srv.requests) if r["kind"] == "main"] rep["last_request"] = mains[-1] if mains else None rep["max_request_bytes"] = max((len(json.dumps(b)) for b in mains), default=0) @@ -471,7 +475,9 @@ def test_agent_turn_liveness(scenario_id: str, runs: dict[str, Future]) -> None: assert rep["probe_request"] is None, "a tripped stale breaker still billed the provider" sent = rep["last_request"]["messages"] else: - assert f"alive {rep_nonce(rep)}" in rep["turn1"]["final"], f"PROBE not answered: {rep['turn1']}" + assert f"alive {rep_nonce(rep)}" in rep["turn1"]["final"], ( + f"PROBE not answered: {rep['turn1']}\nfault calls {rep['fault_calls']}, probe requests " + f"{rep['probe_count']}, stale log:\n" + "\n".join(rep.get("stale_log", []))) assert rep["probe_request"] is not None, "PROBE never reached the provider" sent = rep["probe_request"]["messages"] assert any(f"[[chaos:{scenario_id}]]" in _text(m.get("content")) for m in sent if m.get("role") == "user"), \ diff --git a/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py b/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py index 9a36f6895b..5a1850d76f 100644 --- a/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py +++ b/tests/e2e/core/chaos/test_tui_gateway_turn_liveness.py @@ -288,9 +288,17 @@ def _assert_heartbeat(scn: Scenario, hb: Heartbeat, stats: dict[str, Any], turn_ f"{scn.id}: only {len(hb.latencies)} heartbeats answered in a {turn_s:.1f}s turn") +class ToolOutlivedGateway(Exception): + """The in-flight tool outlived the gateway's exit: its process tree survives and/or its + tool_call was left with no result in state.db. Deliberately NOT an AssertionError: the + known-bug xfail matches only this, so an RPC failure, a crash or any other broken invariant + (heartbeat, exit deadline, integrity) still fails the test.""" + + def _exit_mid_turn(scn: Scenario, gw: TuiGatewayProcess, hb: Heartbeat, tag: str, state_db: Path, stored: str, stats: dict[str, Any]) -> dict[str, Any]: - """The client vanishes (stdin EOF) or the supervisor stops us (SIGTERM) mid-wedge.""" + """The client vanishes (stdin EOF) or the supervisor stops us (SIGTERM) mid-wedge. + Every other invariant is asserted first; the tool leftovers are checked together, last.""" _assert_heartbeat(scn, hb, stats, turn_s=LONG_TIMEOUT_S) t0 = time.monotonic() if scn.action == "sigterm": @@ -303,11 +311,14 @@ def _exit_mid_turn(scn: Scenario, gw: TuiGatewayProcess, hb: Heartbeat, tag: str rc = gw.close_stdin_and_wait(EXIT_TIMEOUT_S) assert rc is not None, f"{scn.id}: gateway still alive {EXIT_TIMEOUT_S}s after {scn.action} mid-turn{gw.tail()}" stats["exit_s"] = round(time.monotonic() - t0, 2) - survivors = wait_no_tagged(tag) - assert survivors == [], f"{scn.id}: orphans after {scn.action} mid-turn: {describe_pids(survivors)}" assert integrity_ok(state_db) == "ok" - assert unanswered_tool_calls(persisted_messages(state_db, stored)) == [], ( - f"{scn.id}: state.db keeps a tool_call with no result after {scn.action} (next resume sends it)") + leftovers = [] + if dangling := unanswered_tool_calls(persisted_messages(state_db, stored)): + leftovers.append(f"state.db keeps tool_call(s) {dangling} with no result (next resume sends them)") + if survivors := wait_no_tagged(tag): + leftovers.append(f"orphans: {describe_pids(survivors)}") + if leftovers: + raise ToolOutlivedGateway(f"{scn.id} after {scn.action} mid-turn: " + "; ".join(leftovers)) return stats @@ -448,10 +459,13 @@ def scenario_futures(request: pytest.FixtureRequest, tmp_path_factory: pytest.Te # Real production bug on base (reported, not fixed here): when the gateway leaves mid-tool — # client closes stdin or supervisor SIGTERMs — _shutdown_sessions() closes the agents but the # in-flight foreground terminal command (its own process group) is never killed, so the -# `bash -c ...` + `sleep 3600` tree survives, reparented to init. strict: flips red once fixed. +# `bash -c ...` + `sleep 3600` tree survives, reparented to init, and its tool_call is left with +# no result in state.db. strict: flips red once fixed. raises= names only the leftovers check's +# exception, so everything before it is asserted normally. _ORPHANED_FOREGROUND_TOOL = pytest.mark.xfail( - strict=True, raises=AssertionError, - reason="tui_gateway exit (EOF/SIGTERM) orphans the running foreground terminal tool's process tree") + strict=True, raises=ToolOutlivedGateway, + reason="tui_gateway exit (EOF/SIGTERM) orphans the running foreground terminal tool's process tree " + "and leaves its tool_call without a result") KNOWN_BUGS = {"stdin_eof_during_hung_tool": _ORPHANED_FOREGROUND_TOOL, "sigterm_during_hung_tool": _ORPHANED_FOREGROUND_TOOL} diff --git a/tests/e2e/core/compaction/_helpers.py b/tests/e2e/core/compaction/_helpers.py index e6b3326eb3..59f2d84fc5 100644 --- a/tests/e2e/core/compaction/_helpers.py +++ b/tests/e2e/core/compaction/_helpers.py @@ -142,6 +142,8 @@ def write_home(hermes_home: Path, base_url: str, extra_config: str) -> None: f" context_length: {CONTEXT_LENGTH}\n" "agent:\n" " api_max_retries: 1\n" + "updates:\n" + " check: false\n" # offline: no GitHub round-trip or git lazy fetch from a test surface + extra_config, encoding="utf-8", ) diff --git a/tests/e2e/core/history/_helpers.py b/tests/e2e/core/history/_helpers.py index 997963f38d..c33bd840dc 100644 --- a/tests/e2e/core/history/_helpers.py +++ b/tests/e2e/core/history/_helpers.py @@ -199,15 +199,16 @@ def _list_diff(a: list[tuple], b: list[tuple], an: str, bn: str) -> str: return "\n".join(lines) -def prefix_breaks(requests: list[dict[str, Any]]) -> list[tuple[int, str]]: +def prefix_breaks(requests: list[dict[str, Any]], *, tools: bool = True) -> list[tuple[int, str]]: """(C17) Indices i where request i is NOT a byte-identical extension of request i-1. - Byte-stability covers the tools array, the system prompt and every earlier message. + Byte-stability covers the tools array (unless ``tools=False``), the system prompt and every + earlier message. """ breaks = [] for i in range(1, len(requests)): prev, cur = requests[i - 1], requests[i] - if canon(prev.get("tools")) != canon(cur.get("tools")): + if tools and canon(prev.get("tools")) != canon(cur.get("tools")): breaks.append((i, "tools array changed: " + tools_diff(prev.get("tools"), cur.get("tools")))) continue pm, cm = prev["messages"], cur["messages"] @@ -222,6 +223,13 @@ def prefix_breaks(requests: list[dict[str, Any]]) -> list[tuple[int, str]]: return breaks +def tools_breaks(requests: list[dict[str, Any]]) -> list[tuple[int, str]]: + """Indices i where request i sends a different tools array than request i-1.""" + return [(i, tools_diff(requests[i - 1].get("tools"), requests[i].get("tools"))) + for i in range(1, len(requests)) + if canon(requests[i - 1].get("tools")) != canon(requests[i].get("tools"))] + + def tools_diff(a: Any, b: Any) -> str: an = {t["function"]["name"]: canon(t) for t in a or ()} bn = {t["function"]["name"]: canon(t) for t in b or ()} @@ -502,19 +510,32 @@ class TuiGateway: class Script: """Main-turn responder: pops scripted actions; an action may be a callable run AT request - arrival (so /steer and interrupt land while the request is genuinely in flight).""" + arrival (so /steer and interrupt land while the request is genuinely in flight). + + Callables run on the fake provider's HTTP handler thread, where a raised assertion only drops + the connection (the agent retries and the test moves on). ``errors`` keeps every such failure + so the test re-raises it on its own thread (``raise_errors``).""" def __init__(self) -> None: self.actions: list[Any] = [] self.n = 0 self.session: Any = None self.steered: list[str] = [] + self.errors: list[str] = [] def __call__(self, record: dict[str, Any]) -> Any: self.n += 1 if self.actions: act = self.actions.pop(0) - return act(record) if callable(act) else act + if not callable(act): + return act + try: + return act(record) + except BaseException: + import traceback + + self.errors.append(traceback.format_exc()) + raise # Unique text + varying usage per answer, so a duplicated row or a double-counted # request is always distinguishable. from tests.fakes.fake_llm_provider import Text @@ -522,6 +543,10 @@ class Script: return Text(f"answer #{self.n}", prompt_tokens=900 + 13 * self.n, completion_tokens=5 + self.n, cached_tokens=400 + self.n) + def raise_errors(self, where: str) -> None: + assert not self.errors, f"{where}: a scripted action failed on the provider thread:\n" + "\n".join( + self.errors) + def big(label: str, n: int = 12000) -> str: """A user message large enough that summarizing a few of them genuinely shrinks the context.""" diff --git a/tests/e2e/core/history/test_prefix_stability.py b/tests/e2e/core/history/test_prefix_stability.py index 9e3476ffa3..9f7b3c1776 100644 --- a/tests/e2e/core/history/test_prefix_stability.py +++ b/tests/e2e/core/history/test_prefix_stability.py @@ -38,6 +38,7 @@ from tests.e2e.core.history._helpers import ( prefix_breaks, row_counts, run_oneshot, + tools_breaks, views, ) from tests.fakes.fake_llm_provider import FakeLLMServer, Text, ToolCall, write_hermes_home @@ -91,6 +92,11 @@ KNOWN_BROKEN = { } +class ToolsArrayDrift(Exception): + """The tools array changed between requests of one session. Not an AssertionError: the + known-bug xfail matches only this, so every other invariant still fails the test.""" + + @pytest.fixture def world(tmp_path): home = tmp_path / "home" @@ -172,7 +178,7 @@ def run_journey(world: dict, hops: list[Hop]) -> tuple[str, list[tuple[int, str] @pytest.mark.parametrize("journey", [ - pytest.param(name, marks=pytest.mark.xfail(strict=True, reason=KNOWN_BROKEN[name])) + pytest.param(name, marks=pytest.mark.xfail(strict=True, raises=ToolsArrayDrift, reason=KNOWN_BROKEN[name])) if name in KNOWN_BROKEN else name for name in JOURNEYS ]) @@ -185,7 +191,9 @@ def test_request_prefix_is_byte_stable_across_processes(world, journey): def where(i: int) -> str: return f"request {i} (in {max((o for o in openings if o[0] <= i), default=(0, '?'))[1]})" - breaks = prefix_breaks(main) + # System prompt + messages first; the tools array is checked last, on its own, so a known + # tools-drift xfail cannot mask a message-prefix, usage or integrity regression. + breaks = prefix_breaks(main, tools=False) unexpected = [(i, why) for i, why in breaks if i != compaction_idx] assert not unexpected, "prompt-cache prefix broke outside the compaction boundary:\n" + "\n".join( f" {where(i)}: {why}" for i, why in unexpected) @@ -195,3 +203,8 @@ def test_request_prefix_is_byte_stable_across_processes(world, journey): assert_usage_matches(home, lineage(home, sid), srv.requests, f"after journey {journey}") integrity_ok(home) + + drift = [(i, why) for i, why in tools_breaks(main) if i != compaction_idx] + if drift: + raise ToolsArrayDrift("tools array changed within one session:\n" + "\n".join( + f" {where(i)}: {why}" for i, why in drift)) diff --git a/tests/e2e/core/history/test_transcript_ledger.py b/tests/e2e/core/history/test_transcript_ledger.py index c4d3797aa2..ca3ada2d5a 100644 --- a/tests/e2e/core/history/test_transcript_ledger.py +++ b/tests/e2e/core/history/test_transcript_ledger.py @@ -20,6 +20,7 @@ prefix at the declared compaction boundaries (C17). from __future__ import annotations import os +import time from pathlib import Path from typing import Any, Callable @@ -61,6 +62,11 @@ from tests.fakes.fake_llm_provider import ( Step = tuple[str, Any] +STEER_TEXT = "also mention the steer marker" +INTERRUPTED_TURN = "this request gets interrupted" +# The interrupted request hangs HANG_S and then drops; a working interrupt ends the turn long before. +INTERRUPT_HANG_S = 60.0 +INTERRUPT_DEADLINE_S = 30.0 def bulky(*labels: str) -> list[Step]: @@ -87,7 +93,7 @@ def _interrupt_during_request(script: Script) -> Callable[[dict], Any]: import threading threading.Thread(target=script.session.agent.interrupt, daemon=True).start() - return Hang(seconds=60) + return Hang(seconds=INTERRUPT_HANG_S) return act @@ -110,13 +116,13 @@ def scenario(name: str, script: Script) -> list[Step]: if name == "steer": return [("turn", "warm up"), ("script", [_steer_then(ToolCall("terminal", {"command": "echo steered-tool"}), - "also mention the steer marker", s), Text("steer seen")]), + STEER_TEXT, s), Text("steer seen")]), ("turn", "do a tool while I steer"), ("turn", "after the steer")] if name == "interrupt": return [("turn", "warm up"), ("script", [_interrupt_during_request(s)]), - ("turn", "this request gets interrupted"), + ("turn", INTERRUPTED_TURN), ("turn", "the follow-up after the interrupt"), ("turn", "one more")] if name == "stream_faults": @@ -206,9 +212,17 @@ def test_transcript_ledger(world, name): script.actions.extend(arg) elif kind == "turn": ledger.inputs.append(arg) + t0 = time.monotonic() result = session.turn(arg) + took = time.monotonic() - t0 + script.raise_errors(label) ledger.inputs += [x for x in script.steered if x not in ledger.inputs] - assert result.get("final_response") is not None or name == "interrupt", f"{label}: {result}" + if arg == INTERRUPTED_TURN: + assert result.get("interrupted") is True and took < INTERRUPT_DEADLINE_S, ( + f"{label}: interrupt did not end the hung request (interrupted=" + f"{result.get('interrupted')!r} after {took:.1f}s)") + else: + assert result.get("final_response") is not None, f"{label}: {result}" compaction = name == "micro_compaction" or ( name == "auto_compaction" and arg == "the turn that must compact first") ledger.step(session.sid, label, compaction=compaction) @@ -220,6 +234,9 @@ def test_transcript_ledger(world, name): f"{res.before_tokens}->{res.after_tokens} tokens") ledger.step(session.sid, label, compaction=True, turn=False) sid = session.sid + if name == "steer": + # The ledger's inputs check then requires the steer row shown exactly once. + assert script.steered == [STEER_TEXT], f"the steer never landed: {script.steered}" finally: session.close() assert not script.actions, f"scripted responses never consumed: {script.actions}" diff --git a/tests/e2e/core/parity/_drive_gateway.py b/tests/e2e/core/parity/_drive_gateway.py index 3f2862e78c..8b93568f58 100644 --- a/tests/e2e/core/parity/_drive_gateway.py +++ b/tests/e2e/core/parity/_drive_gateway.py @@ -204,30 +204,54 @@ def _http(method: str, url: str, *, key: str | None = None, body: dict | None = return exc.code, exc.read().decode("utf-8", "replace") +PORT_ATTEMPTS = 3 + + +def _await_own_api_server(proc: subprocess.Popen, base: str, key: str) -> bool: + """True once OUR child answers on ``base`` (authenticated ``/health/detailed`` reporting its + pid, so a stranger that grabbed the port is never mistaken for it); False when the child + reports the port taken. The port is picked free and then released before the child binds + it, so another process can win that race.""" + deadline = time.monotonic() + TURN_TIMEOUT + while True: + if "already in use" in _log_tail(proc, 20000): + return False + if proc.poll() is not None: + raise AssertionError(f"api server exited {proc.returncode} before ready\n{_log_tail(proc)}") + if time.monotonic() >= deadline: + raise AssertionError(f"api server never became ready on {base}\n{_log_tail(proc)}") + with contextlib.suppress(OSError, urllib.error.URLError, ValueError): + status, raw = _http("GET", f"{base}/health/detailed", key=key, timeout=2.0) + if status == 200 and json.loads(raw).get("pid") == proc.pid: + return True + time.sleep(0.2) + + def drive_api_server(ph: ParityHome, srv: FakeLLMServer, prompt: str) -> DriveResult: ph.pin_terminal_cwd() - port = _free_loopback_port() key = secrets.token_hex(32) - _append_env(ph, {"API_SERVER_ENABLED": "true", "API_SERVER_KEY": key, - "API_SERVER_HOST": "127.0.0.1", "API_SERVER_PORT": str(port)}) - proc = _spawn(ph, [sys.executable, "-m", "gateway.run"], "api_server.stderr.log") - base = f"http://127.0.0.1:{port}" + env_before = (ph.hermes_home / ".env").read_text(encoding="utf-8") if (ph.hermes_home / ".env").exists() else "" + for _attempt in range(PORT_ATTEMPTS): + port = _free_loopback_port() + (ph.hermes_home / ".env").write_text(env_before, encoding="utf-8") + _append_env(ph, {"API_SERVER_ENABLED": "true", "API_SERVER_KEY": key, + "API_SERVER_HOST": "127.0.0.1", "API_SERVER_PORT": str(port)}) + proc = _spawn(ph, [sys.executable, "-m", "gateway.run"], "api_server.stderr.log") + base = f"http://127.0.0.1:{port}" + try: + ready = _await_own_api_server(proc, base, key) + except BaseException: + _stop(ph, proc) + raise + if ready: + break + _stop(ph, proc) + else: + raise AssertionError(f"api server lost the port race {PORT_ATTEMPTS} times\n{_log_tail(proc)}") status: int | None = None payload: dict[str, Any] = {} graceful = False try: - deadline = time.monotonic() + TURN_TIMEOUT - while True: - if proc.poll() is not None: - raise AssertionError(f"api server exited {proc.returncode} before ready\n{_log_tail(proc)}") - if time.monotonic() >= deadline: - raise AssertionError(f"api server never became ready on {base}\n{_log_tail(proc)}") - try: - if _http("GET", f"{base}/health", timeout=2.0)[0] == 200: - break - except (OSError, urllib.error.URLError): - pass - time.sleep(0.2) status, raw = _http( "POST", f"{base}/v1/chat/completions", key=key, timeout=TURN_TIMEOUT, body={"model": "hermes-agent", "messages": [{"role": "user", "content": prompt}], diff --git a/tests/e2e/core/parity/_helpers.py b/tests/e2e/core/parity/_helpers.py index fb0c6dbc2b..db04772e2a 100644 --- a/tests/e2e/core/parity/_helpers.py +++ b/tests/e2e/core/parity/_helpers.py @@ -80,9 +80,15 @@ class ParityHome: """Hermetic env for a subprocess Hermes: fake HOME, no real credentials.""" import pwd # POSIX-only; the suite is Linux-gated + # Refuse only a home the real install would read as live state (its root or a profile). + # A tmp_path under ``~/.hermes/cache/scratch`` (TMPDIR when Hermes itself runs the suite) + # is fine: the child's HOME is the fixture home, so its ``~/.hermes`` never resolves there. real_root = Path(pwd.getpwuid(os.getuid()).pw_dir, ".hermes").resolve() - assert real_root not in (self.hermes_home.resolve(), *self.hermes_home.resolve().parents), ( - f"fixture HERMES_HOME {self.hermes_home} is inside the real {real_root}") + fixture = self.hermes_home.resolve() + assert fixture != real_root and fixture.parent != real_root / "profiles", ( + f"fixture HERMES_HOME {self.hermes_home} is the real install's live home") + assert fixture == (self.home / ".hermes").resolve(), ( + f"fixture HERMES_HOME {self.hermes_home} is not /.hermes") # Allowlist, not denylist: the runner may itself be a Hermes process whose # TERMINAL_CWD / HERMES_* / credential env would silently reroute the child. env = { @@ -168,6 +174,7 @@ def build_parity_home(root: Path, base_url: str, *, grandchild: bool = True, cfg["mcp_single_query_discovery_timeout"] = 120 # Keep turns hermetic and short: no title/aux model chatter decides anything here. cfg.setdefault("display", {})["compact"] = True + cfg["updates"] = {"check": False} # offline: no GitHub round-trip or git lazy fetch (hermes_home / "config.yaml").write_text(yaml.safe_dump(cfg, sort_keys=False), encoding="utf-8") hooks_dir = hermes_home / "agent-hooks" diff --git a/tests/e2e/core/sqlite/_roles.py b/tests/e2e/core/sqlite/_roles.py index 7221bf2349..84e25fbf18 100644 --- a/tests/e2e/core/sqlite/_roles.py +++ b/tests/e2e/core/sqlite/_roles.py @@ -80,7 +80,10 @@ def _patient(a: dict, out: Out, op: str, fn, *, deadline: float = 90.0): try: return fn() except sqlite3.OperationalError as exc: - busy = any(m in str(exc).lower() for m in ("database is locked", "database is busy")) + # By result code, not text: SQLITE_BUSY also surfaces as "vtable constructor failed: + # messages_fts" when the FTS5 table's config read hits the lock during an open. + busy = getattr(exc, "sqlite_errorcode", None) in (sqlite3.SQLITE_BUSY, sqlite3.SQLITE_LOCKED) or any( + m in str(exc).lower() for m in ("database is locked", "database is busy")) if not (busy and a.get("busy_ok")) or time.monotonic() > end: raise out.report(event="busy", op=op, error=repr(exc)) diff --git a/tests/e2e/core/sqlite/test_compaction_contention.py b/tests/e2e/core/sqlite/test_compaction_contention.py index 571f2c9543..4b98ce9862 100644 --- a/tests/e2e/core/sqlite/test_compaction_contention.py +++ b/tests/e2e/core/sqlite/test_compaction_contention.py @@ -28,6 +28,7 @@ turns. Invariants after every episode: from __future__ import annotations +import contextlib import json import random import re @@ -271,6 +272,11 @@ def test_compaction_episode(rig, fault): finally: sampler.stop.set() sampler.join(timeout=10) + # The rig is shared by every fault: a failed episode's live roles must not leak into the next. + for name, _proc in ch.live(): + if name.startswith(fault): + with contextlib.suppress(AssertionError): + ch.stop(name, deadline=30.0) problems: list[str] = [] problems += [f"{n}: {e.get('error')}\n{e.get('tb', '')}" for n, e in ch.errors() if n.startswith(fault)] diff --git a/tests/e2e/core/sqlite/test_torture_chamber.py b/tests/e2e/core/sqlite/test_torture_chamber.py index 7eece48851..7ad91a6bc9 100644 --- a/tests/e2e/core/sqlite/test_torture_chamber.py +++ b/tests/e2e/core/sqlite/test_torture_chamber.py @@ -31,6 +31,7 @@ Randomness (ack thresholds, kill points) is seeded per episode; the seed is in e from __future__ import annotations +import contextlib import os import random import sqlite3 @@ -346,10 +347,15 @@ def test_torture_episode(chamber, episode): before = counts(chamber.db) if chamber.db.exists() else {"__total__": 0} started = time.monotonic() - runs = EPISODES[episode](chamber, episode, rng) - - # Nothing but the long-lived reader may still be running on the file. - stragglers = [n for n, _p in chamber.live() if n != chamber.reader_name] + try: + runs = EPISODES[episode](chamber, episode, rng) + finally: + # Nothing but the long-lived reader may still be running on the file. Stop the rest either + # way: the chamber is shared, so a failed episode's writers must not fail every later one. + stragglers = [n for n, _p in chamber.live() if n != chamber.reader_name] + for name in stragglers: + with contextlib.suppress(AssertionError): + chamber.stop(name, deadline=30.0) assert not stragglers, f"{ctx} roles still running: {stragglers}" problems: list[str] = [] problems += [f"{n}: {e.get('error')}\n{e.get('tb', '')}" for n, e in chamber.errors()] diff --git a/tests/fakes/fake_llm_provider.py b/tests/fakes/fake_llm_provider.py index 50e5ad3deb..6c0065aa62 100644 --- a/tests/fakes/fake_llm_provider.py +++ b/tests/fakes/fake_llm_provider.py @@ -78,7 +78,7 @@ class Error: @dataclass class Hang: - """Accept the request and never answer within ``seconds``.""" + """Accept the request and never answer; the connection is dropped after ``seconds``.""" seconds: float = 3600.0 @@ -272,6 +272,9 @@ def _handler_for(server: FakeLLMServer) -> type[BaseHTTPRequestHandler]: return if isinstance(resp, Hang): server._stop.wait(resp.seconds) + # Drop the socket at the deadline: on a kept-alive HTTP/1.1 connection the client + # would otherwise wait for a response that never comes, far past ``seconds``. + self.close_connection = True return if isinstance(resp, Raw): body = resp.body.encode() From 2b1bb70ac9b14482a996ddf19eaf82b88a179697 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:52:31 +0000 Subject: [PATCH 024/104] test(e2e): micro-compaction display check is a plain test now that #120316 landed Drop the strict xfail on the resumed-display test and re-enable the input check for the micro_compaction ledger scenario. The ledger no-loss oracle now exempts model-only rows (the merged user turn #120316 flags model_only): they stand in for originals that stay displayed and were each tracked when sent. A/B: removing the model_only flag in agent/micro_compaction.py turns both micro cases red; with it, the history suites are green. --- tests/e2e/core/history/_helpers.py | 21 ++++++++++++++++++- .../core/history/test_transcript_ledger.py | 9 +------- 2 files changed, 21 insertions(+), 9 deletions(-) diff --git a/tests/e2e/core/history/_helpers.py b/tests/e2e/core/history/_helpers.py index c33bd840dc..b81d0ed471 100644 --- a/tests/e2e/core/history/_helpers.py +++ b/tests/e2e/core/history/_helpers.py @@ -115,6 +115,18 @@ def views(hermes_home: Path, sid: str) -> tuple[list[dict], list[dict]]: db.close() +def model_only_identities(hermes_home: Path, model: list[dict]) -> set[tuple]: + """Identities of model-view rows stored as model-only (never part of the display projection).""" + ids = [m["_row_id"] for m in model if m.get("_row_id") is not None] + if not ids: + return set() + with db_connect(hermes_home) as con: + hidden = {r[0] for r in con.execute( + f"SELECT id FROM messages WHERE id IN ({','.join('?' * len(ids))}) " + "AND COALESCE(json_extract(display_metadata, '$.model_only'), 0) != 0", ids)} + return {identity(m) for m in model if m.get("_row_id") in hidden} + + def row_counts(hermes_home: Path, sid: str) -> dict[str, int]: with db_connect(hermes_home) as con: total, active = con.execute( @@ -616,6 +628,7 @@ class Ledger: self.srv, self.home, self.check_inputs = srv, hermes_home, check_inputs self.ever: dict[tuple, str] = {} self.inputs: list[str] = [] + self.model_only: set[tuple] = set() self.model_view: list[dict] | None = None self.counts: dict[str, int] | None = None self.compaction_request_idx: list[int] = [] @@ -656,7 +669,13 @@ class Ledger: self.ever.setdefault(identity(m), short(identity(m))) assert_exactly_once(model, f"{where} (model view)") assert_rows_exactly_once(self.home, sid, where) - assert_no_loss(self.ever, display, model, where) + # A model-only row (micro-compaction's merged user turn) stands in for rows that ARE displayed, + # so it is exempt from the display no-loss check; its originals were each sent (and tracked) earlier. + self.model_only |= model_only_identities(self.home, model) + merged = self.model_only + for k in merged: + self.ever.pop(k, None) + assert_no_loss(self.ever, display, [m for m in model if identity(m) not in merged], where) if self.check_inputs: assert_inputs_shown_once(self.inputs, display, model, where) counts = row_counts(self.home, sid) diff --git a/tests/e2e/core/history/test_transcript_ledger.py b/tests/e2e/core/history/test_transcript_ledger.py index ca3ada2d5a..a831140fe2 100644 --- a/tests/e2e/core/history/test_transcript_ledger.py +++ b/tests/e2e/core/history/test_transcript_ledger.py @@ -202,9 +202,7 @@ def test_transcript_ledger(world, name): assert not [(a, b) for a in typed for b in typed if a != b and a in b], "scenario inputs must be unique" session = InProcessSession(srv.base_url, hermes_home, f"ledger-{name}") script.session = session - # Micro-compaction's merged-user-row display duplication is tracked separately (strict xfail - # below) so the scenario still guards every other invariant. - ledger = Ledger(srv, hermes_home, check_inputs=name != "micro_compaction") + ledger = Ledger(srv, hermes_home) try: for i, (kind, arg) in enumerate(scenario(name, script)): label = f"{i}:{kind}:{str(arg)[:24]}" @@ -283,11 +281,6 @@ def test_transcript_ledger(world, name): assert_usage_matches(hermes_home, lineage(hermes_home, sid), srv.requests, f"scenario {name} after resume") -@pytest.mark.xfail(strict=True, reason=( - "PRODUCTION BUG (opt-in micro-compaction): when a newer micro marker supersedes the old one, " - "_merge_adjacent_user_turns persists a merged 'A\\n\\nB' user row while the originals stay " - "compacted=1, so the resumed display history shows both user inputs twice. Flip to a plain test " - "once the display projection (or the merge) stops duplicating them.")) def test_micro_compaction_resumed_display_shows_each_input_once(world): srv, script, hermes_home = world["srv"], world["script"], world["hermes_home"] write_hermes_home(hermes_home, srv.base_url, From 360b9697ac9742a19123b658127f6a86accd97e7 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 08:00:01 -0700 Subject: [PATCH 025/104] fix(gateway): a crash no longer re-answers every recently active chat On an unclean start, suspend_recently_active(120) marked every session touched in the last 120 s resume_pending/restart_interrupted, so startup auto-resume ran a fresh model turn for chats whose turn had already finished and been delivered: one kill re-answered 52 chats in the C12 delivery suite. The durable active-turn markers already name the exact in-flight turns, so the recency sweep is removed. That sweep also hid a real window: _handle_message cleared the turn marker in its finally BEFORE the adapter recorded the delivery obligation, so a kill in between left neither marker nor ledger row and the persisted reply was never sent. The adapter now owns the marker for turns it delivers and clears it right after record_delivery_obligation (or once nothing more is owed). At unclean startup a marked turn whose final reply is already in the transcript has that reply adopted into the delivery ledger (unowned, 'attempting': sent once, marked as a possible duplicate) instead of being regenerated; a marked turn with no reply resumes once, as before. --- gateway/delivery_ledger.py | 22 ++++++ gateway/platforms/base.py | 13 ++++ gateway/run.py | 5 +- gateway/run_inbound.py | 7 +- gateway/run_shutdown.py | 8 +- gateway/run_startup.py | 74 +++++++++++++++---- gateway/session_lifecycle.py | 20 ----- tests/gateway/test_active_turn_recovery.py | 70 +++++++++++------- tests/gateway/test_clean_shutdown_marker.py | 52 +------------ tests/gateway/test_restart_resume_pending.py | 26 +------ tests/gateway/test_session.py | 2 - .../test_session_store_runtime_stale_guard.py | 1 - tests/gateway/test_startup_restart_race.py | 1 - .../gateway-session-lifecycle.md | 40 ++++++---- website/docs/user-guide/messaging/index.md | 2 + 15 files changed, 181 insertions(+), 162 deletions(-) diff --git a/gateway/delivery_ledger.py b/gateway/delivery_ledger.py index e846402ccd..6b4b45549f 100644 --- a/gateway/delivery_ledger.py +++ b/gateway/delivery_ledger.py @@ -285,6 +285,28 @@ def record_obligation(*, obligation_id: str, session_key: str, platform: str, ch _prune_unlocked(conn, now) +def record_crash_left_reply(*, obligation_id: str, session_key: str, platform: str, chat_id: str, + thread_id: Optional[str], content: str, since: float, + adapter_profile: Optional[str] = None) -> None: + """Adopt a reply a killed process persisted but never ledgered. Unowned, so this boot's sweep + claims it, and 'attempting', because a streamed reply may already be on screen: it is + redelivered once, with the recovered marker. A no-op when the same reply was already ledgered + since *since* (the turn start), and idempotent across boots that die before their sweep.""" + now = time.time() + with _DB_LOCK, _transaction() as conn: + conn.execute( + """INSERT OR IGNORE INTO delivery_obligations + (obligation_id, session_key, platform, chat_id, thread_id, + content, state, attempts, created_at, updated_at, + owner_pid, owner_started_at, adapter_profile) + SELECT ?, ?, ?, ?, ?, ?, 'attempting', 0, ?, ?, NULL, NULL, ? + WHERE NOT EXISTS (SELECT 1 FROM delivery_obligations + WHERE session_key = ? AND content = ? AND created_at >= ?)""", + (obligation_id, session_key, platform, str(chat_id), str(thread_id) if thread_id else None, + content, now, now, str(adapter_profile).strip() if adapter_profile else "default", + session_key, content, since)) + + def mark_attempting(obligation_id: str) -> None: _update_state(obligation_id, "attempting") diff --git a/gateway/platforms/base.py b/gateway/platforms/base.py index 4eb1ab4c08..d938cdc793 100644 --- a/gateway/platforms/base.py +++ b/gateway/platforms/base.py @@ -4267,12 +4267,21 @@ class BasePlatformAdapter(ABC): len(text_content), event.source.chat_id) obligation_id = await self._record_delivery_obligation( event, session_key, text_content, delivery_adapter, is_ephemeral_response) + if obligation_id is not None: + await self._release_turn_marker(event) # the ledger now owns the crash recovery result = await delivery_adapter._send_with_retry( chat_id=event.source.chat_id, content=text_content, reply_to=reply_to, metadata=metadata) if obligation_id is not None: await self._finalize_delivery_obligation(obligation_id, result, event, delivery_adapter) return result, delivery_adapter + async def _release_turn_marker(self, event: MessageEvent) -> None: + """Clear the crash-recovery marker the runner handed to this delivery lifecycle + (``_turn_marker_handoff``): only once the final reply is ledgered or nothing more is owed, + so no kill leaves a persisted reply with neither marker nor ledger row. Idempotent.""" + if getattr(event, "_turn_marker_handoff", False) and getattr(event, "_gateway_active_turn_token", None): + await self.gateway_runner._clear_durable_active_turn(event) + async def _send_final_text( self, event: MessageEvent, session_key: str, text_content: str, metadata: Dict[str, Any], is_ephemeral_response: bool, ephemeral_ttl: int, record_delivery: Callable) -> None: @@ -4442,6 +4451,7 @@ class BasePlatformAdapter(ABC): typing_task = self._start_typing_refresh(event, interrupt_event, _thread_metadata) try: await self._run_processing_hook("on_processing_start", event) + event._turn_marker_handoff = self.gateway_runner is not None # it can release the marker response = await self._message_handler(event) # A muted diagnostic wake ran for the session; its reply is not presented. The # policy read binds the routed profile; delivery itself stays in the launch scope. @@ -4501,6 +4511,7 @@ class BasePlatformAdapter(ABC): event, extracted, _final_thread_metadata, anything_sent=delivery_attempted or _tts_caption_delivered, record_delivery=_record_delivery) + await self._release_turn_marker(event) processing_ok = delivery_succeeded if delivery_attempted else not bool(response) # Clean up the per-turn streaming-TTS flag. self._streaming_tts_completed_turns.discard(self._streaming_tts_turn_key( @@ -4533,6 +4544,8 @@ class BasePlatformAdapter(ABC): if isinstance(e, (SystemExit, KeyboardInterrupt)): raise finally: + await self._release_turn_marker(event) + event._turn_marker_handoff = False # a later run of this object clears its own marker # Stop typing BEFORE the post-delivery callback: a stuck callback must not keep it # alive. await self._stop_typing_refresh(event.source.chat_id, typing_task, metadata=_thread_metadata) diff --git a/gateway/run.py b/gateway/run.py index 3f334263cc..3f028576ee 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -4047,8 +4047,9 @@ class GatewayRunner( _STUCK_LOOP_THRESHOLD = 3 # restarts while active before auto-suspend _STUCK_LOOP_FILE = ".restart_failure_counts" - # Reasons set by _stop_impl() on force-interrupt; "restart_interrupted" by suspend_recently_active() - # on crash recovery (no .clean_shutdown marker). All mean "killed mid-turn" -> startup auto-resume. + # Reasons set by _stop_impl() on force-interrupt; "restart_interrupted" by recover_interrupted_turns() + # for a crash-left turn marker (no .clean_shutdown marker). All mean "killed mid-turn" -> startup + # auto-resume. _AUTO_RESUME_REASONS = frozenset({"restart_timeout", "shutdown_timeout", "restart_interrupted"}) _MAX_SUPERVISED_RESTARTS = 5 diff --git a/gateway/run_inbound.py b/gateway/run_inbound.py index e893dab272..48d793d7a7 100644 --- a/gateway/run_inbound.py +++ b/gateway/run_inbound.py @@ -1375,8 +1375,11 @@ class GatewayInboundMixin: # exception, interrupt); the generation guard makes a displaced turn's finalizer a no-op. self._restore_pending_one_turn_model_override(_quick_key, _run_generation) # SIGKILL/OOM skips finally, leaving the durable marker for the next unclean startup's - # recovery pass. - await self._clear_durable_active_turn(event) + # recovery pass. A turn the adapter delivers hands its marker to that lifecycle, which + # clears it only once the reply is in the delivery ledger (else a kill in between + # left neither marker nor ledger row and the persisted reply was never sent). + if not getattr(event, "_turn_marker_handoff", False): + await self._clear_durable_active_turn(event) # Release only this turn's generation. Eviction may immediately admit a replacement # through the cold path; an unconditional release here would then clear the replacement # sentinel/agent and lease. Reset/stop release their stale slot before installing a diff --git a/gateway/run_shutdown.py b/gateway/run_shutdown.py index b9f7ea8667..d555a04616 100644 --- a/gateway/run_shutdown.py +++ b/gateway/run_shutdown.py @@ -1329,7 +1329,7 @@ class GatewayShutdownMixin: atomic_json_write(path, {key: counts.get(key, 0) + 1 for key in active_session_keys}, indent=None) def _suspend_stuck_loop_sessions(self) -> int: - """Suspend sessions active across too many restarts (startup, AFTER suspend_recently_active()).""" + """Suspend sessions active across too many restarts (startup, AFTER crash-turn recovery).""" path = self._stuck_loop_counts_path() if not path.exists(): return 0 @@ -2043,15 +2043,15 @@ class GatewayShutdownMixin: from gateway.status import remove_pid_file, release_gateway_runtime_lock remove_pid_file() release_gateway_runtime_lock() - # Clean-shutdown marker skips suspend_recently_active() next boot; a timed-out drain left - # half-finished sessions, so no marker — the next startup suspends them. + # Clean-shutdown marker skips crash-turn recovery next boot; a timed-out drain left + # half-finished sessions, so no marker — the next startup recovers their turn markers. if not ctx.timed_out: with suppress(Exception): (_hermes_home / ".clean_shutdown").touch() else: logger.info( "Skipping .clean_shutdown marker — drain timed out with " - "interrupted agents; next startup will suspend recently active sessions." + "interrupted agents; next startup will recover their interrupted turns." ) # Stuck-loop counter: sessions active across 3 consecutive restarts are auto-suspended next boot. if ctx.active_agents: diff --git a/gateway/run_startup.py b/gateway/run_startup.py index a4e1448afc..e7f46ac769 100644 --- a/gateway/run_startup.py +++ b/gateway/run_startup.py @@ -708,18 +708,60 @@ class GatewayStartupMixin: return discarded async def _recover_unclean_sessions(self) -> tuple[int, int]: - """Recover exact active turns, then run the legacy recency fallback.""" + """Recover only the turns the dead process left marked: one whose reply is already in the + transcript is owed delivery, not a new answer; any other resumes once. An unmarked session + finished its turn (the marker is held until the reply is ledgered), so nothing re-runs it — + the old 120 s recency sweep re-answered every recently active chat. Returns (resumed, + ledgered).""" from gateway.run import _float_env - exact = 0 - fallback = 0 + resumed = ledgered = 0 + max_age = max(60 * 60, int(max(1.0, _float_env("HERMES_AGENT_TIMEOUT", 1800)) * 2)) + with _log_suppressed(logging.WARNING, "Crash-left reply recovery on startup failed: %s"): + ledgered = await self._ledger_crash_left_replies(max_age) with _log_suppressed(logging.WARNING, "Exact active-turn recovery on startup failed: %s"): - agent_timeout = max(1.0, _float_env("HERMES_AGENT_TIMEOUT", 1800)) - exact = await self.async_session_store.recover_interrupted_turns( - max_age_seconds=max(60 * 60, int(agent_timeout * 2)) - ) - with _log_suppressed(logging.WARNING, "Legacy session recovery on startup failed: %s"): - fallback = await self.async_session_store.suspend_recently_active(max_age_seconds=120) - return exact, fallback + resumed = await self.async_session_store.recover_interrupted_turns(max_age_seconds=max_age) + return resumed, ledgered + + async def _ledger_crash_left_replies(self, max_age_seconds: int) -> int: + """Hand every marked turn whose final reply was persisted to the delivery ledger and clear + its marker, so the boot sweep delivers the stored reply instead of auto-resume + regenerating it. Without the ledger such turns stay marked and resume.""" + from gateway.delivery_ledger import compute_obligation_id, ledger_enabled, record_crash_left_reply + from gateway.platforms.base import _strip_media_directives + from gateway.run import _sanitize_gateway_final_response + if not await asyncio.to_thread(ledger_enabled): + return 0 + cutoff = time.time() - max_age_seconds # older markers are cleared, never acted on + with self.session_store._lock: # noqa: SLF001 — snapshot under lock + self.session_store._ensure_loaded_locked() # noqa: SLF001 + marked = [ + (e.session_key, e.session_id, e.active_turn_token, e.active_turn_started_at, e.origin, + e.transport_profile) + for e in self.session_store._entries.values() # noqa: SLF001 + if e.active_turn_token and e.active_turn_started_at and e.origin and not e.suspended + ] + ledgered = 0 + for key, session_id, token, started_at, origin, profile in marked: + history = await self.async_session_store.load_transcript(session_id) + last = next((m for m in reversed(history) if m.get("role") not in ("session_meta", "system")), None) + if (started_at.timestamp() < cutoff or not last or last.get("role") != "assistant" + or last.get("tool_calls") + or not isinstance(last.get("content"), str) + or float(last.get("timestamp") or 0) < started_at.timestamp()): + continue # the turn never produced its final reply: it resumes + text = _strip_media_directives( + _sanitize_gateway_final_response(origin.platform, last["content"])).strip() + if not text: + continue + await asyncio.to_thread( + record_crash_left_reply, + obligation_id=compute_obligation_id(key, f"crash:{token}", text), session_key=key, + platform=str(getattr(origin.platform, "value", origin.platform)), chat_id=origin.chat_id, + thread_id=origin.thread_id, content=text, since=started_at.timestamp(), + adapter_profile=profile) + if await self.async_session_store.clear_turn_active(key, token): + ledgered += 1 + return ledgered @staticmethod def _start_hosted_room_worker_sync(): @@ -1041,8 +1083,8 @@ class GatewayStartupMixin: recovered += self._recover_secondary_process_checkpoints(process_registry) if recovered: logger.info("Recovered %s background process(es) from previous run", recovered) - # Recover sessions active at last exit (exact turn markers + 120s recency fallback for - # marker-less older turns). SKIP after a clean exit — the previous process already drained. + # Recover the turns the last process left marked (in flight, or reply not yet ledgered). + # SKIP after a clean exit — the previous process already drained. _clean_marker = _hermes_home / ".clean_shutdown" if _clean_marker.exists(): logger.info("Previous gateway exited cleanly — skipping session suspension") @@ -1057,11 +1099,11 @@ class GatewayStartupMixin: if discarded: logger.info("Discarded %d orphan active-turn marker(s) after clean shutdown", discarded) else: - exact, fallback = await self._recover_unclean_sessions() - if exact + fallback: + resumed, ledgered = await self._recover_unclean_sessions() + if resumed + ledgered: logger.info( - "Marked %d in-flight session(s) as resumable from previous run " - "(%d exact, %d legacy)", exact + fallback, exact, fallback, + "Recovered %d interrupted turn(s) from previous run (%d to resume, %d reply(ies) " + "owed delivery)", resumed + ledgered, resumed, ledgered, ) # Stuck-loop detection: a session active across 3+ consecutive restarts is auto-suspended. with _log_suppressed(logging.DEBUG, "Stuck-loop detection failed: %s"): diff --git a/gateway/session_lifecycle.py b/gateway/session_lifecycle.py index 47b3a95489..e1815afde7 100644 --- a/gateway/session_lifecycle.py +++ b/gateway/session_lifecycle.py @@ -241,23 +241,3 @@ class SessionLifecycleMixin: logger.info("SessionStore pruned %d entries older than %d days", len(removed_keys), max_age_days) return len(removed_keys) - - def suspend_recently_active(self, max_age_seconds: int = 120) -> int: - """Mark sessions active within *max_age_seconds* as ``resume_pending`` after a crash/fast - restart (already-pending and suspended entries are skipped). Returns the number marked. - - Called on gateway startup after a crash or fast restart to preserve in-flight sessions instead of - destroying their conversation history (#7536). Only marks sessions updated within *max_age_seconds* - to avoid touching long-idle sessions. Sets ``resume_pending=True`` so the next incoming message on - the same session_key auto-resumes from the existing transcript. - """ - cutoff = _now() - timedelta(seconds=max_age_seconds) - - def _mark(entry: SessionEntry) -> bool: - if entry.resume_pending or entry.suspended or entry.updated_at < cutoff: - return False - entry.resume_pending = True - entry.resume_reason = "restart_interrupted" - entry.last_resume_marked_at = _now() - return True - return self._update_all_entries_locked(_mark) diff --git a/tests/gateway/test_active_turn_recovery.py b/tests/gateway/test_active_turn_recovery.py index 5fa76c112a..3c2ecd1bdb 100644 --- a/tests/gateway/test_active_turn_recovery.py +++ b/tests/gateway/test_active_turn_recovery.py @@ -458,35 +458,53 @@ async def test_runner_active_turn_clear_stops_after_bounded_retries(): assert not hasattr(event, "_gateway_active_turn_token") -@pytest.mark.asyncio -async def test_unclean_recovery_promotes_exact_markers_before_legacy_fallback( - monkeypatch, -): +def _db_runner(tmp_path) -> tuple[GatewayRunner, SessionStore]: runner = object.__new__(GatewayRunner) - calls: list[str] = [] + runner.session_store = _make_db_store(tmp_path) + return runner, runner.session_store - monkeypatch.delenv("HERMES_AGENT_TIMEOUT", raising=False) - async def _recover(*, max_age_seconds): - assert max_age_seconds == ACTIVE_TURN_MAX_AGE_SECONDS - calls.append("exact") - return 1 +def _turn(store: SessionStore, chat_id: str, *, marked: bool, reply: str | None) -> SessionSource: + source = _make_source(chat_id) + entry = store.get_or_create_session(source) + if marked: + store.mark_turn_active(entry.session_key) + store.append_to_transcript(entry.session_id, {"role": "user", "content": f"question {chat_id}"}) + if reply is not None: + store.append_to_transcript(entry.session_id, {"role": "assistant", "content": reply}) + return source - async def _fallback(*, max_age_seconds): - assert max_age_seconds == 120 - calls.append("fallback") - return 2 - runner.session_store = MagicMock() - setattr( - runner, - "_async_session_store", - SimpleNamespace( - _store=runner.session_store, - recover_interrupted_turns=_recover, - suspend_recently_active=_fallback, - ), - ) +@pytest.mark.asyncio +async def test_unclean_restart_resumes_only_the_turn_left_in_flight(tmp_path): + """A kill re-arms the marked turn that had no reply yet, never a chat whose turn finished just + before it (the removed 120 s recency sweep re-answered every recently active chat).""" + runner, store = _db_runner(tmp_path) + finished = _turn(store, "finished", marked=False, reply="answered and delivered") + in_flight = _turn(store, "in-flight", marked=True, reply=None) - assert await runner._recover_unclean_sessions() == (1, 2) - assert calls == ["exact", "fallback"] + assert await runner._recover_unclean_sessions() == (1, 0) + + assert not _entry_for(store, finished).resume_pending + resumed = _entry_for(store, in_flight) + assert (resumed.resume_pending, resumed.resume_reason) == (True, "restart_interrupted") + _close_store_db(store) + + +@pytest.mark.asyncio +async def test_unclean_restart_delivers_a_persisted_unledgered_reply_instead_of_regenerating(tmp_path): + """Killed after the reply was persisted but before it reached the delivery ledger: the stored + reply is ledgered for this boot's sweep (marked, it may already be on screen), not resumed.""" + from gateway.delivery_ledger import sweep_recoverable + + runner, store = _db_runner(tmp_path) + source = _turn(store, "replied", marked=True, reply="the stored answer") + + assert await runner._recover_unclean_sessions() == (0, 1) + + entry = _entry_for(store, source) + assert (entry.resume_pending, entry.active_turn_token) == (False, None) + rows = sweep_recoverable(deliverable_platforms={"discord"}) + assert [(r["content"], r["needs_marker"], r["chat_id"], r["thread_id"]) for r in rows] == [ + ("the stored answer", True, "replied", "thread-1")] + _close_store_db(store) diff --git a/tests/gateway/test_clean_shutdown_marker.py b/tests/gateway/test_clean_shutdown_marker.py index 25c79f674b..e9dca67cb4 100644 --- a/tests/gateway/test_clean_shutdown_marker.py +++ b/tests/gateway/test_clean_shutdown_marker.py @@ -2,9 +2,7 @@ When the gateway shuts down gracefully (hermes update, gateway restart, /restart), it writes a .clean_shutdown marker. On the next startup, if the marker exists, -suspend_recently_active() is skipped so users don't lose their sessions. - -After a crash (no marker), suspension still fires as a safety net for stuck sessions. +crash-turn recovery is skipped and orphan turn markers are discarded. """ from datetime import datetime, timedelta @@ -28,28 +26,6 @@ def _make_store(tmp_path): return SessionStore(sessions_dir=tmp_path, config=config) -# --------------------------------------------------------------------------- -# SessionStore.suspend_recently_active -# --------------------------------------------------------------------------- - -class TestSuspendRecentlyActive: - """Verify suspend_recently_active only marks recent sessions.""" - - def test_suspends_recently_active_sessions(self, tmp_path): - store = _make_store(tmp_path) - source = _make_source() - entry = store.get_or_create_session(source) - assert not entry.suspended - - count = store.suspend_recently_active() - assert count == 1 - - # Re-fetch — should be resume_pending (preserved, not wiped) - refreshed = store.get_or_create_session(source) - assert refreshed.resume_pending - assert refreshed.session_id == entry.session_id # same session preserved - - # --------------------------------------------------------------------------- # Clean shutdown marker integration # --------------------------------------------------------------------------- @@ -100,32 +76,6 @@ class TestCleanShutdownMarker: assert marker.exists(), ".clean_shutdown marker should exist after graceful stop" - def test_no_marker_triggers_suspension(self, tmp_path, monkeypatch): - """Without .clean_shutdown marker (crash), suspension should fire.""" - monkeypatch.setattr("gateway.run._hermes_home", tmp_path) - - marker = tmp_path / ".clean_shutdown" - assert not marker.exists() - - # Create a store with a recently active session - store = _make_store(tmp_path) - source = _make_source() - entry = store.get_or_create_session(source) - assert not entry.suspended - - # Simulate what start() does: - if marker.exists(): - marker.unlink() - else: - store.suspend_recently_active() - - # Session SHOULD be resume_pending (crash recovery preserves history) - with store._lock: - store._ensure_loaded_locked() - resume_count = sum(1 for e in store._entries.values() if e.resume_pending) - assert resume_count == 1, "Session should be resume_pending after crash (no marker)" - - # --------------------------------------------------------------------------- # resume_pending freshness gate (#46934) # --------------------------------------------------------------------------- diff --git a/tests/gateway/test_restart_resume_pending.py b/tests/gateway/test_restart_resume_pending.py index 1646395e7e..9ac0142975 100644 --- a/tests/gateway/test_restart_resume_pending.py +++ b/tests/gateway/test_restart_resume_pending.py @@ -7,9 +7,8 @@ PRs #9850, #9934, #7536): 1. When a gateway restart drain times out and agents are force-interrupted, the affected sessions are flagged ``resume_pending=True`` — not ``suspended`` — so the next user message on the same session_key - auto-resumes from the existing transcript instead of getting routed - through ``suspend_recently_active()`` and converted into a fresh - session. + auto-resumes from the existing transcript instead of being converted + into a fresh session. 2. ``suspended=True`` (from ``/stop`` or stuck-loop escalation) still wins over ``resume_pending`` — the forced-wipe path is preserved. @@ -246,25 +245,6 @@ class TestGetOrCreateResumePending: mock_tip.assert_called_with(original_sid) -# --------------------------------------------------------------------------- -# SessionStore.suspend_recently_active skip behaviour -# --------------------------------------------------------------------------- - - -class TestSuspendRecentlyActiveSkipsResumePending: - def test_resume_pending_entries_not_suspended(self, tmp_path): - store = _make_store(tmp_path) - source = _make_source() - entry = store.get_or_create_session(source) - store.mark_resume_pending(entry.session_key) - - count = store.suspend_recently_active() - assert count == 0 - e = store._entries[entry.session_key] - assert e.suspended is False - assert e.resume_pending is True - - # --------------------------------------------------------------------------- # Restart-resume system-note injection # --------------------------------------------------------------------------- @@ -552,7 +532,7 @@ class TestFreshnessHelpers: async def test_drain_timeout_marks_resume_pending(): """End-to-end: a drain timeout during gateway stop should flag every active session as resume_pending BEFORE the interrupt fires, so the - next startup's suspend_recently_active() does not destroy them.""" + next startup auto-resumes them.""" runner, adapter = make_restart_runner() adapter.disconnect = AsyncMock() runner._restart_drain_timeout = 0.05 diff --git a/tests/gateway/test_session.py b/tests/gateway/test_session.py index d30f62c4d7..48764de403 100644 --- a/tests/gateway/test_session.py +++ b/tests/gateway/test_session.py @@ -1136,8 +1136,6 @@ class TestSessionMetadata: assert store.set_session_metadata(entry.session_key, "k", "v") assert entry.updated_at == idle - # And the restart freshness gate must still see it as idle. - assert store.suspend_recently_active(max_age_seconds=120) == 0 class TestRewriteTranscriptPreservesReasoning: diff --git a/tests/gateway/test_session_store_runtime_stale_guard.py b/tests/gateway/test_session_store_runtime_stale_guard.py index 782db55e0c..8d12e4a72f 100644 --- a/tests/gateway/test_session_store_runtime_stale_guard.py +++ b/tests/gateway/test_session_store_runtime_stale_guard.py @@ -221,4 +221,3 @@ class TestAdvanceCompressionSession: assert result is not None assert result.updated_at == idle - assert store.suspend_recently_active(max_age_seconds=120) == 0 diff --git a/tests/gateway/test_startup_restart_race.py b/tests/gateway/test_startup_restart_race.py index 7157363431..356e862573 100644 --- a/tests/gateway/test_startup_restart_race.py +++ b/tests/gateway/test_startup_restart_race.py @@ -88,7 +88,6 @@ def make_startup_runner(tmp_path): runner.hooks.discover_and_load = MagicMock() runner.hooks.emit = AsyncMock() runner.session_store = MagicMock() - runner.session_store.suspend_recently_active.return_value = 0 runner.delivery_router = MagicMock() runner.delivery_router.adapters = {} diff --git a/website/docs/developer-guide/gateway-session-lifecycle.md b/website/docs/developer-guide/gateway-session-lifecycle.md index 2df83f49cc..d92802bc52 100644 --- a/website/docs/developer-guide/gateway-session-lifecycle.md +++ b/website/docs/developer-guide/gateway-session-lifecycle.md @@ -98,7 +98,7 @@ behavior on the next access. | `is_fresh_reset` | `bool` | `False` | Set by explicit `/new` or `/reset`. Triggers topic/channel skill re-injection on first message. Distinguished from `was_auto_reset` to avoid misleading "session expired" notices. | | `expiry_finalized` | `bool` | `False` | Historical finalization fence retained for recovery; no timer writes it. | | `suspended` | `bool` | `False` | Hard force-wipe signal. Set by `/stop` or stuck-loop escalation (3+ consecutive restart failures). On next `get_or_create_session()`, forces a new `session_id` regardless of `resume_pending`. | -| `resume_pending` | `bool` | `False` | Soft recovery marker. Set by `suspend_recently_active()` (crash recovery) or drain timeout. On next access, preserves the existing `session_id` — the user continues on the same transcript. Cleared after the next successful turn completes. | +| `resume_pending` | `bool` | `False` | Soft recovery marker. Set by `recover_interrupted_turns()` (crash recovery of a marked, unreplied turn) or drain timeout. On next access, preserves the existing `session_id` — the user continues on the same transcript. Cleared after the next successful turn completes. | | `resume_reason` | `Optional[str]` | `None` | Why resume was marked: `"restart_timeout"`, `"shutdown_timeout"`, `"restart_interrupted"`. | | `last_resume_marked_at` | `Optional[datetime]` | `None` | Timestamp of the last resume-pending marking. | @@ -174,7 +174,7 @@ SessionStore(sessions_dir: Path, config: GatewayConfig, has_active_processes_fn= | `suspend_session(session_key)` | Mark session as `suspended=True` (from `/stop`). Forces auto-reset on next access. | | `mark_resume_pending(session_key, reason)` | Mark session as `resume_pending=True` (from drain timeout). Preserves session_id on next access. Will NOT override `suspended=True`. | | `clear_resume_pending(session_key)` | Clear `resume_pending` after a successful resumed turn. Called from gateway after `run_conversation()` returns. | -| `suspend_recently_active(max_age_seconds=120)` | Crash recovery: mark recently-active sessions as `resume_pending=True`. Skips already-pending and already-suspended entries. Called on startup after unclean shutdown. | +| `recover_interrupted_turns(max_age_seconds)` | Crash recovery: promote durable active-turn markers the dead process left behind to `resume_pending=True` (`restart_interrupted`). Sessions without a marker finished their turn and are left alone. Called on startup after unclean shutdown. | | `prune_old_entries(max_age_days)` | Drop entries older than `max_age_days` (based on `updated_at`). Skips `suspended` entries and sessions with active processes. | | `list_sessions(active_minutes=None)` | Return all sessions, optionally filtered by recent activity. Sorted by `updated_at` descending. | | `lookup_by_session_id(session_id)` | Find the active `SessionEntry` for a persisted session ID. | @@ -324,8 +324,10 @@ Gateway starts │ Missing ▼ ┌───────────────────────────────┐ -│ session_store │── Marks sessions updated within -│ .suspend_recently_active() │ last 120 seconds as resume_pending +│ _recover_unclean_sessions() │── Marked turn with a persisted reply +│ │ → delivery ledger (sent, marked); +│ │ marked turn without one → +│ │ resume_pending (once) └───────────────────────────────┘ │ ▼ @@ -351,15 +353,25 @@ Gateway starts └───────────────────────────────┘ ``` -### suspend_recently_active(max_age_seconds=120) +### Crash recovery (`_recover_unclean_sessions`) -Called on gateway startup when no `.clean_shutdown` marker exists (indicating a crash or -unexpected exit). For each session updated within the last 120 seconds: +Called on gateway startup when no `.clean_shutdown` marker exists (a crash or unexpected +exit). It acts only on durable active-turn markers, never on recency: a chat that was merely +active shortly before the crash finished its turn and is not answered again. -- Sets `resume_pending=True`, `resume_reason="restart_interrupted"`, - `last_resume_marked_at=now`. -- Skips entries already `resume_pending=True` (no double-mark). -- Skips entries explicitly `suspended=True` (hard wipe should stay). +The marker is set when a turn starts and is held until the final reply is in the delivery +ledger (the adapter releases it right after `record_delivery_obligation`), or until nothing +more is owed (streamed reply, suppressed or empty response). So a marker left at startup means +one of two things: + +- **The reply was persisted but never ledgered.** The stored transcript reply is recorded as + an unowned ledger row and the marker is cleared; the boot sweep delivers it once with the + "Recovered reply" notice. The turn is not regenerated. +- **No reply was persisted.** `recover_interrupted_turns()` sets `resume_pending=True`, + `resume_reason="restart_interrupted"`, and the turn auto-resumes once. + +A turn already in the ledger is redelivered by the ledger sweep, which also clears any +`resume_pending` for that session, so it is never both delivered and re-answered. ### Stuck-Loop Detection (`_suspend_stuck_loop_sessions`) @@ -374,7 +386,7 @@ session that was mid-turn when the drain timeout fired. Reasons: - `"restart_timeout"` — killed during restart drain - `"shutdown_timeout"` — killed during shutdown drain -- `"restart_interrupted"` — crash recovery (from `suspend_recently_active`) +- `"restart_interrupted"` — crash recovery of a marked, unreplied turn (from `recover_interrupted_turns`) All three reasons are in `_AUTO_RESUME_REASONS` and eligible for startup auto-resume. @@ -395,8 +407,8 @@ When `get_or_create_session()` encounters `resume_pending=True`: Written at the end of a graceful shutdown. On next startup: -- If present: skip `suspend_recently_active()` entirely. Active agents were already - drained, so no sessions are stuck. +- If present: skip crash recovery entirely and discard orphan turn markers. Active agents + were already drained, so no sessions are stuck. - Then delete the marker. This prevents unwanted auto-resets after `hermes update`, `hermes gateway restart`, diff --git a/website/docs/user-guide/messaging/index.md b/website/docs/user-guide/messaging/index.md index 9650818e8c..1ad2373964 100644 --- a/website/docs/user-guide/messaging/index.md +++ b/website/docs/user-guide/messaging/index.md @@ -862,6 +862,8 @@ Set `typing_indicator: false` on any platform where the indicator is unwanted. S When the gateway shuts down with an in-flight tool call or generation, the affected sessions are flagged as `restart_interrupted`. On the next startup, the gateway schedules an auto-resume for each one — the user gets a short heads-up in the chat ("Send any message after restart and I'll try to resume where you left off.") and the session picks up from the last committed turn when they reply. +Only turns that were actually in flight are resumed, and each resumes once. A chat whose turn had already finished is never answered again just because it was active shortly before a crash. If the gateway was killed after the agent finished a reply but before it was sent, the stored reply is delivered (with a "Recovered reply" notice) instead of being regenerated. + This behaviour is on by default and is logged at gateway start: ``` From 1136f135dd640fff082af7859386637d347dbcd2 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:23:28 +0000 Subject: [PATCH 026/104] fix(gateway): crash recovery never redelivers a suppressed reply; tz-safe turn marker Review follow-ups on the crash-left reply adoption: - A persisted reply is now judged the way live delivery would have judged it. A bare silence marker ([SILENT] / SILENT / NO_REPLY ...) on an internal turn, or the reply to a diagnostic wake whose chat policy mutes diagnostics (read in the routed profile's scope, as the adapter does), is owed nothing: the marker is cleared and nothing is sent or resumed. A human turn's bare silence marker becomes the same "returned only a silence marker" notice the live path sends. Before, the raw "NO_REPLY" reached the user as a "Recovered reply". - The active-turn marker start is written as aware UTC and compared as epoch seconds (startup adoption and recover_interrupted_turns). A naive local wall clock read by a process in another zone (DST, container vs unit TZ) was hours off: a fresh in-flight turn was dropped as stale, or a previous turn's reply could be adopted as this one's. updated_at stays naive local for the older binary's recency heuristic; a pre-upgrade naive marker still reads as local time. - Reply timestamps go through coerce_epoch instead of float(), so one odd transcript row cannot abort the whole recovery pass. --- gateway/run_startup.py | 75 ++++++++++++------- gateway/session_lifecycle.py | 30 ++++---- tests/gateway/test_active_turn_recovery.py | 70 ++++++++++++++++- .../gateway-session-lifecycle.md | 11 ++- 4 files changed, 143 insertions(+), 43 deletions(-) diff --git a/gateway/run_startup.py b/gateway/run_startup.py index e7f46ac769..743335d9e6 100644 --- a/gateway/run_startup.py +++ b/gateway/run_startup.py @@ -14,7 +14,7 @@ import logging import os import signal import time -from contextlib import suppress +from contextlib import nullcontext, suppress from contextvars import copy_context from datetime import datetime from pathlib import Path @@ -723,14 +723,12 @@ class GatewayStartupMixin: return resumed, ledgered async def _ledger_crash_left_replies(self, max_age_seconds: int) -> int: - """Hand every marked turn whose final reply was persisted to the delivery ledger and clear - its marker, so the boot sweep delivers the stored reply instead of auto-resume - regenerating it. Without the ledger such turns stay marked and resume.""" + """Settle every marked turn whose final reply was persisted and clear its marker, so + auto-resume does not regenerate it: a reply live delivery would have suppressed is owed + nothing, any other goes to the delivery ledger for the boot sweep. Without the ledger a + presentable reply stays marked and resumes.""" from gateway.delivery_ledger import compute_obligation_id, ledger_enabled, record_crash_left_reply - from gateway.platforms.base import _strip_media_directives - from gateway.run import _sanitize_gateway_final_response - if not await asyncio.to_thread(ledger_enabled): - return 0 + ledger_on = await asyncio.to_thread(ledger_enabled) cutoff = time.time() - max_age_seconds # older markers are cleared, never acted on with self.session_store._lock: # noqa: SLF001 — snapshot under lock self.session_store._ensure_loaded_locked() # noqa: SLF001 @@ -742,27 +740,54 @@ class GatewayStartupMixin: ] ledgered = 0 for key, session_id, token, started_at, origin, profile in marked: - history = await self.async_session_store.load_transcript(session_id) - last = next((m for m in reversed(history) if m.get("role") not in ("session_meta", "system")), None) - if (started_at.timestamp() < cutoff or not last or last.get("role") != "assistant" - or last.get("tool_calls") - or not isinstance(last.get("content"), str) - or float(last.get("timestamp") or 0) < started_at.timestamp()): - continue # the turn never produced its final reply: it resumes - text = _strip_media_directives( - _sanitize_gateway_final_response(origin.platform, last["content"])).strip() - if not text: + started = started_at.timestamp() # aware UTC marker; a pre-upgrade naive one reads as local + if started < cutoff: continue - await asyncio.to_thread( - record_crash_left_reply, - obligation_id=compute_obligation_id(key, f"crash:{token}", text), session_key=key, - platform=str(getattr(origin.platform, "value", origin.platform)), chat_id=origin.chat_id, - thread_id=origin.thread_id, content=text, since=started_at.timestamp(), - adapter_profile=profile) - if await self.async_session_store.clear_turn_active(key, token): + text = self._crash_left_reply(await self.async_session_store.load_transcript(session_id), + started, origin) + if text is None or (text and not ledger_on): + continue # no final reply to deliver: the turn resumes + if text: + await asyncio.to_thread( + record_crash_left_reply, + obligation_id=compute_obligation_id(key, f"crash:{token}", text), session_key=key, + platform=str(getattr(origin.platform, "value", origin.platform)), chat_id=origin.chat_id, + thread_id=origin.thread_id, content=text, since=started, adapter_profile=profile) + if await self.async_session_store.clear_turn_active(key, token) and text: ledgered += 1 return ledgered + def _crash_left_reply(self, history: list, started: float, origin) -> Optional[str]: + """What a crash-left turn owes, judged as live delivery would have: ``None`` when it never + persisted a final reply after *started*; ``""`` when nothing would have been presented (a + silence marker on a machinery turn, a muted diagnostic wake); else the text to send, with a + human turn's bare silence marker replaced by the same notice the live path sends.""" + from gateway.platforms.base import _strip_media_directives + from gateway.response_filters import is_intentional_silence_response, is_machinery_display_kind + from gateway.run import _sanitize_gateway_final_response + from gateway.run_turn import _UNEXPECTED_SILENCE_REPLY + from gateway.warning_notifications import diagnostic_turn_muted + from hermes_cli.timefmt import coerce_epoch + visible = [m for m in history if m.get("role") not in ("session_meta", "system")] + last = visible[-1] if visible else {} + if (last.get("role") != "assistant" or last.get("tool_calls") or not isinstance(last.get("content"), str) + or (coerce_epoch(last.get("timestamp")) or 0) < started): + return None + prompt = next((m for m in reversed(visible) if m.get("role") == "user"), {}) + machinery = is_machinery_display_kind(prompt.get("display_kind")) + if machinery: + try: # the owning profile's display policy, as the adapter reads it at delivery + scope = self._media_delivery_scope_for_source(origin) + except Exception: + logger.debug("Crash-left reply: no routed scope for %s", origin.chat_id, exc_info=True) + scope = nullcontext() + with scope: + if diagnostic_turn_muted(prompt.get("display_metadata"), origin.platform): + return "" + if is_intentional_silence_response(last["content"]): + return "" if machinery else _UNEXPECTED_SILENCE_REPLY + return _strip_media_directives(_sanitize_gateway_final_response(origin.platform, last["content"])).strip() or None + @staticmethod def _start_hosted_room_worker_sync(): """Start the local Group Chat worker without importing the dashboard.""" diff --git a/gateway/session_lifecycle.py b/gateway/session_lifecycle.py index e1815afde7..1c70f89daf 100644 --- a/gateway/session_lifecycle.py +++ b/gateway/session_lifecycle.py @@ -4,8 +4,9 @@ from __future__ import annotations import logging import os +import time import uuid -from datetime import datetime, timedelta +from datetime import datetime, timedelta, timezone from typing import TYPE_CHECKING, Optional from hermes_state_ids import new_session_id @@ -117,15 +118,16 @@ class SessionLifecycleMixin: candidate = entry.to_dict() candidate["active_turn_token"] = token candidate["active_turn_started_at"] = _iso(started_at) - if started_at is not None: + touched = _now() if started_at is not None else None + if touched is not None: # Keeps the legacy 120s startup heuristic working for an older binary during a rolling # downgrade/upgrade window. - candidate["updated_at"] = started_at.isoformat() + candidate["updated_at"] = touched.isoformat() self._save_entry(session_key, entry_data=candidate, lock_held=True) entry.active_turn_token = token entry.active_turn_started_at = started_at - if started_at is not None: - entry.updated_at = started_at + if touched is not None: + entry.updated_at = touched def mark_turn_active(self, session_key: str) -> Optional[str]: """Persist exact ownership of the running agent turn; returns the opaque token for @@ -136,7 +138,9 @@ class SessionLifecycleMixin: entry = self._entry_locked(session_key) if entry is None: return None - self._set_turn_marker_locked(session_key, entry, token, _now()) + # Aware UTC, unlike the local wall clock elsewhere: the next process compares it with + # epoch transcript timestamps and may run in another zone (DST, container vs unit TZ). + self._set_turn_marker_locked(session_key, entry, token, datetime.now(timezone.utc)) return token def clear_turn_active(self, session_key: str, token: str) -> bool: @@ -153,8 +157,7 @@ class SessionLifecycleMixin: """Promote crash-left turn markers into ``resume_pending`` (unclean startup only). Old/invalid markers are cleared without resuming; suspended sessions are never re-armed. Returns the number of newly promoted sessions.""" - now = _now() - max_age = timedelta(seconds=max(0, max_age_seconds)) + now, epoch_now = _now(), time.time() promoted = 0 def _promote(entry: SessionEntry) -> bool: @@ -162,13 +165,10 @@ class SessionLifecycleMixin: if not entry.active_turn_token: return False started_at = entry.active_turn_started_at - try: - marker_is_stale = started_at is None or ( - max_age_seconds > 0 and now - started_at > max_age - ) - except TypeError: - # Mixed aware/naive timestamps: clear rather than risk an unsafe old resume. - marker_is_stale = True + # Epoch arithmetic: a pre-upgrade naive marker reads as local time, an aware one exactly. + marker_is_stale = started_at is None or ( + max_age_seconds > 0 and epoch_now - started_at.timestamp() > max_age_seconds + ) if not marker_is_stale and not entry.suspended: if entry.resume_pending: # A drain-timeout marker is more specific; keep it. diff --git a/tests/gateway/test_active_turn_recovery.py b/tests/gateway/test_active_turn_recovery.py index 3c2ecd1bdb..8c781ec2f9 100644 --- a/tests/gateway/test_active_turn_recovery.py +++ b/tests/gateway/test_active_turn_recovery.py @@ -6,7 +6,10 @@ marker, compare-and-swap cleanup, and promotion into the existing ``resume_pending`` recovery path after an unclean exit. """ +import os +import time from datetime import datetime, timedelta +from pathlib import Path from types import SimpleNamespace from typing import Any, cast from unittest.mock import AsyncMock, MagicMock, PropertyMock, patch @@ -464,12 +467,12 @@ def _db_runner(tmp_path) -> tuple[GatewayRunner, SessionStore]: return runner, runner.session_store -def _turn(store: SessionStore, chat_id: str, *, marked: bool, reply: str | None) -> SessionSource: +def _turn(store: SessionStore, chat_id: str, *, marked: bool, reply: str | None, **prompt: Any) -> SessionSource: source = _make_source(chat_id) entry = store.get_or_create_session(source) if marked: store.mark_turn_active(entry.session_key) - store.append_to_transcript(entry.session_id, {"role": "user", "content": f"question {chat_id}"}) + store.append_to_transcript(entry.session_id, {"role": "user", "content": f"question {chat_id}", **prompt}) if reply is not None: store.append_to_transcript(entry.session_id, {"role": "assistant", "content": reply}) return source @@ -508,3 +511,66 @@ async def test_unclean_restart_delivers_a_persisted_unledgered_reply_instead_of_ assert [(r["content"], r["needs_marker"], r["chat_id"], r["thread_id"]) for r in rows] == [ ("the stored answer", True, "replied", "thread-1")] _close_store_db(store) + + +_WAKE = {"display_kind": "internal_notification"} + + +@pytest.mark.asyncio +@pytest.mark.parametrize(("reply", "prompt", "owed"), [ + ("[SILENT]", _WAKE, []), + ("NO_REPLY", _WAKE, []), + ("disk is 91% full", {**_WAKE, "display_metadata": {"notification_category": "diagnostic"}}, []), + ("NO_REPLY", {}, ["⚠️ The model returned only a silence marker for a message that needed a reply. " + "Try again or rephrase."]), +]) +async def test_unclean_restart_never_redelivers_a_reply_live_delivery_suppressed(tmp_path, reply, prompt, owed): + """A crash-left reply is owed exactly what live delivery would have sent: nothing for a silence + marker on a machinery turn or a muted diagnostic wake (and the finished turn is not resumed), the + unexpected-silence notice for a human turn, never the raw marker.""" + from gateway.delivery_ledger import sweep_recoverable + + (Path(os.environ["HERMES_HOME"]) / "config.yaml").write_text("display: {suppress_warning_notifications: true}\n", encoding="utf-8") + runner, store = _db_runner(tmp_path) + source = _turn(store, "quiet", marked=True, reply=reply, **prompt) + + assert await runner._recover_unclean_sessions() == (0, len(owed)) + + entry = _entry_for(store, source) + assert (entry.resume_pending, entry.active_turn_token) == (False, None) + assert [r["content"] for r in sweep_recoverable(deliverable_platforms={"discord"})] == owed + _close_store_db(store) + + +@pytest.mark.asyncio +@pytest.mark.skipif(not hasattr(time, "tzset"), reason="needs a POSIX process timezone switch") +async def test_turn_marker_start_survives_a_timezone_change_across_the_crash(tmp_path): + """The dead process's local zone is not the new one's (DST, container vs unit TZ): the marked + turn still resumes, and the previous turn's answer persisted a minute before it is not re-sent + (a naive wall-clock marker read in the new zone was 7 h off and dropped the turn as stale).""" + from gateway.delivery_ledger import sweep_recoverable + + runner, store = _db_runner(tmp_path) + source = _make_source("tz") + entry = store.get_or_create_session(source) + store.append_to_transcript(entry.session_id, {"role": "user", "content": "earlier question"}) + store.append_to_transcript(entry.session_id, {"role": "assistant", "content": "earlier answer", + "timestamp": time.time() - 60}) + original_tz = os.environ.get("TZ") + try: + os.environ["TZ"] = "Etc/GMT+7" # UTC-7 when the turn starts ... + time.tzset() + store.mark_turn_active(entry.session_key) + os.environ["TZ"] = "UTC" # ... UTC when the gateway comes back + time.tzset() + assert await runner._recover_unclean_sessions() == (1, 0) + finally: + if original_tz is None: + os.environ.pop("TZ", None) + else: + os.environ["TZ"] = original_tz + time.tzset() + + assert _entry_for(store, source).resume_pending is True + assert sweep_recoverable(deliverable_platforms={"discord"}) == [] + _close_store_db(store) diff --git a/website/docs/developer-guide/gateway-session-lifecycle.md b/website/docs/developer-guide/gateway-session-lifecycle.md index d92802bc52..b9edf3f9c6 100644 --- a/website/docs/developer-guide/gateway-session-lifecycle.md +++ b/website/docs/developer-guide/gateway-session-lifecycle.md @@ -366,10 +366,19 @@ one of two things: - **The reply was persisted but never ledgered.** The stored transcript reply is recorded as an unowned ledger row and the marker is cleared; the boot sweep delivers it once with the - "Recovered reply" notice. The turn is not regenerated. + "Recovered reply" notice. The turn is not regenerated. The reply is judged the way live + delivery would have judged it: a bare silence marker (`[SILENT]`, `NO_REPLY`, ...) on an + internal turn, or the reply to a diagnostic wake the chat's policy mutes, is owed nothing + (the marker is cleared, nothing is sent or resumed). A human turn's bare silence marker + becomes the same "returned only a silence marker" notice the live path sends. - **No reply was persisted.** `recover_interrupted_turns()` sets `resume_pending=True`, `resume_reason="restart_interrupted"`, and the turn auto-resumes once. +The marker's start time is stored as aware UTC and compared as epoch seconds, so a restart +in a different local zone (DST change, container vs. unit `TZ`) neither drops a fresh marker +as stale nor adopts the previous turn's reply as this one's. A marker written by an older +build (naive local time) is read as host-local time. + A turn already in the ledger is redelivered by the ledger sweep, which also clears any `resume_pending` for that session, so it is never both delivered and re-answered. From 58f3bf10c05315fe566bff4c8f9b922bab409b83 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:10:57 +0000 Subject: [PATCH 027/104] fix: startup update check no longer lazy-fetches history in partial clones In a partial (blobless/treeless) clone, the passive update check's `git merge-base --is-ancestor HEAD` asked about an object the clone had never fetched, so git spawned `git fetch` from the promisor remote to download it (~233k objects for a real install, no negotiation). The 5 s probe timeout killed only the merge-base process; the fetch child kept running orphaned and left partial packs behind, on every launch. Every banner git probe and every `bounded_git_probe` (session-start coding context, TUI gateway branch/root probes) now runs with GIT_NO_LAZY_FETCH=1 (NO_LAZY_FETCH_ENV). A missing object makes the probe fail fast; the ancestry check then falls through to the compare API, which is already the path for tips not in local history. `hermes update` keeps lazy fetch: its checkout/merge legitimately needs missing blobs. --- hermes_cli/_subprocess_compat.py | 11 ++- hermes_cli/banner.py | 9 ++- .../test_git_probe_no_lazy_fetch.py | 75 +++++++++++++++++++ website/docs/user-guide/configuration.md | 4 +- 4 files changed, 93 insertions(+), 6 deletions(-) create mode 100644 tests/hermes_cli/test_git_probe_no_lazy_fetch.py diff --git a/hermes_cli/_subprocess_compat.py b/hermes_cli/_subprocess_compat.py index 6195103048..3c050c8e86 100644 --- a/hermes_cli/_subprocess_compat.py +++ b/hermes_cli/_subprocess_compat.py @@ -28,6 +28,7 @@ __all__ = [ "bounded_probe_run", "noninteractive_git_env", "NO_DRIVER_DIFF_FLAGS", + "NO_LAZY_FETCH_ENV", "pid_is_hermes", ] @@ -211,6 +212,14 @@ def windows_detach_popen_kwargs() -> dict: return {"start_new_session": True} +# Read-only probes must never lazy-fetch. In a partial (blobless/treeless) clone a missing object makes +# git spawn ``git fetch`` from the promisor remote, and a probe's timeout kills only its own git: the +# startup update check's ``merge-base --is-ancestor `` started a ~233k-object +# history download on every launch that ran on orphaned, piling up partial packs. With this set the +# probe fails fast on the missing object instead (git >= 2.44; older git ignores the variable). +NO_LAZY_FETCH_ENV = {"GIT_NO_LAZY_FETCH": "1"} + + # GIT_CONFIG_KEY_n/VALUE_n overrides for internal git children: no credential/askpass prompts, no # repo-configured fsmonitor/hooks/pager/editor/external-diff programs. _GIT_CONFIG_INJECT_PREFIXES = ("GIT_CONFIG_KEY_", "GIT_CONFIG_VALUE_") @@ -583,7 +592,7 @@ def bounded_git_probe(argv: Sequence[str], *, timeout: float) -> str: openai/codex#36793). ``process_group`` only changes which group the child belongs to; it does not detach the terminal or alter the fast path. """ - result = bounded_probe_run(argv, timeout=timeout, env=noninteractive_git_env()) + result = bounded_probe_run(argv, timeout=timeout, env={**noninteractive_git_env(), **NO_LAZY_FETCH_ENV}) if result is None or result.returncode != 0: return "" return (result.stdout or "").strip() diff --git a/hermes_cli/banner.py b/hermes_cli/banner.py index 4416a46d7a..b7c74b2746 100644 --- a/hermes_cli/banner.py +++ b/hermes_cli/banner.py @@ -179,15 +179,16 @@ def _git_run(args: list[str], *, cwd: Optional[Path] = None, timeout: int = 5, t git output is UTF-8; on Windows ``text=True`` defaults to the ANSI code page and a byte like the 3rd of 🐛 in a commit subject crashes the stdlib reader thread (#52649), hence the explicit encoding. ``network=True`` (ls-remote/fetch) detaches stdin and disables git/GCM prompts so a - passive update check can never hang on a ``Username for 'https://github.com':`` prompt. + passive update check can never hang on a ``Username for 'https://github.com':`` prompt. No probe + here may lazy-fetch from a partial clone's promisor remote (see ``NO_LAZY_FETCH_ENV``). """ - from hermes_cli._subprocess_compat import noninteractive_git_env, windows_hide_flags + from hermes_cli._subprocess_compat import NO_LAZY_FETCH_ENV, noninteractive_git_env, windows_hide_flags # The banner/update probes run from GUI-hosted backends too (desktop-spawned # ``hermes serve``), where a bare git child flashes a console window. - kwargs: dict = {"creationflags": windows_hide_flags()} + kwargs: dict = {"creationflags": windows_hide_flags(), "env": {**os.environ, **NO_LAZY_FETCH_ENV}} if network: - kwargs.update({"stdin": subprocess.DEVNULL, "env": noninteractive_git_env()}) + kwargs.update({"stdin": subprocess.DEVNULL, "env": {**noninteractive_git_env(), **NO_LAZY_FETCH_ENV}}) try: return subprocess.run( ["git", *args], capture_output=True, timeout=timeout, cwd=str(cwd) if cwd is not None else None, diff --git a/tests/hermes_cli/test_git_probe_no_lazy_fetch.py b/tests/hermes_cli/test_git_probe_no_lazy_fetch.py new file mode 100644 index 0000000000..7443004d67 --- /dev/null +++ b/tests/hermes_cli/test_git_probe_no_lazy_fetch.py @@ -0,0 +1,75 @@ +"""Read-only git probes never lazy-fetch from a partial clone's promisor remote. + +In a blobless clone, asking about an object the clone has not fetched (the fresh upstream tip the +startup update check compares against) makes git download it — for a real install, the whole +commit/tree history — and the probe's timeout kills only its own git, orphaning the fetch. +""" + +import os +import re +import subprocess +from pathlib import Path + +import pytest + +from hermes_cli import banner +from hermes_cli._subprocess_compat import bounded_git_probe + +_ENV = {**os.environ, "GIT_CONFIG_GLOBAL": os.devnull, "GIT_CONFIG_NOSYSTEM": "1"} + + +def _git(*args, cwd=None, env=_ENV): + return subprocess.run(["git", "-c", "user.name=t", "-c", "user.email=t@t", *args], cwd=cwd, env=env, + check=True, capture_output=True, text=True).stdout.strip() + + +def _git_supports_no_lazy_fetch() -> bool: + m = re.search(r"(\d+)\.(\d+)", _git("--version")) + return bool(m) and (int(m[1]), int(m[2])) >= (2, 44) + + +pytestmark = pytest.mark.skipif(not _git_supports_no_lazy_fetch(), reason="GIT_NO_LAZY_FETCH needs git >= 2.44") + + +@pytest.fixture +def partial_clone(tmp_path: Path): + """(clone, local_head, unfetched_upstream_tip) for a blobless clone of a local upstream.""" + seed, up, clone = tmp_path / "seed", tmp_path / "up.git", tmp_path / "clone" + _git("init", "-q", "-b", "main", str(seed)) + for i in range(2): + (seed / f"f{i}.txt").write_text(f"v{i}\n" * 50, encoding="utf-8") + _git("add", "-A", cwd=seed) + _git("commit", "-qm", f"c{i}", cwd=seed) + _git("clone", "-q", "--bare", str(seed), str(up)) + _git("config", "uploadpack.allowFilter", "true", cwd=up) + _git("config", "uploadpack.allowAnySHA1InWant", "true", cwd=up) + _git("clone", "-q", "--filter=blob:none", "--no-checkout", up.as_uri(), str(clone)) + (seed / "new.txt").write_text("upstream moved\n", encoding="utf-8") + _git("add", "-A", cwd=seed) + _git("commit", "-qm", "upstream", cwd=seed) + _git("push", "-q", str(up), "main", cwd=seed) + return clone, _git("rev-parse", "HEAD", cwd=clone), _git("rev-parse", "main", cwd=up) + + +def _fetched(clone: Path, sha: str) -> bool: + env = {**_ENV, "GIT_NO_LAZY_FETCH": "1"} + return subprocess.run(["git", "cat-file", "-e", sha], cwd=clone, env=env, capture_output=True).returncode == 0 + + +def test_update_check_ancestry_probe_never_fetches_from_the_promisor(partial_clone, monkeypatch): + clone, head, upstream_tip = partial_clone + monkeypatch.setattr(banner, "_github_compare_behind", lambda cur, tgt: None) + + assert banner._tips_behind(head, upstream_tip, clone) == banner.UPDATE_AVAILABLE_NO_COUNT + assert not _fetched(clone, upstream_tip), "the update check downloaded upstream history" + # Control: an upstream tip already in local history still reads as up to date. + parent = _git("rev-parse", "HEAD~1", cwd=clone) + assert banner._tips_behind(head, parent, clone) == 0 + + +def test_bounded_git_probe_never_fetches_from_the_promisor(partial_clone): + clone, head, upstream_tip = partial_clone + + assert bounded_git_probe(["git", "-C", str(clone), "log", "-1", "--format=%H", upstream_tip], timeout=10) == "" + assert not _fetched(clone, upstream_tip), "a session-start git probe downloaded upstream history" + assert bounded_git_probe(["git", "-C", str(clone), "log", "-1", "--format=%H", head], timeout=10) == head diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 6ddbc5404d..b87b25d27d 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -170,7 +170,9 @@ Leaving these unset keeps the legacy defaults (`HERMES_API_TIMEOUT=1800`s, `HERM Passive update checks (CLI banner, TUI badge, dashboard, desktop app) ask the GitHub REST API for the tip of `main` and, when it differs from your checkout, the compare endpoint for the exact count and changelog. They never run -`git fetch`, and every install asks at most **once per 24 hours** (a failed check +`git fetch` — in a partial (`--filter=blob:none`) clone they also never +download missing objects from the promisor remote (Git 2.44 or newer) — and +every install asks at most **once per 24 hours** (a failed check retries after an hour). Applying an update (`hermes update`, or the desktop's Update button) always fetches fresh and invalidates the cached answer. Explicit checks — `hermes update --check`, the desktop's "Check for Updates…" menu item, From eca8babe07b91238ee89efe2b9706713106ea1db Mon Sep 17 00:00:00 2001 From: brooklyn! Date: Wed, 23 Sep 2026 10:36:06 -0500 Subject: [PATCH 028/104] fix(desktop): localize bot creation and appearance dialogs Preserve the existing draft, routing, credential-sharing and avatar behavior. Extend the landed Bot Mode catalogs from #96726/#96878 and #113430; addresses the dialog report in #88798 by NealZhouPanda. Broader #90810 and #101305 remain independent. --- apps/desktop/src/i18n/bots-editor.test.tsx | 88 ++++++ .../src/plugins/hermes-bots/avatar-picker.tsx | 16 +- .../src/plugins/hermes-bots/create-dialog.tsx | 86 +++--- .../hermes-bots/edit-profile-dialog.tsx | 12 +- apps/desktop/src/plugins/hermes-bots/i18n.ts | 259 ++++++++++++++++++ 5 files changed, 393 insertions(+), 68 deletions(-) create mode 100644 apps/desktop/src/i18n/bots-editor.test.tsx diff --git a/apps/desktop/src/i18n/bots-editor.test.tsx b/apps/desktop/src/i18n/bots-editor.test.tsx new file mode 100644 index 0000000000..3a4288f2e5 --- /dev/null +++ b/apps/desktop/src/i18n/bots-editor.test.tsx @@ -0,0 +1,88 @@ +import type * as HermesSdk from '@hermes/plugin-sdk' +import { QueryClient, QueryClientProvider } from '@tanstack/react-query' +import { act, cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterEach, beforeAll, expect, it, vi } from 'vitest' + +import { I18nProvider, useI18n } from '@/i18n' +import type { I18nContextValue } from '@/i18n' +import { registerPluginLocales } from '@/i18n/plugin-i18n' +import { $imagenAvailable } from '@/plugins/hermes-bots/avatar-image' +import { CreateAgentDialog } from '@/plugins/hermes-bots/create-dialog' +import { EditProfileDialog } from '@/plugins/hermes-bots/edit-profile-dialog' +import { BOTS_LOCALES } from '@/plugins/hermes-bots/i18n' +import { translateBotsIn } from '@/plugins/hermes-bots/i18n-test-helper' + +const mocks = vi.hoisted(() => ({ + request: vi.fn(async (_method: string) => ({ skills: [], toolsets: [], mcp_servers: [] })), + connections: vi.fn(async () => []) +})) + +vi.mock('@hermes/plugin-sdk', async importOriginal => { + const sdk = await importOriginal() + + return { ...sdk, host: { ...sdk.host, request: mocks.request, connections: mocks.connections } } +}) +let i18n: I18nContextValue +let dispose: (() => void) | undefined + +function Controls() { + i18n = useI18n() + + return null +} + +function mount(children: React.ReactNode) { + dispose = registerPluginLocales('hermes-bots', BOTS_LOCALES) + + return render( + + + + {children} + + + ) +} + +beforeAll(() => { + Element.prototype.scrollIntoView = () => undefined + $imagenAvailable.set(false) +}) +afterEach(() => { + cleanup() + dispose?.() +}) +it('keeps a new bot draft when the language changes and localizes advanced controls', async () => { + mount( undefined} open roster={[{ connectionId: 'local', name: 'default' }]} />) + const zh = translateBotsIn('zh') + expect(screen.getByText(zh('editor.newDescription'))).toBeTruthy() + fireEvent.change(screen.getByPlaceholderText('inbox-triage'), { target: { value: 'draft-test' } }) + fireEvent.click(screen.getByRole('button', { name: zh('bot.advanced') })) + expect(screen.getByText(zh('editor.cloneFrom'))).toBeTruthy() + expect(screen.getByText(zh('editor.shareKeysHint'))).toBeTruthy() + await act(() => i18n.setLocale('ja')) + const ja = translateBotsIn('ja') + expect(screen.getByText(ja('editor.cloneFrom'))).toBeTruthy() + expect(screen.getByDisplayValue('draft-test')).toBeTruthy() + expect(screen.getByRole('button', { name: ja('editor.createBot') })).toBeTruthy() + expect(mocks.request.mock.calls.some(([method]) => method === 'profiles.create')).toBe(false) +}) +it('localizes Edit Profile and image controls without losing user content', async () => { + mount( + undefined} + open + /> + ) + const zh = translateBotsIn('zh') + expect(screen.getByText(zh('editor.title'))).toBeTruthy() + expect(screen.getByText(zh('editor.description'))).toBeTruthy() + expect(screen.getByText(zh('editor.editDescription', 'Fixture Bot', 'fixture-bot'))).toBeTruthy() + fireEvent.click(screen.getByRole('button', { name: zh('avatar.upload') })) + expect(screen.getByRole('button', { name: zh('editor.chooseImage') })).toBeTruthy() + await act(() => i18n.setLocale('zh-hant')) + const hant = translateBotsIn('zh-hant') + expect(screen.getByRole('button', { name: hant('editor.chooseImage') })).toBeTruthy() + expect(screen.getByDisplayValue('user authored description')).toBeTruthy() +}) diff --git a/apps/desktop/src/plugins/hermes-bots/avatar-picker.tsx b/apps/desktop/src/plugins/hermes-bots/avatar-picker.tsx index 099d3e8149..a997c1675a 100644 --- a/apps/desktop/src/plugins/hermes-bots/avatar-picker.tsx +++ b/apps/desktop/src/plugins/hermes-bots/avatar-picker.tsx @@ -161,7 +161,7 @@ export function AvatarPicker({ shape, color, image, onShape, onColor, onImage, g width: 44, height: 44 }} - title={k || 'Auto — the name decides'} + title={k || b.editor.autoHint} > {k ? ( ) : ( - Auto + {b.editor.auto} )} ))} @@ -196,11 +196,11 @@ export function AvatarPicker({ shape, color, image, onShape, onColor, onImage, g variant="ghost" > - {locked ? 'Unlock' : 'Lock face'} + {locked ? b.editor.unlock : b.editor.lockFace}
- {locked ? 'Face locked — renaming won\u2019t change it.' : 'Face follows the name.'} + {locked ? b.editor.lockedHint : b.editor.unlockedHint}
{describe.trim() ? null : (
{b.bot.descriptionHint}
@@ -276,16 +276,14 @@ export function AvatarPicker({ shape, color, image, onShape, onColor, onImage, g ) : (
- {imagen === false - ? 'No image model available. If you just enabled one (or updated Hermes), restart the gateway: Ctrl+K → "Restart gateway".' - : 'Checking image backend…'} + {imagen === false ? b.editor.noImageModel : b.editor.checkingImage}
) ) : null} {tab === 'upload' ? ( ) : null} {tab === 'pet' ? : null} diff --git a/apps/desktop/src/plugins/hermes-bots/create-dialog.tsx b/apps/desktop/src/plugins/hermes-bots/create-dialog.tsx index 12a0a3afb5..7a556ebdac 100644 --- a/apps/desktop/src/plugins/hermes-bots/create-dialog.tsx +++ b/apps/desktop/src/plugins/hermes-bots/create-dialog.tsx @@ -509,7 +509,7 @@ export function CreateAgentDialog({ open, onClose, roster }: CreateAgentDialogPr if (!slugCreated) { setBusy(false) - setError('Could not create the bot.') + setError(b.bot.createFailed) return } @@ -517,14 +517,8 @@ export function CreateAgentDialog({ open, onClose, roster }: CreateAgentDialogPr host.notify({ kind: 'success', message: remoteTarget - ? `Bot "${displayName({ - name: slug, - title: botTitle - })}" created on ${targetLabel}` - : `Bot "${displayName({ - name: slug, - title: botTitle - })}" created` + ? b.editor.createdOn(displayName({ name: slug, title: botTitle }), targetLabel) + : b.editor.created(displayName({ name: slug, title: botTitle })) }) const wasRemote = remoteTarget reset() @@ -600,9 +594,7 @@ export function CreateAgentDialog({ open, onClose, roster }: CreateAgentDialogPr > {b.bot.newTitle} - - A named teammate with its own memory, skills, and chat. It can message your other agents. - + {b.editor.newDescription}
@@ -628,14 +620,12 @@ export function CreateAgentDialog({ open, onClose, roster }: CreateAgentDialogPr shape={shape} /> {labeled( - 'Name', + b.editor.name, setName(event.target.value)} placeholder="inbox-triage" value={name} /> )} {taken ? (
- {remoteTarget - ? `An agent named "${slug}" already exists on ${targetLabel}.` - : `An agent named "${slug}" already exists.`} + {remoteTarget ? b.editor.nameTakenOn(slug, targetLabel) : b.editor.nameTaken(slug)}
) : null} {/* Multi-connection desktops choose WHERE the agent lives. Hidden */ @@ -643,7 +633,7 @@ export function CreateAgentDialog({ open, onClose, roster }: CreateAgentDialogPr /* possible home, exactly the old behavior. */} {Array.isArray(connections) && connections.length > 1 ? labeled( - 'Create on', + b.editor.createOn, setTitle(event.target.value)} placeholder="Inbox Triage" value={title} /> )} {labeled( - 'Description', + b.editor.description,