From 6ee218602d5ceccb3de025f1cfe5010165165cb3 Mon Sep 17 00:00:00 2001 From: David Gardner Date: Thu, 10 Sep 2026 13:23:24 -0700 Subject: [PATCH 01/98] feat(relay): pass API request metadata to Relay tracking Signed-off-by: David Gardner --- agent/relay_runtime.py | 18 +++++- agent/turn_facade.py | 9 ++- gateway/platforms/api_server.py | 25 +++++++-- gateway/platforms/api_server_openai_routes.py | 7 ++- tests/agent/test_relay_session_segments.py | 30 ++++++++++ tests/gateway/test_api_server.py | 55 +++++++++++++++++++ 6 files changed, 136 insertions(+), 8 deletions(-) diff --git a/agent/relay_runtime.py b/agent/relay_runtime.py index a204ec5b76..7c0b78d2ea 100644 --- a/agent/relay_runtime.py +++ b/agent/relay_runtime.py @@ -922,7 +922,14 @@ class RelaySessionCoordinator: return host.register_subagent(event, metadata=metadata) return host.ensure_session({"session_id": session_id}, metadata=metadata) - def begin_turn(self, lease: ConversationLease, *, turn_id: str, task_id: str) -> RelayTurnContext: + def begin_turn( + self, + lease: ConversationLease, + *, + turn_id: str, + task_id: str, + metadata: dict[str, Any] | None = None, + ) -> RelayTurnContext: if lease.released: raise RuntimeError("Hermes Relay conversation lease is released") turn = RelayTurnContext(lease=lease, turn_id=turn_id, task_id=task_id) @@ -942,10 +949,17 @@ class RelaySessionCoordinator: if host is not None: # Rotation happens HERE: no live turn scope on the stack, so the session scope can close/reopen LIFO. _warn_on_error("segment rotation", self._maybe_rotate_segment, host, lease.session) + turn_metadata = dict(metadata or {}) + turn_metadata.update( + runtime_metadata( + host.runtime_id, + **{"hermes.execution_surface": lease.platform or "unknown"}, + ) + ) turn.handle = _warn_on_error( "turn initialization", host.run_in_session, lease.session, host.relay.scope.push, TURN_SCOPE, host.relay.ScopeType.Function, handle=lease.session.handle, input={}, - metadata=runtime_metadata(host.runtime_id, **{"hermes.execution_surface": lease.platform or "unknown"}), + metadata=turn_metadata, timeout=_SCOPE_OP_TIMEOUT, ) turn._previous_turn = _CURRENT_TURN.get() diff --git a/agent/turn_facade.py b/agent/turn_facade.py index acd3bc97ed..5eb2b500d8 100644 --- a/agent/turn_facade.py +++ b/agent/turn_facade.py @@ -27,6 +27,7 @@ class TurnFacadeMixin: persist_user_display_metadata: Optional[Dict[str, Any]]=None, persist_user_platform_id: Optional[str]=None, moa_config: Optional[dict[str, Any]]=None, turn_author: Optional[Dict[str, Any]] = None, + relay_metadata: Optional[Dict[str, Any]] = None, ) -> Dict[str, Any]: """Forwarder — see ``agent.conversation_loop.run_conversation``.""" # A review shares this session_id for cache parity: fence review startup or interrupt @@ -94,8 +95,14 @@ class TurnFacadeMixin: parent_session_id=relay_parent_session_id, model=str(getattr(self, "model", None) or ""), ) + relay_turn_kwargs: Dict[str, Any] = { + "turn_id": relay_turn_id, + "task_id": effective_task_id, + } + if relay_metadata: + relay_turn_kwargs["metadata"] = relay_metadata relay_turn = relay_runtime.SESSION_COORDINATOR.begin_turn( - relay_lease, turn_id=relay_turn_id, task_id=effective_task_id + relay_lease, **relay_turn_kwargs ) # Minimal relay-runtime shims may lack the opt-out flag: default enabled. if getattr(relay_turn, "relay_enabled", True): diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index c7d2c05cec..7c279ce4ec 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -345,6 +345,16 @@ def _request_agent_overrides( return overrides +def _request_relay_metadata(body: Any) -> Dict[str, Any]: + """Extract Relay metadata from an OpenAI request body.""" + if not isinstance(body, dict): + return {} + metadata = body.get("metadata") + if not isinstance(metadata, dict): + return {} + return dict(metadata) + + def _is_compressed_summary_message(message: Any) -> bool: """Recognize every compaction carrier shape via the compressor's own classifier (SessionDB drops the in-process marker; a prefix scan misses merge-into-tail carriers).""" @@ -3641,7 +3651,8 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): route: Optional[Dict[str, Any]] = None, session_model: Optional[str] = None, requested_runtime: Optional[Dict[str, Any]] = None, route_source: str = "global", confirmed_runtime_lock: bool = False, bind_declared_conversation: bool = False, - session_history_delivery: str = "", turn_author: Optional[Dict[str, Any]] = None) -> tuple: + session_history_delivery: str = "", turn_author: Optional[Dict[str, Any]] = None, + relay_metadata: Optional[Dict[str, Any]] = None) -> tuple: """Create an agent and run one turn in a thread executor -> ``(result, usage)``. ``agent_ref[0]`` receives the agent so SSE writers can interrupt it; ``active_run_id`` registers it in ``_active_run_agents``. Under a confirmed model lock the actual @@ -3695,9 +3706,15 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): self._shutdown_interruptible_agents[id(agent)] = agent # Passed only when set: a human turn keeps today's call shape. author_kwargs = {"turn_author": turn_author} if turn_author is not None else {} - result = agent.run_conversation( - user_message=user_message, conversation_history=conversation_history, - task_id=effective_task_id, **author_kwargs) + conversation_kwargs = dict( + user_message=user_message, + conversation_history=conversation_history, + task_id=effective_task_id, + **author_kwargs, + ) + if relay_metadata: + conversation_kwargs["relay_metadata"] = relay_metadata + result = agent.run_conversation(**conversation_kwargs) return self._finish_turn_result( agent, result, session_id, route=route, requested_runtime=requested_runtime, route_source=route_source, confirmed_runtime_lock=confirmed_runtime_lock) diff --git a/gateway/platforms/api_server_openai_routes.py b/gateway/platforms/api_server_openai_routes.py index 998954d39f..3d4c92dce9 100644 --- a/gateway/platforms/api_server_openai_routes.py +++ b/gateway/platforms/api_server_openai_routes.py @@ -421,6 +421,8 @@ class OpenAICompatRoutesMixin: body = await request.json() except Exception: return _error_response("Invalid JSON in request body", 400) + from gateway.platforms.api_server import _request_relay_metadata + relay_metadata = _request_relay_metadata(body) messages = body.get("messages") if not messages or not isinstance(messages, list): return _invalid_request("Missing or invalid 'messages' field") @@ -502,6 +504,7 @@ class OpenAICompatRoutesMixin: user_message=user_message, conversation_history=history, ephemeral_system_prompt=system_prompt, session_id=session_id, gateway_session_key=gateway_session_key, **agent_overrides, route=route, + relay_metadata=relay_metadata, # #98619: only an explicitly provided X-Hermes-Session-Id is wake-capable (the # header is 403-gated on API_SERVER_KEY, so the wake self-post can authenticate # and the client can resume the session by sending it again). A fingerprint-derived @@ -766,6 +769,8 @@ class OpenAICompatRoutesMixin: body = await request.json() except Exception: return _invalid_request("Invalid JSON in request body") + from gateway.platforms.api_server import _request_relay_metadata + relay_metadata = _request_relay_metadata(body) raw_input = body.get("input") if raw_input is None: return _error_response("Missing 'input' field", 400) @@ -846,7 +851,7 @@ class OpenAICompatRoutesMixin: user_message=user_message, conversation_history=conversation_history, ephemeral_system_prompt=instructions, session_id=session_id, gateway_session_key=gateway_session_key, bind_declared_conversation=_declared_selected, - **agent_overrides, route=route) + **agent_overrides, route=route, relay_metadata=relay_metadata) if stream: _stream_q = ThreadSafeAsyncQueue() diff --git a/tests/agent/test_relay_session_segments.py b/tests/agent/test_relay_session_segments.py index 2e778c4d47..54311eadae 100644 --- a/tests/agent/test_relay_session_segments.py +++ b/tests/agent/test_relay_session_segments.py @@ -167,6 +167,36 @@ def _run_turn(coordinator, lease, turn_id): return turn +class TestTurnMetadata: + def test_includes_request_metadata_without_overriding_runtime_fields( + self, coordinator + ): + fake = _FakeRelay() + runtime = _make_runtime(fake) + lease = _acquire(coordinator, runtime) + + turn = coordinator.begin_turn( + lease, + turn_id="t1", + task_id="task1", + metadata={ + "request_id": "req-123", + "context": {"tenant": "example"}, + relay_runtime.RUNTIME_INSTANCE_KEY: "caller-supplied", + }, + ) + + turn_metadata = [ + push + for push in fake.scope.pushes + if push["name"] == relay_runtime.TURN_SCOPE + ][-1]["metadata"] + assert turn_metadata["request_id"] == "req-123" + assert turn_metadata["context"] == {"tenant": "example"} + assert turn_metadata[relay_runtime.RUNTIME_INSTANCE_KEY] == runtime.runtime_id + coordinator.end_turn(turn, outcome="success") + + class TestDefaultsNeverRotate: def test_no_rotation_across_many_turns_and_compactions(self, coordinator): fake = _FakeRelay() diff --git a/tests/gateway/test_api_server.py b/tests/gateway/test_api_server.py index f9f2a5e0b9..8abb8daf90 100644 --- a/tests/gateway/test_api_server.py +++ b/tests/gateway/test_api_server.py @@ -35,6 +35,7 @@ from gateway.platforms.api_server import ( _hermes_version, _redact_api_error_text, _request_agent_overrides, + _request_relay_metadata, check_api_server_requirements, cors_middleware, security_headers_middleware, @@ -429,6 +430,36 @@ class TestAgentExecution: assert mock_agent._gateway_turn_process_baseline == frozenset() +class TestRelayMetadataForwarding: + @pytest.mark.asyncio + @pytest.mark.parametrize( + ("endpoint", "payload"), + [ + ( + "/v1/chat/completions", + {"messages": [{"role": "user", "content": "hi"}]}, + ), + ("/v1/responses", {"input": "hi"}), + ], + ) + async def test_openai_requests_forward_metadata_to_relay( + self, adapter, endpoint, payload + ): + app = _create_app(adapter) + metadata = {"request_id": "req-123", "context": {"tenant": "example"}} + with patch.object(adapter, "_run_agent", new_callable=AsyncMock) as mock_run: + mock_run.return_value = ( + {"final_response": "ok", "messages": [], "api_calls": 1}, + {"input_tokens": 1, "output_tokens": 1, "total_tokens": 2}, + ) + async with TestClient(TestServer(app)) as cli: + response = await cli.post(endpoint, json={**payload, "metadata": metadata}) + + assert response.status == 200 + assert mock_run.call_args.kwargs["relay_metadata"] == metadata + assert mock_run.call_args.kwargs["relay_metadata"] is not metadata + + class TestDisconnectedAgentReap: """#76188 review: SSE disconnect handlers must reap only the background processes the disconnected turn created, and must no-op when no turn @@ -2844,6 +2875,30 @@ class TestKeyRejectionSetsNonRetryableFatalError: await self._assert_key_rejection_is_fatal(adapter) +# --------------------------------------------------------------------------- +# Relay metadata extraction +# --------------------------------------------------------------------------- + + +class TestRequestRelayMetadata: + def test_copies_all_metadata_fields(self): + metadata = { + "request_id": "req-123", + "attempt": 2, + "tags": ["batch", "evaluation"], + "context": {"tenant": "example"}, + } + + extracted = _request_relay_metadata({"metadata": metadata}) + + assert extracted == metadata + assert extracted is not metadata + + @pytest.mark.parametrize("body", [None, [], {}, {"metadata": "invalid"}]) + def test_ignores_non_object_metadata(self, body): + assert _request_relay_metadata(body) == {} + + # --------------------------------------------------------------------------- # Bare-model opt-in gate (direct_model_requests) for _request_agent_overrides # --------------------------------------------------------------------------- From bd2c7e245768092608158f15195d229ed2d497d2 Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Fri, 11 Sep 2026 02:13:07 -0700 Subject: [PATCH 02/98] fix(state): drop orphan FTS5 shadow tables per family on SessionDB open `sqlite3 state.db .recover` re-emits the FTS5 shadow tables (messages_fts_data, _idx, _docsize, _config, _content) as ordinary tables but cannot re-emit the CREATE VIRTUAL TABLE row. The next SessionDB open ran the FTS DDL in _ensure_fts_schema and died with "fts5: error creating shadow table messages_fts_data: table 'messages_fts_data' already exists", so a recovered database was unusable until someone hand-dropped the shadows. _init_fts now runs _drop_orphan_fts_shadow_tables before any FTS DDL. It is per-family and exact-name scoped: a family's shadows are dropped only when its own vtable row is absent from sqlite_master (type='table' AND sql LIKE 'CREATE VIRTUAL TABLE%'), so a healthy messages_fts_trigram survives a base-family repair untouched. A repaired base/trigram family is then treated like a missing-trigger repair and rebuilt from the canonical messages table under the cross-process rebuild admission; the shadows are derived index state, nothing is lost. Live repro: real `sqlite3 x.db .recover | sqlite3 y.db` on sqlite 3.50.4 keeps the vtable rows (the shell emits CREATE VIRTUAL TABLE), so the deterministic fixture removes the vtable row via writable_schema leaving the shadows behind: BEFORE OperationalError on open; AFTER opens, fts_enabled, MATCH returns every message, trigram sqlite_master rowids unchanged. Salvaged from PR #56824 (intent applied onto the current hermes_state_fts / hermes_state_schema siblings). The ownership-safety point (never touch a live family's shadows) was raised by @ggoldani in #103868 / #103840. Refs #103840 Refs #56815 Co-authored-by: ggoldani --- hermes_state_fts.py | 42 ++++++++++++ hermes_state_schema.py | 13 +++- tests/state/test_fts_orphan_shadow_repair.py | 67 ++++++++++++++++++++ 3 files changed, 120 insertions(+), 2 deletions(-) create mode 100644 tests/state/test_fts_orphan_shadow_repair.py diff --git a/hermes_state_fts.py b/hermes_state_fts.py index 56b8ff60d7..33815f73a0 100644 --- a/hermes_state_fts.py +++ b/hermes_state_fts.py @@ -6,6 +6,7 @@ import logging import os import sqlite3 from pathlib import Path +from typing import Sequence from hermes_constants import get_hermes_home from hermes_state_common import FTS_CJK_STALE_KEY, FTS_STALE_KEY, _FTS_CJK_TRIGGERS, _FTS_TRIGGERS @@ -122,6 +123,47 @@ def load_fts5_cjk_extension(conn: sqlite3.Connection) -> bool: return False +# FTS5 shadow tables the virtual-table engine owns. `sqlite3 .recover` re-emits them +# as ordinary tables but cannot re-emit the CREATE VIRTUAL TABLE row, so every later +# CREATE VIRTUAL TABLE fails with "fts5: error creating shadow table : table +# already exists" until the orphans are dropped (#103840). +_FTS5_SHADOW_SUFFIXES = ("content", "data", "docsize", "idx", "config") + + +def _drop_orphan_fts_shadow_tables(cursor: sqlite3.Cursor, families: Sequence[str]) -> list[str]: + """Drop, per family, shadow tables whose virtual table row is absent from sqlite_master. + + Matches exact shadow names only (never a prefix LIKE, so the base family cannot reach + ``messages_fts_trigram_*``) and leaves a family alone whenever its vtable is live. The + shadows are derived index state; the caller recreates and rebuilds from ``messages``. + Returns the families that were repaired. + """ + repaired: list[str] = [] + for family in families: + vtable_live = cursor.execute( + "SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = ? " + "AND sql LIKE 'CREATE VIRTUAL TABLE%'", + (family,), + ).fetchone() + if vtable_live: + continue + shadows = [f"{family}_{suffix}" for suffix in _FTS5_SHADOW_SUFFIXES] + orphans = [row[0] for row in cursor.execute( + f"SELECT name FROM sqlite_master WHERE type = 'table' AND name IN ({','.join('?' for _ in shadows)})", + shadows, + ).fetchall()] + if not orphans: + continue + for name in orphans: + cursor.execute(f'DROP TABLE "{name}"') + logger.warning( + "Dropped orphan FTS5 shadow tables of %s (%s); the index is recreated from messages", + family, ", ".join(orphans), + ) + repaired.append(family) + return repaired + + class SessionFtsSetupMixin: """FTS table/trigger lifecycle shared by schema init, optimize and the write path.""" diff --git a/hermes_state_schema.py b/hermes_state_schema.py index 977ee163e4..eafa1835ad 100644 --- a/hermes_state_schema.py +++ b/hermes_state_schema.py @@ -26,6 +26,7 @@ from hermes_state_common import ( LEGACY_FTS_TRIGRAM_SQL, SCHEMA_SQL, SCHEMA_VERSION, _FTS_CJK_TRIGGERS, _FTS_TRIGGERS, _ephemeral_child_sql, _sql_json_extract, fts_rebuild_admission, ) +from hermes_state_fts import _drop_orphan_fts_shadow_tables from hermes_state_holders import _read_proc_argv # Pre-split logger identity so log filtering/capture is unchanged. @@ -1129,6 +1130,12 @@ class SessionSchemaMixin: OPT-IN v23 boundary: a legacy v22 inline install keeps its inline schema + triggers (the v23 DDL would create the trigram source VIEW and leave a mixed state).""" legacy_fts = self._db_has_legacy_inline_fts(cursor) + # A `.recover`-restored image keeps the shadow tables but not the vtable rows; the DDL + # below would fail on the first shadow. Drop only orphaned families, then rebuild the + # recreated (empty) index like a missing-trigger repair (#103840). + orphan_repaired = _drop_orphan_fts_shadow_tables( + cursor, ("messages_fts", "messages_fts_trigram", "messages_fts_cjk"), + ) if not self._fts_stale: self._migrate_bounded_tool_fts_triggers(cursor, legacy=legacy_fts) if self._fts_stale: @@ -1142,8 +1149,10 @@ class SessionSchemaMixin: # Measure before any DDL. Publishing missing base triggers before rebuild admission lets # another process write through an index whose bootstrap/repair has no owner (#105790). base_triggers_missing = self._fts_triggers_missing(cursor, _FTS_BASE_TRIGGERS) or getattr( - self, "_fts_tool_prefix_migration_requires_rebuild", False) - trigram_triggers_missing = self._fts_triggers_missing(cursor, _FTS_TRIGRAM_TRIGGERS) + self, "_fts_tool_prefix_migration_requires_rebuild", False) or "messages_fts" in orphan_repaired + trigram_triggers_missing = ( + self._fts_triggers_missing(cursor, _FTS_TRIGRAM_TRIGGERS) or "messages_fts_trigram" in orphan_repaired + ) def ensure_and_rebuild() -> None: self._fts_enabled = self._ensure_fts_schema(cursor, "messages_fts", base_sql) diff --git a/tests/state/test_fts_orphan_shadow_repair.py b/tests/state/test_fts_orphan_shadow_repair.py new file mode 100644 index 0000000000..f7661b20c6 --- /dev/null +++ b/tests/state/test_fts_orphan_shadow_repair.py @@ -0,0 +1,67 @@ +"""#103840: a ``sqlite3 .recover`` restore re-emits FTS5 shadow tables as ordinary tables +but cannot re-emit the ``CREATE VIRTUAL TABLE`` row. The next SessionDB open then failed +in ``_ensure_fts_schema`` with "fts5: error creating shadow table messages_fts_data: table +already exists". Only families whose vtable row is absent may be repaired; a healthy family's +shadows must survive untouched. +""" + +import sqlite3 + +from hermes_state import SessionDB + + +def _orphan_family(db_path, family: str) -> None: + """Emulate the ``.recover`` residue for one family: vtable row gone, shadows kept.""" + raw = sqlite3.connect(db_path) + raw.isolation_level = None + raw.execute("PRAGMA writable_schema=ON") + raw.execute( + "DELETE FROM sqlite_master WHERE name = ? AND sql LIKE 'CREATE VIRTUAL TABLE%'", (family,), + ) + version = raw.execute("PRAGMA schema_version").fetchone()[0] + raw.execute(f"PRAGMA schema_version={version + 1}") + raw.execute("PRAGMA writable_schema=OFF") + raw.close() + + +def _fts_master_rows(db_path, prefix: str) -> list: + raw = sqlite3.connect(db_path) + try: + return raw.execute( + "SELECT rowid, type, name FROM sqlite_master WHERE name LIKE ? ESCAPE '\\' ORDER BY rowid", + (prefix.replace("_", "\\_") + "%",), + ).fetchall() + finally: + raw.close() + + +def test_orphaned_base_family_is_repaired_and_healthy_trigram_untouched(tmp_path): + db_path = tmp_path / "state.db" + db = SessionDB(db_path=db_path) + db.create_session("s1", source="cli", model="m") + for i in range(3): + db.append_message("s1", "user", f"recovered orphan {i}") + db.close() + + _orphan_family(db_path, "messages_fts") + trigram_before = _fts_master_rows(db_path, "messages_fts_trigram") + assert trigram_before, "fixture needs a live trigram family" + raw = sqlite3.connect(db_path) + orphan_shadows = raw.execute( + "SELECT count(*) FROM sqlite_master WHERE name IN ('messages_fts_data', 'messages_fts_config')" + ).fetchone()[0] + raw.close() + assert orphan_shadows == 2, "fixture must leave the base shadows behind" + + reopened = SessionDB(db_path=db_path) + try: + assert reopened._fts_enabled is True + with reopened._lock: + hits = reopened._conn.execute( + "SELECT count(*) FROM messages_fts WHERE messages_fts MATCH 'orphan'" + ).fetchone()[0] + assert hits == 3, "recreated index must be rebuilt from the canonical messages table" + finally: + reopened.close() + # Same rowids: the healthy family was neither dropped nor recreated. + assert _fts_master_rows(db_path, "messages_fts_trigram") == trigram_before From 1dcc0b06e29b25c22aa3440f342edfdb0819c091 Mon Sep 17 00:00:00 2001 From: Gunwoo Lee Date: Fri, 11 Sep 2026 02:18:12 -0700 Subject: [PATCH 03/98] fix(recovery): skip rows the destination schema rejects during partial salvage MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `hermes sessions recover --allow-partial` walks damaged tables by rowid range and falls back to an exact-rowid lookup for the last cell of a broken range. A phantom row produced by page damage (a `sessions` row with a NULL `started_at`, an `async_delegations` row with a NULL `state`) reads fine from the source but violates the destination's NOT NULL constraint, and the resulting sqlite3.IntegrityError escaped the exact-lookup boundary and aborted the whole recovery — losing every healthy row behind it. Catch IntegrityError at that boundary, count it under `destination_rejected_rows` and record the rowid as a skipped singleton ("destination constraint rejected row: ...") so the run completes, verifies, and the report shows exactly which rows were dropped. Live repro (real fixture: source schema's NOT NULL relaxed via writable_schema, phantom `sessions` row inserted): BEFORE sqlite3.IntegrityError "NOT NULL constraint failed: sessions.started_at" at session_recovery.py recover_exact_rowid; AFTER status=partial copied=3 destination_rejected_rows=1, verified=True. Reporter @i8ei confirmed the same patch recovers the field database (22 sessions / 2,248 messages) in #102240. Salvaged from PR #91413 (rebased onto the _RowidRangeSalvage refactor; test reduced to one invariant). Refs #102240 Reported-by: i8ei --- contributors/emails/gunwoo@wustl.edu | 2 ++ hermes_cli/session_recovery.py | 14 ++++++-- tests/hermes_cli/test_session_recovery.py | 42 +++++++++++++++++++++++ 3 files changed, 55 insertions(+), 3 deletions(-) create mode 100644 contributors/emails/gunwoo@wustl.edu diff --git a/contributors/emails/gunwoo@wustl.edu b/contributors/emails/gunwoo@wustl.edu new file mode 100644 index 0000000000..3e5c997d18 --- /dev/null +++ b/contributors/emails/gunwoo@wustl.edu @@ -0,0 +1,2 @@ +leegunwoo98 +# PR #91413 salvage (skip destination-rejected rows in partial recovery; #102240) diff --git a/hermes_cli/session_recovery.py b/hermes_cli/session_recovery.py index 73cf4645b1..28fd11f208 100644 --- a/hermes_cli/session_recovery.py +++ b/hermes_cli/session_recovery.py @@ -511,8 +511,15 @@ class _RowidRangeSalvage: if not self._keep([value]): result["excluded_rows"] += 1 return True - with _immediate_transaction(self.destination): - self.destination.execute(self.insert_sql, value) + try: + with _immediate_transaction(self.destination): + self.destination.execute(self.insert_sql, value) + except sqlite3.IntegrityError as exc: + # A phantom row from a damaged page (NULL in a NOT NULL column, FK to nothing) is rejected + # by the destination schema; report it as a skipped singleton instead of aborting the table. + result["destination_rejected_rows"] += 1 + self._skip(rowid, rowid, f"destination constraint rejected row: {exc}") + return True result["copied_rows"] += 1 result["exact_lookup_recovered"] += 1 return True @@ -568,7 +575,8 @@ def _copy_table_salvage( """Best-effort rowid-range copy that continues past damaged source pages.""" result: dict[str, Any] = { "mode": "rowid_range_salvage", "source_rows": source_rows, "copied_rows": 0, "excluded_rows": 0, - "columns": [], "range_queries": 0, "exact_lookup_recovered": 0, "skipped_rowid_ranges": [], + "columns": [], "range_queries": 0, "exact_lookup_recovered": 0, "destination_rejected_rows": 0, + "skipped_rowid_ranges": [], } columns = _compatible_columns(source, destination, table, result) if columns is None: diff --git a/tests/hermes_cli/test_session_recovery.py b/tests/hermes_cli/test_session_recovery.py index c9549018d7..899906de2c 100644 --- a/tests/hermes_cli/test_session_recovery.py +++ b/tests/hermes_cli/test_session_recovery.py @@ -834,3 +834,45 @@ def test_lost_and_found_direct_copy_creates_lazy_delivery_ledger(tmp_path: Path) lf_conn.close() dest.close() assert rows == [("ob-1", "pending", None), ("ob-2", "failed", "boom")] + + + +def test_partial_recovery_skips_phantom_row_rejected_by_destination_schema( + tmp_path: Path, +) -> None: + """#102240: a phantom ``sessions`` row with NULL ``started_at`` must be reported as a skipped + singleton, not abort the whole ``--allow-partial`` run at the exact-lookup boundary.""" + source = tmp_path / "phantom-state.db" + output = tmp_path / "phantom-recovered.db" + _make_source(source) + + # Relax the source's NOT NULL in place (schema text only, the pages stay identical) so the + # source can hold a row the destination's canonical schema rejects. + with sqlite3.connect(str(source), isolation_level=None) as conn: + conn.execute("PRAGMA writable_schema=ON") + conn.execute( + "UPDATE sqlite_master SET sql = replace(sql, 'started_at REAL NOT NULL', 'started_at REAL') " + "WHERE type = 'table' AND name = 'sessions'" + ) + version = conn.execute("PRAGMA schema_version").fetchone()[0] + conn.execute(f"PRAGMA schema_version={version + 1}") + conn.execute("PRAGMA writable_schema=OFF") + with sqlite3.connect(str(source), isolation_level=None) as conn: + conn.execute( + "INSERT INTO sessions (id, source, started_at, title) VALUES ('phantom', 'cli', NULL, 'Phantom')" + ) + assert conn.execute("SELECT count(*) FROM sessions").fetchone()[0] == 4 + + report = recover_session_database(source, output, work_dir=tmp_path, chunk_size=16, allow_partial=True) + + copied = report["copy"]["sessions"] + assert copied["status"] == "partial" + assert copied["copied_rows"] == 3 + assert copied["destination_rejected_rows"] == 1 + assert [item["error"] for item in copied["skipped_rowid_ranges"]] == [ + "destination constraint rejected row: NOT NULL constraint failed: sessions.started_at", + ] + assert report["verified"] is True + with sqlite3.connect(str(output)) as conn: + assert conn.execute("SELECT count(*) FROM sessions WHERE id = 'phantom'").fetchone()[0] == 0 + assert conn.execute("SELECT count(*) FROM messages").fetchone()[0] == 21 From 8a9f9eca2ea0fbf9938265f3af3554332ab4b19d Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 11 Sep 2026 02:26:48 -0700 Subject: [PATCH 04/98] fix(recovery): seed a damaged rowid edge from min()/max() before the INT64 domain MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When the leftmost (or rightmost) leaf of a table b-tree is damaged, the edge probe `SELECT rowid ... ORDER BY rowid ASC LIMIT 1` walks the table tree and raises, and _salvage_rowid_bounds fell back to INT64_MIN. The gallop from the surviving edge cannot cap that side either (every probe crosses the damaged leaf), so bisection burned the entire 10,000-query budget moving the bound inward by a few thousand rowids out of 9.2e18 and the table was lost — a 4-row gateway_routing table in #98050, sessions + session_model_usage in #100313. `SELECT min(rowid), max(rowid)` is answered by the planner from any covering index (every Hermes table has at least the PRIMARY KEY autoindex) without touching the damaged leaf, which is exactly what the reporter verified by hand. Ask it for the missing edge(s) first; only when it fails too does the domain fallback + gallop run as before. Reported under `aggregate_edges` so recovery.json still shows how the bound was obtained. Live repro (real fixture: leftmost `sessions` leaf cell count overwritten, 400 rows): BEFORE bounds low=-9223372036854775808 copied=0 range_queries=10000 query_limit_reached=True status=failed; AFTER low=1 high=400 copied=391 range_queries=40 status=partial (only the damaged leaf's rows are lost). Refs #98050 Refs #100313 Reported-by: Ace-Kelly Corroborated-by: Proff506 --- hermes_cli/session_recovery.py | 16 ++++++++ tests/hermes_cli/test_session_recovery.py | 39 +++++++++++++++++++ .../test_session_recovery_lost_and_found.py | 5 ++- 3 files changed, 58 insertions(+), 2 deletions(-) diff --git a/hermes_cli/session_recovery.py b/hermes_cli/session_recovery.py index 28fd11f208..9659beb9fd 100644 --- a/hermes_cli/session_recovery.py +++ b/hermes_cli/session_recovery.py @@ -414,6 +414,22 @@ def _salvage_rowid_bounds(source: sqlite3.Connection, table: str) -> dict[str, A result["empty" if not result["errors"] else "unavailable"] = True return result + # An ordered LIMIT 1 walks the table b-tree and dies on a damaged edge leaf, while the + # aggregate lets the planner answer from any covering index (every Hermes table has at + # least a PRIMARY KEY autoindex). Ask it before falling back to the synthetic domain: + # bisecting from INT64_MIN burned the whole query budget on a 4-row table (#98050). + missing = [edge for edge in ("low", "high") if rows[edge] is None] + if missing: + try: + aggregate = source.execute(f'SELECT min(rowid), max(rowid) FROM "{table}"').fetchone() + except sqlite3.DatabaseError as exc: + result["errors"].append(f"aggregate rowid bounds: {exc}") + else: + for edge, value in zip(("low", "high"), aggregate): + if rows[edge] is None and value is not None: + rows[edge] = int(value) + result.setdefault("aggregate_edges", []).append(edge) + # A damaged edge can stop one ordered probe. Keep the readable edge and bound the other side by the # SQLite rowid domain, so bisection never assumes user databases hold only positive ids. if rows["low"] is None: diff --git a/tests/hermes_cli/test_session_recovery.py b/tests/hermes_cli/test_session_recovery.py index 899906de2c..1318202036 100644 --- a/tests/hermes_cli/test_session_recovery.py +++ b/tests/hermes_cli/test_session_recovery.py @@ -876,3 +876,42 @@ def test_partial_recovery_skips_phantom_row_rejected_by_destination_schema( with sqlite3.connect(str(output)) as conn: assert conn.execute("SELECT count(*) FROM sessions WHERE id = 'phantom'").fetchone()[0] == 0 assert conn.execute("SELECT count(*) FROM messages").fetchone()[0] == 21 + + + +def test_salvage_bounds_damaged_low_edge_from_the_aggregate_not_the_int64_domain( + tmp_path: Path, +) -> None: + """#98050: with the leftmost leaf damaged, ``ORDER BY rowid ASC LIMIT 1`` fails while + ``min(rowid)`` still answers via the covering index. Bisecting from INT64_MIN burned the + whole 10,000-query budget and lost every row; the aggregate must seed the bound instead.""" + source = tmp_path / "low-edge.db" + sessions_root = _make_many_sessions_source(source, session_count=180) + page_size, leaf_pages = _btree_leaf_pages(source, sessions_root) + assert len(leaf_pages) >= 3 + first_leaf = leaf_pages[0] + data = bytearray(source.read_bytes()) + header_offset = (first_leaf - 1) * page_size + assert data[header_offset] == 0x0D + data[header_offset + 3 : header_offset + 5] = b"\xff\xff" + source.write_bytes(data) + + conn = sqlite3.connect(str(source)) + try: + with pytest.raises(sqlite3.DatabaseError): + conn.execute('SELECT rowid FROM "sessions" ORDER BY rowid ASC LIMIT 1').fetchone() + bounds = session_recovery._salvage_rowid_bounds(conn, "sessions") + assert bounds["low"] == 1 and bounds["high"] == 180 + assert bounds["fallback_edges"] == [] + + destination = sqlite3.connect(":memory:") + destination.execute("CREATE TABLE sessions (id TEXT PRIMARY KEY, source TEXT, started_at REAL)") + result = session_recovery._copy_table_salvage( + conn, destination, "sessions", chunk_size=16, progress_cb=None, source_rows=180, + ) + finally: + conn.close() + assert result["query_limit_reached"] is False + assert result["range_queries"] < 200 + # Only the rows on the damaged leaf are lost; everything behind it is recovered. + assert result["copied_rows"] >= 180 - 60 diff --git a/tests/hermes_cli/test_session_recovery_lost_and_found.py b/tests/hermes_cli/test_session_recovery_lost_and_found.py index 0927ceaa15..a8711dde86 100644 --- a/tests/hermes_cli/test_session_recovery_lost_and_found.py +++ b/tests/hermes_cli/test_session_recovery_lost_and_found.py @@ -157,9 +157,10 @@ def test_exact_lookup_recovers_tail_row_next_to_damaged_high_edge( copied = report["copy"]["messages"] bounds = copied["rowid_bounds"] - # Premise check: the high edge probe really failed and fell back. + # Premise check: the high edge probe really failed; the bound came from the aggregate + # (#98050) or, when that fails too, the synthetic-domain fallback. assert any("high rowid" in error for error in bounds["errors"]), bounds - assert "high" in bounds["fallback_edges"] + assert "high" in bounds["fallback_edges"] or "high" in bounds.get("aggregate_edges", ()) conn = sqlite3.connect(str(output)) try: From b811dfd0eee5b941209c98db8b38625213f0632a Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Sat, 5 Sep 2026 20:49:48 +0800 Subject: [PATCH 05/98] fix(state): exclude the cjk index family from the legacy FTS demote rename The demote enumeration (name LIKE 'messages_fts_%') sweeps the messages_fts_cjk vtable and its shadow tables into the fts_v22_trash_* renames. Renaming the cjk vtable cascades to its shadow tables and breaks the vtable constructor chain, so 'hermes sessions optimize-storage' aborts with 'vtable constructor failed: messages_fts_cjk' on every DB that carries both a legacy inline FTS layout and an established cjk index (#103647). The cjk family is an independent v23+ index, not part of the demoted legacy layout: skip it in the enumeration. --- hermes_state_search.py | 6 ++- tests/test_fts_cjk_bigram.py | 79 ++++++++++++++++++++++++++++++++++++ 2 files changed, 84 insertions(+), 1 deletion(-) diff --git a/hermes_state_search.py b/hermes_state_search.py index f1dadec66d..3cd3427862 100644 --- a/hermes_state_search.py +++ b/hermes_state_search.py @@ -530,7 +530,11 @@ class SessionSearchMixin: for row in conn.execute( "SELECT name FROM sqlite_master WHERE type = 'table' " "AND (name LIKE 'messages_fts_%' ESCAPE '\\' " - "OR name LIKE 'messages_fts_trigram_%' ESCAPE '\\')" + "OR name LIKE 'messages_fts_trigram_%' ESCAPE '\\') " + # messages_fts_cjk* is an independent v23+ index, not part of the + # demoted legacy layout: renaming it (or its shadow tables) breaks + # the vtable constructor chain (#103647). + "AND name NOT LIKE 'messages_fts_cjk%'" ).fetchall(): conn.execute(f"ALTER TABLE {row[0]} RENAME TO fts_v22_trash_{row[0]}") # Claim the backfill BEFORE the empty v23 tables exist so a crash before diff --git a/tests/test_fts_cjk_bigram.py b/tests/test_fts_cjk_bigram.py index 01c2dfa312..19c20e6460 100644 --- a/tests/test_fts_cjk_bigram.py +++ b/tests/test_fts_cjk_bigram.py @@ -212,6 +212,85 @@ def test_legacy_v22_optimize_lands_on_cjk(cjk_so, tmp_path, monkeypatch): d.close() +def test_optimize_demote_leaves_established_cjk_index_intact(cjk_so, tmp_path, monkeypatch): + """#103647: a legacy inline-FTS DB that ALSO carries an established (backfilled, + trigger-live) messages_fts_cjk index must demote without touching the cjk family. + The demote enumeration matches 'messages_fts_%', which sweeps the cjk vtable and + its shadow tables into the fts_v22_trash_* renames; renaming them breaks the + vtable constructor chain ('vtable constructor failed: messages_fts_cjk') and + aborts optimize-storage on every CJK-enabled host before any space is reclaimed.""" + import time as _time + + from hermes_state_common import SCHEMA_SQL + from hermes_state_fts import FTS_CJK_TABLE_SQL, FTS_CJK_TRIGGER_SQL + + monkeypatch.setenv("HERMES_FTS5_CJK_SO", str(cjk_so)) + db_path = tmp_path / "state.db" + + # Hand-build the coexistence shape: legacy inline FTS + a live cjk index. + conn = sqlite3.connect(str(db_path)) + try: + conn.enable_load_extension(True) + conn.load_extension(str(cjk_so)) + conn.executescript(SCHEMA_SQL) + conn.executescript(""" + DROP TABLE IF EXISTS messages_fts; + DROP TABLE IF EXISTS messages_fts_trigram; + DROP VIEW IF EXISTS messages_fts_trigram_src; + CREATE VIRTUAL TABLE messages_fts USING fts5(content); + CREATE TRIGGER messages_fts_insert AFTER INSERT ON messages BEGIN + INSERT INTO messages_fts(rowid, content) VALUES (new.id, COALESCE(new.content,'')); + END; + """) + # Established cjk index: tables + triggers, no backfill markers (complete). + conn.executescript(FTS_CJK_TABLE_SQL) + conn.executescript(FTS_CJK_TRIGGER_SQL) + conn.execute("DELETE FROM schema_version") + conn.execute("INSERT INTO schema_version (version) VALUES (10)") + conn.execute( + "INSERT INTO sessions (id, source, started_at) VALUES ('s1', 'cli', ?)", + (_time.time(),), + ) + for role, content in ( + ("user", "레거시 일본 메시지"), + ("assistant", "legacy english reply"), + ): + conn.execute( + "INSERT INTO messages (session_id, timestamp, role, content) " + "VALUES ('s1', ?, ?, ?)", + (_time.time(), role, content), + ) + conn.commit() + finally: + conn.close() + + d = SessionDB(db_path=db_path) + try: + assert d.fts_optimize_available(), "legacy inline layout must be eligible" + assert d._fts_cjk_loaded + with d._lock: + cjk_triggers = d._conn.execute( + "SELECT COUNT(*) FROM sqlite_master WHERE type = 'trigger' " + "AND name LIKE 'messages_fts_cjk_%'" + ).fetchone()[0] + assert cjk_triggers == 3, "cjk index is established and trigger-live" + result = d.optimize_fts_storage(vacuum=False) + assert result["ok"] + # The cjk index survives demote untouched: same tables, same service path. + assert d._fts_cjk_available + assert d.fts_cjk_rebuild_status() is None + assert d._describe_search_path("일본") == "fts_cjk" + assert d.search_messages("일본", limit=10) + with d._lock: + trash_cjk = d._conn.execute( + "SELECT COUNT(*) FROM sqlite_master " + "WHERE name LIKE 'fts_v22_trash_messages_fts_cjk%'" + ).fetchone()[0] + assert trash_cjk == 0, "no cjk table may be renamed into the trash family" + finally: + d.close() + + def test_pure_latin_embedded_in_cjk_recovered_via_cjk_index(db): """#54242 residual: a pure-Latin query for a token embedded in CJK text (no whitespace) misses on unicode61; with the cjk index available the From c816bfab20fe77419a7f2677fb0d90f72bab22f1 Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Sat, 5 Sep 2026 21:27:33 +0800 Subject: [PATCH 06/98] fix(state): escape underscores in the cjk family exclusion LIKE pattern Review feedback on #103657: the sibling clauses in the same statement declare ESCAPE, and the trash enumeration two blocks up escapes its underscores via replace. Use the same escaped ESCAPE form so every underscore in the statement is a literal match instead of a single-character wildcard. Behavior on the fixed schema is unchanged (same enumeration split); this is consistency plus defense against future lookalike table names. Also reword the exclusion comment to the verified failure mechanism: fts5 xRename renames the whole shadow family in one step, so sweeping the cjk vtable aborts the loop on the next shadow entry and drags the _config table (read by the vtable constructor) into the trash family. --- hermes_state_search.py | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/hermes_state_search.py b/hermes_state_search.py index 3cd3427862..5986a4f840 100644 --- a/hermes_state_search.py +++ b/hermes_state_search.py @@ -532,9 +532,11 @@ class SessionSearchMixin: "AND (name LIKE 'messages_fts_%' ESCAPE '\\' " "OR name LIKE 'messages_fts_trigram_%' ESCAPE '\\') " # messages_fts_cjk* is an independent v23+ index, not part of the - # demoted legacy layout: renaming it (or its shadow tables) breaks - # the vtable constructor chain (#103647). - "AND name NOT LIKE 'messages_fts_cjk%'" + # demoted legacy layout: fts5's xRename renames the entire shadow + # family in one step, so sweeping the cjk vtable here aborts the + # loop on the next cjk shadow entry and drags _config — needed by + # the vtable constructor — into the trash family (#103647). + "AND name NOT LIKE 'messages\\_fts\\_cjk%' ESCAPE '\\'" ).fetchall(): conn.execute(f"ALTER TABLE {row[0]} RENAME TO fts_v22_trash_{row[0]}") # Claim the backfill BEFORE the empty v23 tables exist so a crash before From 10777823fe801ee35eb59158d281ccd9a476f74b Mon Sep 17 00:00:00 2001 From: TaoMasterCoder Date: Fri, 11 Sep 2026 02:38:55 -0700 Subject: [PATCH 07/98] fix(sessions): salvage a page-1-header-damaged state.db in the lost_and_found lane MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A SIGKILL mid-write (OOM killer, #106667) can leave state.db with a garbage page-1 header. SQLite refuses the file outright ("file is not a database", SQLITE_NOTADB) and the sqlite3 shell's .recover opens the file like any other client, so the lost_and_found lane failed with the same rc=26 twice although every data page after the header survived. The #106587 quarantine now preserves such a file as state.db.notadb--.bak; this makes that preserved file recoverable with `hermes sessions recover --source --allow-partial`. When both .recover attempts fail with "not a database", zero the 100-byte header of the lane's private snapshot copy and rerun them. .recover trips only on the magic check and infers page size and layout from the pages themselves, so a zeroed header is enough; a spliced donor header (the PR's original mechanism) instead advertises a database size / freelist that contradicts the file and yields "database disk image is malformed" on a direct open — verified live on a 139-page fixture, which also showed the zeroed header recovers 60/60 sessions and 300/300 messages whether the damage covers 100 bytes or the whole first page. The user's file is never written; the report carries `sqlite3_cli.header_zeroed` and a warning about the WAL boundary. Live repro (sqlite3 shell 3.53.1 on PATH, header overwritten with random bytes): BEFORE "page-level .recover salvage failed: ... file is not a database (26)"; AFTER "Recovered 60 sessions and 300 messages", source md5 unchanged. Salvaged from PR #102808 (intent; trimmed from 302 to ~40 source LOC by dropping the donor-header/page-size sweep and the redundant preopen probe — the shell's own refusal is the detector). Independent review on the PR by @strzhao. Refs #106667 Refs #106587 Reported-by: TaoMasterCoder Cross-referenced-by: kshitijk4poor --- contributors/emails/devops@77hub.com | 2 + hermes_cli/session_lost_and_found.py | 42 ++++++++++++---- hermes_cli/session_recovery.py | 6 +++ .../test_session_recovery_lost_and_found.py | 48 +++++++++++++++++++ 4 files changed, 90 insertions(+), 8 deletions(-) create mode 100644 contributors/emails/devops@77hub.com diff --git a/contributors/emails/devops@77hub.com b/contributors/emails/devops@77hub.com new file mode 100644 index 0000000000..5c1f36eb5b --- /dev/null +++ b/contributors/emails/devops@77hub.com @@ -0,0 +1,2 @@ +TaoMasterCoder +# PR #102808 salvage (header-damaged state.db in lost_and_found lane; #106667) diff --git a/hermes_cli/session_lost_and_found.py b/hermes_cli/session_lost_and_found.py index ee2f36cd66..d915532c24 100644 --- a/hermes_cli/session_lost_and_found.py +++ b/hermes_cli/session_lost_and_found.py @@ -180,10 +180,41 @@ def _cli_supports_recover(binary: str) -> bool: shutil.rmtree(scratch_dir, ignore_errors=True) +SQLITE_HEADER_LENGTH = 100 + + def run_cli_lost_and_found_recover( source: Path, lf_path: Path, sqlite3_bin: str, *, timeout: float = 3600.0, ) -> dict[str, Any]: - """Run ``sqlite3 .recover`` streamed into a fresh scratch DB.""" + """Run ``sqlite3 .recover`` streamed into a fresh scratch DB. + + A file whose page-1 header is garbage (SIGKILL mid-write) is refused outright by the shell + (``file is not a database``, rc 26) although the data pages after it survive. ``.recover`` + walks pages via sqlite_dbpage and only trips on the magic check, so on that refusal the + 100-byte header of the private snapshot is zeroed and the attempts rerun; a zeroed header + makes .recover infer page size and layout from the pages themselves (a spliced donor header + would instead report a database size/freelist that contradicts the file). ``source`` is + the caller's snapshot copy, never the user's file (#106667). + """ + attempts = _cli_recover_attempts(source, lf_path, sqlite3_bin, timeout=timeout) + if attempts[-1]["usable"]: + return {"binary": sqlite3_bin, "attempts": attempts} + if any("not a database" in a["dump_stderr_tail"] for a in attempts): + with source.open("r+b") as handle: + handle.write(bytes(SQLITE_HEADER_LENGTH)) + attempts += _cli_recover_attempts(source, lf_path, sqlite3_bin, timeout=timeout) + if attempts[-1]["usable"]: + return {"binary": sqlite3_bin, "attempts": attempts, "header_zeroed": True} + details = "; ".join( + f"[{a['command']}] dump rc={a['dump_returncode']} load rc={a['load_returncode']} " + f"{a['dump_stderr_tail'] or a['load_stderr_tail']}".strip() + for a in attempts + ) + raise LostAndFoundError(f"sqlite3 .recover did not produce a usable lost_and_found database: {details}") + + +def _cli_recover_attempts(source: Path, lf_path: Path, sqlite3_bin: str, *, timeout: float) -> list[dict[str, Any]]: + """``--ignore-freelist`` first (no resurrected deleted rows), plain ``.recover`` for older shells.""" attempts: list[dict[str, Any]] = [] for command in (".recover --ignore-freelist", ".recover"): if lf_path.exists(): @@ -211,13 +242,8 @@ def run_cli_lost_and_found_recover( "usable": _lost_and_found_db_usable(lf_path), }) if attempts[-1]["usable"]: - return {"binary": sqlite3_bin, "attempts": attempts} - details = "; ".join( - f"[{a['command']}] dump rc={a['dump_returncode']} load rc={a['load_returncode']} " - f"{a['dump_stderr_tail'] or a['load_stderr_tail']}".strip() - for a in attempts - ) - raise LostAndFoundError(f"sqlite3 .recover did not produce a usable lost_and_found database: {details}") + break + return attempts def _lost_and_found_db_usable(lf_path: Path) -> bool: diff --git a/hermes_cli/session_recovery.py b/hermes_cli/session_recovery.py index 9659beb9fd..6dae877432 100644 --- a/hermes_cli/session_recovery.py +++ b/hermes_cli/session_recovery.py @@ -1040,6 +1040,12 @@ def _recover_via_lost_and_found( "BEST-EFFORT page-level salvage: the source table schemas were unreadable, so rows were rebuilt from raw " "pages and mapped heuristically. Review every count before trusting this output." ) + if cli_report.get("header_zeroed"): + verification["warnings"].append( + "header salvage: SQLite refused the source outright (page-1 header damaged, 'file is not a " + "database'); the header of the private snapshot copy was zeroed so .recover could walk the " + "surviving pages. Rows written only to a -wal after the last checkpoint are not included." + ) verification.update(loss_detected=True, complete=False) # Structural checks cannot see a positional mis-mapping: every row still inserts, so integrity/FK/FTS # stay green. A systematic timestamp violation is the semantic tell — never report such a salvage as verified. diff --git a/tests/hermes_cli/test_session_recovery_lost_and_found.py b/tests/hermes_cli/test_session_recovery_lost_and_found.py index a8711dde86..0046804c1d 100644 --- a/tests/hermes_cli/test_session_recovery_lost_and_found.py +++ b/tests/hermes_cli/test_session_recovery_lost_and_found.py @@ -305,6 +305,54 @@ def test_lost_and_found_lane_recovers_schema_unreadable_source( recovered_db.close() + +@pytest.mark.skipif( + not HAVE_SQLITE3_CLI, + reason="sqlite3 CLI not on PATH; .recover is a shell-only feature", +) +def test_lost_and_found_lane_recovers_page1_header_damaged_source(tmp_path: Path) -> None: + """#106667: a garbage page-1 header makes SQLite (and the shell's .recover) refuse the file + with 'file is not a database' although every data page survives. The lane must still + salvage the rows, and must do it on its snapshot — the user's file stays byte-identical.""" + source = tmp_path / "header-damaged.db" + output = tmp_path / "header-recovered.db" + db = SessionDB(db_path=source) + try: + for session_number in range(3): + session_id = f"hdr-session-{session_number}" + db.create_session(session_id, "cli", cwd="/tmp/hdr") + for message_number in range(9): + db.append_message(session_id, "user", f"payload {session_number} {message_number}") + finally: + db.close() + conn = sqlite3.connect(str(source), isolation_level=None) + try: + conn.execute("PRAGMA wal_checkpoint(TRUNCATE)") + conn.execute("PRAGMA journal_mode=DELETE") + finally: + conn.close() + data = bytearray(source.read_bytes()) + data[0:100] = bytes(range(1, 101)) # not the magic, not zeroes: the incident shape + source.write_bytes(data) + damaged_bytes = source.read_bytes() + + with pytest.raises(sqlite3.DatabaseError, match="not a database"): + sqlite3.connect(str(source)).execute("SELECT count(*) FROM sqlite_master").fetchone() + + report = recover_session_database(source, output, work_dir=tmp_path, allow_partial=True) + + assert report["mode"] == "lost_and_found_salvage" + assert report["sqlite3_cli"]["header_zeroed"] is True + assert any("header salvage" in warning for warning in report["verification"]["warnings"]) + assert source.read_bytes() == damaged_bytes + conn = sqlite3.connect(str(output)) + try: + assert conn.execute("SELECT COUNT(*) FROM sessions").fetchone()[0] == 3 + assert conn.execute("SELECT COUNT(*) FROM messages").fetchone()[0] == 27 + finally: + conn.close() + + # ── mapper unit tests (no sqlite3 CLI required) ───────────────────────────── From 17b2768b86300485ce95355b151f40fe5754afee Mon Sep 17 00:00:00 2001 From: nftpoetrist <264138787+nftpoetrist@users.noreply.github.com> Date: Thu, 10 Sep 2026 00:10:38 +0300 Subject: [PATCH 08/98] fix(state): quarantine a corrupt/replaced handle out of rebuild_fts() too MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit optimize_fts() and vacuum() already refuse to run against a quarantined handle (_db_corrupt / _db_replaced / _db_wal_generation_lost): both would rewrite index/file pages in place, turning contained, diagnosable corruption into an amplified one. rebuild_fts() never got the same guard, despite being the more destructive of the two ("discards and recreates the index data entirely", per its own docstring, vs. optimize_fts's segment merge). It's also independently reachable outside _execute_write's own quarantine check: gateway/session_transcript.py's _rebuild_fts_once() calls db.rebuild_fts() directly from the FTS-corruption transcript-retry path, with no quarantine check of its own (only a WAL split-brain / foreign-holder check, a different concern). A quarantined handle hitting that retry path would run a full FTS rebuild — and commit it — on a corrupt, replaced, or split-WAL-generation file. Add the same self._raise_if_db_corrupt()/self._raise_if_db_replaced() pair optimize_fts() already has, at the top of rebuild_fts(), before it enters the cross-process rebuild admission. --- hermes_state_search.py | 5 +++- .../test_state_db_corrupt_quarantine.py | 23 +++++++++++++++++++ 2 files changed, 27 insertions(+), 1 deletion(-) diff --git a/hermes_state_search.py b/hermes_state_search.py index 5986a4f840..a96c7c9fe1 100644 --- a/hermes_state_search.py +++ b/hermes_state_search.py @@ -1216,7 +1216,8 @@ class SessionSearchMixin: Uses the FTS5 ``'rebuild'`` command, which rewrites the internal b-tree segments from the content rows. Unlike ``optimize_fts`` (which merges existing segments), ``rebuild`` discards and recreates - the index data entirely. See #50502. + the index data entirely — the more destructive of the two, so it is quarantined the same way. See + #50502. A full structural rebuild must never run concurrently in two processes sharing one state.db — that interleaving has structurally corrupted the database in production (PR #93200) — so this admits through the cross-process ``fts_rebuild_admission`` authority and FAILS CLOSED: if another process @@ -1225,6 +1226,8 @@ class SessionSearchMixin: path, which retries in-process from the gateway housekeeping tick (``retry_deferred_fts_recovery``) and at next startup. """ + self._raise_if_db_corrupt() + self._raise_if_db_replaced() rebuilt = 0 with fts_rebuild_admission(self.db_path) as admitted: if not admitted: diff --git a/tests/hermes_state/test_state_db_corrupt_quarantine.py b/tests/hermes_state/test_state_db_corrupt_quarantine.py index 05475ba3aa..4228abf660 100644 --- a/tests/hermes_state/test_state_db_corrupt_quarantine.py +++ b/tests/hermes_state/test_state_db_corrupt_quarantine.py @@ -307,3 +307,26 @@ class TestVacuumAndMaintenanceRespectQuarantine: _clear_flag(db, flag_name) db._conn = real_conn db.close() + + @pytest.mark.parametrize("flag_name,expected_exc", _QUARANTINE_FLAGS) + def test_rebuild_fts_refuses_when_quarantined(self, tmp_path, flag_name, expected_exc): + """rebuild_fts() is reachable outside _execute_write's own quarantine check — the gateway's + FTS-corruption transcript-retry path (gateway/session_transcript.py::_rebuild_fts_once) + calls it directly. Unlike optimize_fts ("merges existing segments"), rebuild_fts "discards + and recreates the index data entirely" — strictly more destructive — so it must refuse at + least as eagerly.""" + db = SessionDB(db_path=tmp_path / "state.db") + real_conn = db._conn + try: + db.create_session(session_id="s1", source="cli", model="test") + db.append_message("s1", role="user", content="hello world") + recorder = _RecordingConn(real_conn) + db._conn = recorder + _force_flag(db, flag_name) + with pytest.raises(expected_exc): + db.rebuild_fts() + assert recorder.recorded == [] + finally: + _clear_flag(db, flag_name) + db._conn = real_conn + db.close() From 38adfe90a4e6b995678bba2ffdebf92804e64c5c Mon Sep 17 00:00:00 2001 From: Sulthan Zahran Date: Fri, 11 Sep 2026 02:49:25 -0700 Subject: [PATCH 09/98] fix(state): classify FTS-scoped corruption as fts_index, never whole-file damage MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit classify_persistence_error bucketed every _DB_CORRUPTION_MARKERS hit as "corrupt", so an error SQLite itself scoped to the FTS5 index layer (SQLITE_CORRUPT_VTAB, or an `fts5: corrupt structure record for table "messages_fts"` report) that escaped the write path — the detach in _enter_fts_fail_open refused (generation/lock check), or a read/search path with no fail-open at all — reached the turn boundary and the gateway startup notice as structural corruption: the turn ended with `.recover` / restore-backup advice on a file whose canonical tables were provably healthy. One provenance rule, hermes_state_errors.is_fts_scoped_corruption_error, now feeds both the write-repair gate (SessionDB._is_fts_write_corruption_error delegates to it, so the gateway transcript retry inherits it) and the classifier: a known result code outranks prose (only SQLITE_CORRUPT_VTAB is FTS-scoped; bare SQLITE_CORRUPT/NOTADB and any contradictory code fail closed), and without a code the text must both carry a corruption marker and name a messages_fts* object. The new "fts_index" cause renders index-scoped guidance (doctor --fix / restart, do not run recovery) in the turn explainer and the home-channel notice. The structural fail-close is untouched: bare malformed / not-a-database still quarantine and still classify "corrupt". Salvaged from PR #97843 (SulthanZahran1), trimmed: the quick_check-backed "corrupt_unconfirmed" tier is dropped — on a live handle that just observed an unscoped SQLITE_CORRUPT, PRAGMA quick_check on a damaged shadow b-tree raises rather than reports on 3.53.1, so the probe could never downgrade the exact shape it was built for, and a verdict that softens quarantine guidance on prose alone weakens the fail-close. #97841 (Finn763) reached the same fts_index cause via text markers only; its LIKE-degradation intent already lives in _search_messages_impl (_fts_stale). Fixes #97794 Co-authored-by: finn763 <165816600+finn763@users.noreply.github.com> --- agent/turn_explainers.py | 10 ++ contributors/emails/zsulthan9@gmail.com | 2 + gateway/run_notifications.py | 12 +- hermes_state_errors.py | 47 ++++++- hermes_state_fts.py | 13 +- .../test_turn_completion_explainer.py | 78 ++++++++++ tests/state/test_fts_index_fail_open.py | 133 ++++++++++++++++++ 7 files changed, 283 insertions(+), 12 deletions(-) create mode 100644 contributors/emails/zsulthan9@gmail.com create mode 100644 tests/state/test_fts_index_fail_open.py diff --git a/agent/turn_explainers.py b/agent/turn_explainers.py index 0ee38115e3..feda8388cf 100644 --- a/agent/turn_explainers.py +++ b/agent/turn_explainers.py @@ -151,6 +151,16 @@ _PERSISTENCE_CAUSE_EXPLANATIONS: Dict[str, str] = { "3. Restore from a backup in {backups_dir}/\n" "Then send your message again." ), + # SQLite scoped the corruption to the FTS index and the derived indexes could not be + # detached, so this write did not land; the message store itself is intact (#97794). + "fts_index": ( + "the turn was stopped because the session search index (FTS5) " + "is corrupt and could not be detached, so this message was not " + "saved. The message store itself is not damaged: do not run " + "recovery tools or restore a backup. Run `hermes doctor --fix` " + "(or restart Hermes, which repairs the index on open), then " + "send your message again." + ), "disk": ( "the turn was stopped because session storage could not " "be written (the transcript would have been lost on " diff --git a/contributors/emails/zsulthan9@gmail.com b/contributors/emails/zsulthan9@gmail.com new file mode 100644 index 0000000000..62ff403726 --- /dev/null +++ b/contributors/emails/zsulthan9@gmail.com @@ -0,0 +1,2 @@ +SulthanZahran1 +# PR #97843 salvage diff --git a/gateway/run_notifications.py b/gateway/run_notifications.py index 38d69aa422..65e89351e6 100644 --- a/gateway/run_notifications.py +++ b/gateway/run_notifications.py @@ -835,7 +835,8 @@ class GatewayNotificationsMixin: return from hermes_constants import get_default_hermes_root from hermes_state import _default_db_path, classify_persistence_error, format_session_db_unavailable - if classify_persistence_error(error) == "corrupt": + cause = classify_persistence_error(error) + if cause == "corrupt": # Copy-pasteable, so name the real store (profiles / HERMES_HOME do not live under ~/.hermes). db_path = _default_db_path() backups_dir = get_default_hermes_root() / "backups" @@ -854,6 +855,15 @@ class GatewayNotificationsMixin: f"3. Restore from a backup in {backups_dir}/\n" "Run `hermes doctor` for sanitized diagnostics." ) + elif cause == "fts_index": + # Index-scoped corruption: the message tables are not damaged, so the recover / + # restore advice above would be destructive on a healthy file (#97794). + message = ( + "⚠️ Session database reported a corruption error confined to the search index " + "(FTS5); the message tables are not damaged. Messages may not be persisted until " + "it is repaired: run `hermes doctor --fix`, then restart the gateway. Do not run " + "recovery tools or restore a backup unless `hermes doctor` confirms damage." + ) else: message = ( f"⚠️ Session database unavailable — messages may not be persisted. " diff --git a/hermes_state_errors.py b/hermes_state_errors.py index c1523c377e..6856610439 100644 --- a/hermes_state_errors.py +++ b/hermes_state_errors.py @@ -3,6 +3,7 @@ Shared by hermes_state and its mixins; string predicates match wrapped RPC strings as well as live sqlite3 exceptions.""" import errno +import re import sqlite3 # Malformed schema: ``sqlite_master`` itself is inconsistent (typically a DUPLICATE @@ -74,8 +75,8 @@ def is_disk_full_error(exc: BaseException | str | None) -> bool: # Every classify_persistence_error bucket; consumers enumerate this tuple. PERSISTENCE_ERROR_CAUSES = ( - "locked", "compression", "compression_closed", "turn_lease", "corrupt", "replaced", - "deleted_wal", "disk", "unknown", + "locked", "compression", "compression_closed", "turn_lease", "corrupt", "fts_index", + "replaced", "deleted_wal", "disk", "unknown", ) @@ -88,6 +89,38 @@ _DB_CORRUPTION_MARKERS = ( "malformed", "file is not a database", "not a database", "database corruption", ) +# The module constant exists on Python 3.11+; the numeric value is stable across SQLite releases. +SQLITE_CORRUPT_VTAB = getattr(sqlite3, "SQLITE_CORRUPT_VTAB", 267) + +# Every FTS object hangs off this prefix: the virtual tables and their _data/_idx/_content/ +# _docsize/_config shadow b-trees. FTS5 names the table in its own corruption reports. +_FTS_OBJECT_RE = re.compile(r"\bmessages_fts\w*") + + +def is_fts_scoped_corruption_error(exc_or_str) -> bool: + """Corruption SQLite itself attributes to the FTS index layer: the ONE provenance rule + shared by the write-repair gate (``SessionDB._is_fts_write_corruption_error``), the + gateway transcript retry and :func:`classify_persistence_error` (#96038, #97794). + + A known result code outranks prose: ``SQLITE_CORRUPT_VTAB`` is FTS-scoped even with + the generic malformed-image text older SQLite builds emit, while bare ``SQLITE_CORRUPT`` + / ``SQLITE_NOTADB`` carry no object scope and any other known code contradicts + FTS-looking prose, so both fail closed. Only without a code (Python < 3.11, RPC-wrapped + strings) does the text decide, and then only an ``fts5:`` corruption report or a + corruption marker that names a ``messages_fts*`` object counts. + """ + if exc_or_str is None: + return False + code = getattr(exc_or_str, "sqlite_errorcode", None) + if code is not None: + return code == SQLITE_CORRUPT_VTAB + text = (exc_or_str if isinstance(exc_or_str, str) else str(exc_or_str)).lower() + if not _FTS_OBJECT_RE.search(text): + return False + if text.startswith("fts5:") and "corrupt" in text: + return True + return any(marker in text for marker in _DB_CORRUPTION_MARKERS) + class CompressionSessionClosedError(RuntimeError): """A durable write targeted a parent already closed by compression.""" @@ -205,8 +238,10 @@ def classify_persistence_error(exc_or_str) -> str: matches: "locked" = busy, retry; "disk" = full/read-only/permissions; "compression" = a live lease refused the write; "compression_closed" = adopt the rotated session id; "turn_lease" = fencing, not storage; "corrupt" = - file damage (repair path, not disk space); "replaced" = main-file replacement; - "deleted_wal" = a retired sidecar generation requiring capture inspection.""" + file damage (repair path, not disk space); "fts_index" = SQLite scoped the + corruption to the FTS index (the transcript store is not damaged); "replaced" = + main-file replacement; "deleted_wal" = a retired sidecar generation requiring + capture inspection.""" if exc_or_str is None: return "unknown" # Lease refusals contain neither "locked" nor "busy": match by type first, @@ -216,6 +251,10 @@ def classify_persistence_error(exc_or_str) -> str: for exc_type, cause in _PERSISTENCE_CAUSE_BY_TYPE: if isinstance(exc_or_str, exc_type): return cause + # Provenance before prose: an FTS-scoped result code (or, without one, an fts5 report + # naming messages_fts*) is index damage, never whole-file corruption (#97794). + if is_fts_scoped_corruption_error(exc_or_str): + return "fts_index" text = str(exc_or_str).lower() for markers, cause in _PERSISTENCE_CAUSE_BY_PHRASE: if any(marker in text for marker in markers): diff --git a/hermes_state_fts.py b/hermes_state_fts.py index 33815f73a0..bde71ef790 100644 --- a/hermes_state_fts.py +++ b/hermes_state_fts.py @@ -10,6 +10,7 @@ from typing import Sequence from hermes_constants import get_hermes_home from hermes_state_common import FTS_CJK_STALE_KEY, FTS_STALE_KEY, _FTS_CJK_TRIGGERS, _FTS_TRIGGERS +from hermes_state_errors import is_fts_scoped_corruption_error # caplog tests pin the "hermes_state" logger name. logger = logging.getLogger("hermes_state") @@ -341,13 +342,11 @@ class SessionFtsSetupMixin: @staticmethod def _is_fts_write_corruption_error(exc: sqlite3.DatabaseError) -> bool: - """Corruption SQLite identifies as FTS-scoped (SQLITE_CORRUPT_VTAB, or an - ``fts5:`` message on older builds); a bare malformed image is structural.""" - error_code = getattr(exc, "sqlite_errorcode", None) - if error_code is not None: - return error_code == getattr(sqlite3, "SQLITE_CORRUPT_VTAB", 267) - msg = str(exc).lower() - return msg.startswith("fts5:") and "corrupt structure" in msg + """Corruption SQLite identifies as FTS-scoped (SQLITE_CORRUPT_VTAB, or an ``fts5:`` + report naming ``messages_fts*`` on builds without result codes); a bare malformed + image is structural. One rule, shared with ``classify_persistence_error`` and the + gateway transcript retry: see :func:`hermes_state_errors.is_fts_scoped_corruption_error`.""" + return is_fts_scoped_corruption_error(exc) def _enter_fts_fail_open(self, exc: sqlite3.DatabaseError) -> bool: """Detach corrupt FTS indexes so canonical writes can continue. Breadcrumb + diff --git a/tests/run_agent/test_turn_completion_explainer.py b/tests/run_agent/test_turn_completion_explainer.py index a4b624e401..6137423291 100644 --- a/tests/run_agent/test_turn_completion_explainer.py +++ b/tests/run_agent/test_turn_completion_explainer.py @@ -173,6 +173,26 @@ def test_explanation_persistence_corrupt_backups_dir_follows_hermes_home(monkeyp assert "{backups_dir}" not in out +def test_explanation_persistence_fts_index_never_advises_recovery(): + """#97794: an FTS-scoped failure must never send the user down the recover / + restore-backup path on a healthy file, and must not claim the transcript was lost.""" + out = AIAgent._format_turn_completion_explanation( + "session_persistence_failed", "fts_index" + ) + lower = out.lower() + assert out.strip() != "" + assert "sessions recover" not in lower + assert ".recover" not in lower + # Negative advice ("do not ... restore a backup") is fine; instructions are not. + assert "recovery options" not in lower + assert "restore from a backup" not in lower and "backups/" not in lower + assert "would have been lost" not in lower + assert "free" not in lower # never disk-space advice + assert "hermes doctor" in lower + assert "search index" in lower and "not damaged" in lower + assert "send your message again" in lower # the handle stays live + + def test_explanation_persistence_replaced_cause_forbids_inplace_repair(): out = AIAgent._format_turn_completion_explanation( "session_persistence_failed", "replaced" @@ -350,6 +370,7 @@ def test_persistence_error_causes_tuple_matches_classifier(): "Session 'abc' is being compressed by another writer", "Session turn lease lost; refusing transcript write for 'abc'", "database disk image is malformed", + 'fts5: corrupt structure record for table "messages_fts"', "FATAL: state.db was replaced underneath the gateway", "FATAL: a live process holds a deleted state.db-wal or state.db-shm inode.", "database or disk is full", @@ -360,6 +381,63 @@ def test_persistence_error_causes_tuple_matches_classifier(): assert classify_persistence_error(probe) in PERSISTENCE_ERROR_CAUSES +def test_classify_persistence_error_fts_provenance_order(): + """Result code first, prose only without one — the #96038 rule the write-repair gate + already enforces, now shared with the classifier so there is one definition of + "provably FTS-only" (#97794 review).""" + import sqlite3 + + from hermes_state import SessionDB, classify_persistence_error + from hermes_state_errors import SQLITE_CORRUPT_VTAB, is_fts_scoped_corruption_error + + def _err(text, code=None, cls=sqlite3.DatabaseError): + exc = cls(text) + if code is not None: + exc.sqlite_errorcode = code + return exc + + # Tier 1 — a known result code decides. SQLITE_CORRUPT_VTAB is FTS-scoped even with the + # generic text older SQLite builds emit; bare SQLITE_CORRUPT / SQLITE_NOTADB are unscoped. + vtab = _err("database disk image is malformed", SQLITE_CORRUPT_VTAB) + assert classify_persistence_error(vtab) == "fts_index" + assert SessionDB._is_fts_write_corruption_error(vtab) # same verdict as the write gate + assert classify_persistence_error( + _err("database disk image is malformed", sqlite3.SQLITE_CORRUPT) + ) == "corrupt" + assert classify_persistence_error( + _err("file is not a database", sqlite3.SQLITE_NOTADB) + ) == "corrupt" + # A contradictory known code outranks FTS-looking prose (the #96038 regression shape). + contradictory = _err( + 'fts5: corrupt structure record for table "messages_fts"', + sqlite3.SQLITE_CONSTRAINT_TRIGGER, + sqlite3.IntegrityError, + ) + assert not is_fts_scoped_corruption_error(contradictory) + assert classify_persistence_error(contradictory) != "fts_index" + assert not SessionDB._is_fts_write_corruption_error(contradictory) + + # Tier 2 — no code (Python < 3.11, RPC-wrapped strings): the report must name a + # messages_fts* object. Both shapes from #97794's evidence logs qualify. + assert classify_persistence_error( + _err('fts5: corrupt structure record for table "messages_fts"') + ) == "fts_index" + assert classify_persistence_error( + 'fts5: corruption found reading blob 2061584302081 from table "messages_fts"' + ) == "fts_index" + assert classify_persistence_error( + 'fts5: corrupt structure record for table "messages_fts_trigram"' + ) == "fts_index" + assert classify_persistence_error( + "malformed inverted index for FTS5 table main.messages_fts" + ) == "fts_index" + # Generic markers without provenance stay conservative; fts5 text without corruption, + # or an FTS name without a corruption marker, is not corruption at all. + assert classify_persistence_error("database disk image is malformed") == "corrupt" + assert classify_persistence_error('fts5: syntax error near "x"') == "unknown" + assert classify_persistence_error("no such table: messages_fts") == "unknown" + + # -------------------------------------------------------------------------- # 2. Enable/disable seam # -------------------------------------------------------------------------- diff --git a/tests/state/test_fts_index_fail_open.py b/tests/state/test_fts_index_fail_open.py new file mode 100644 index 0000000000..834d5e49ba --- /dev/null +++ b/tests/state/test_fts_index_fail_open.py @@ -0,0 +1,133 @@ +"""Regression tests for #97794: an FTS5-index-only failure must not kill the turn, and an +FTS-scoped error that escapes must not be rendered as whole-file damage. + +The write path already fails open on provenance-proven FTS corruption (detach the derived +indexes, retry the canonical write) and quarantines the handle on unscoped corruption +(#97940 / #90837). These tests pin the contract at the boundaries the issue was filed +against — the agent flush whose failure ends the turn, and the cause that drives the +user-facing guidance: + +* the flush succeeds after an FTS-only stomp and the exact user message is durable in + ``messages`` (the turn proceeds); +* an FTS-scoped error that still escapes (detach refused) classifies as ``fts_index`` and + never quarantines the handle; +""" + +import sqlite3 +from types import SimpleNamespace + +import pytest + +from hermes_state import SessionDB +from run_agent import AIAgent + + +def _flush_agent(db, session_id): + """Bind the real flush methods onto a stand-in over a live SessionDB.""" + agent = SimpleNamespace( + _session_db=db, + _session_db_created=True, + _persist_disabled=False, + session_id=session_id, + _session_persist_lock=None, + _flushed_db_message_ids=set(), + _flushed_db_message_session_id=None, + _last_flushed_db_idx=0, + _db_flush_scan_prefix=None, + _persist_user_message_idx=None, + _persist_user_message_override=None, + _persist_user_message_timestamp=None, + _pending_cli_user_message=None, + _active_session_turn_lease_holder=None, + _last_persistence_error_cause=None, + _compression_adoption_failed=False, + ) + agent._ensure_db_session = lambda: None + agent._flush_messages_to_session_db = ( + AIAgent._flush_messages_to_session_db.__get__(agent, AIAgent) + ) + agent._flush_messages_to_session_db_unlocked = ( + AIAgent._flush_messages_to_session_db_unlocked.__get__(agent, AIAgent) + ) + return agent + + +def _seed(db, rows=60): + if not db._fts_enabled: + pytest.skip("FTS5 unavailable in this build") + db.create_session("s1", source="cli") + for i in range(rows): + db.append_message("s1", "user", f"seed row {i} " + "z" * 200) + + +def _stomp_fts_shadow(db_path): + """Overwrite the messages_fts shadow b-tree blocks: FTS5 raises SQLITE_CORRUPT_VTAB on the + next MATCH / sync-trigger insert while every canonical row stays intact.""" + raw = sqlite3.connect(str(db_path)) + raw.execute("UPDATE messages_fts_data SET block = X'DEADBEEFDEADBEEFDEADBEEFDEADBEEF'") + raw.commit() + raw.close() + + +def _contents(db_path): + raw = sqlite3.connect(str(db_path)) + try: + return [r[0] for r in raw.execute("SELECT content FROM messages ORDER BY id").fetchall()] + finally: + raw.close() + + +def test_turn_flush_survives_fts_only_corruption(tmp_path): + """The turn's transcript write succeeds after an FTS-only stomp: the flush reports + success (the turn proceeds, no ``session_persistence_failed``) and the exact user + message is durable in ``messages``. The derived indexes are detached, the handle is + not quarantined.""" + db_path = tmp_path / "state.db" + db = SessionDB(db_path=db_path) + try: + _seed(db) + _stomp_fts_shadow(db_path) + agent = _flush_agent(db, "s1") + + ok = agent._flush_messages_to_session_db( + [{"role": "user", "content": "lands after stomp"}], [] + ) + + assert ok is True + assert agent._last_persistence_error_cause is None + assert _contents(db_path)[-1] == "lands after stomp" + assert db._db_corrupt is False + # Builds whose sync trigger walks the stomped structure record detach the derived + # indexes; builds that defer the read pass the insert through untouched. Either way + # the canonical write landed, which is the contract. + assert db._fts_stale in (True, False) + assert db.get_session("s1") is not None + finally: + db.close() + + +def test_escaped_fts_only_error_is_index_scoped_not_quarantined(tmp_path, monkeypatch): + """When the detach itself is refused the FTS-scoped error escapes to the agent. It must + classify as ``fts_index`` (guidance names the index, not the file) and must not + quarantine the handle or touch the derived indexes.""" + db_path = tmp_path / "state.db" + db = SessionDB(db_path=db_path) + try: + _seed(db) + _stomp_fts_shadow(db_path) + monkeypatch.setattr(db, "_enter_fts_fail_open", lambda exc: False) + agent = _flush_agent(db, "s1") + + ok = agent._flush_messages_to_session_db( + [{"role": "user", "content": "refused detach"}], [] + ) + if ok is True: + pytest.skip("this SQLite build defers FTS shadow corruption past the insert trigger") + + assert ok is False + assert agent._last_persistence_error_cause == "fts_index" + assert db._db_corrupt is False + assert db._fts_stale is False + assert "refused detach" not in _contents(db_path) + finally: + db.close() From 3d167a827196197ad8a61d1ddce43f58851f1d65 Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Fri, 11 Sep 2026 03:06:35 -0700 Subject: [PATCH 10/98] fix(doctor): name structural state.db corruption honestly and route it to sessions recover MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `hermes doctor` reported every write-health-probe failure as "state.db FTS write corruption" and `--fix` ran the FTS repair ladder — rebuild, REINDEX, sqlite_master surgery + VACUUM — on the damaged file. When the damage is structural (canonical tables/indexes), none of those rungs can fix it, each one writes to the torn file in place, and the operator is then told to "restore from the backup copy beside state.db": a `.malformed-backup` that is a snapshot of the same corrupt image. `hermes sessions recover`, the tool that actually rebuilds canonical rows into a fresh file, was never mentioned (#88587; the 1.7 GB field incident lost days to it). Discriminate before mutating. hermes_state_repair.integrity_damage_is_structural maps `PRAGMA integrity_check` output onto the file: a `Tree N` id resolved through sqlite_master.rootpage, an index named in `row N missing from index X`, or a `Freelist:` line is structural unless the object is a Hermes-owned messages_fts* table/shadow (full-matched, so a user lookalike such as archive_fts_data is never swept into the rebuildable set). state_db_has_structural_damage runs it read-only on a fresh connection; an integrity_check that RAISES under the walk (torn root page) is structural too — no FTS-only fixture does that while sessions/messages read cleanly. doctor's state check consults it first: structural damage becomes a manual issue naming `hermes [-p ] sessions recover --source --inspect-only` (profile pinned, #105887) and explicitly warning off the .malformed-backup; nothing is mutated and no backup is written. FTS-only damage keeps the existing in-place repair path. Verified against real fixtures: a torn `sessions` root page (before: "FTS write corruption", --fix wrote a 1:1 malformed-backup and failed; after: structural, recover guidance, no writes) and the 16-byte DEADBEEF messages_fts_data stomp (still repaired in place via rebuild_fts). Salvaged from PR #88604 (liuhao1024) onto the split doctor_state.py; the classifier lives beside the repair ladder in hermes_state_repair so the ladder itself can consult it next. Fixes #88587 --- hermes_cli/doctor_state.py | 19 +++- hermes_constants.py | 9 ++ hermes_state_repair.py | 57 ++++++++++++ .../test_doctor_structural_corruption.py | 93 +++++++++++++++++++ 4 files changed, 177 insertions(+), 1 deletion(-) create mode 100644 tests/hermes_cli/test_doctor_structural_corruption.py diff --git a/hermes_cli/doctor_state.py b/hermes_cli/doctor_state.py index 40bb61d0d4..8ec42f2e56 100644 --- a/hermes_cli/doctor_state.py +++ b/hermes_cli/doctor_state.py @@ -162,6 +162,8 @@ def _session_count(state_db_path: Path): # Corruption class -> (ok label, not-fixed label, failed issue, fix hint). ``{count}`` = recovered sessions. +# ``structural`` has no in-place repair: an FTS rebuild cannot fix a canonical b-tree, and the +# ``.malformed-backup`` the repair path would leave beside state.db is a copy of the same damage (#88587). _STATE_DB_REPAIRS = { "fts": ("Repaired state.db FTS write health", "state.db FTS write-health repair did not recover automatically", @@ -172,10 +174,21 @@ _STATE_DB_REPAIRS = { "state.db schema malformed and auto-repair failed — restore from the backup copy beside state.db", "state.db schema malformed — run 'hermes doctor --fix' (or 'hermes sessions repair') to recover hidden sessions"), } +_STATE_DB_STRUCTURAL_ISSUE = ( + "state.db structural corruption (canonical tables/indexes damaged, not the FTS index) — an FTS rebuild " + "cannot repair it. Stop the gateway, then run 'hermes {profile_arg}sessions recover --source {db_path} " + "--inspect-only' and, if it reports recoverable, 'hermes {profile_arg}sessions recover --source {db_path} " + "--output recovered-state.db'. Do NOT restore a .malformed-backup copy beside state.db: it is a snapshot " + "of the same corrupt file." +) def _repair_state_db(f: Finding, should_fix: bool, state_db_path: Path, kind: str) -> None: """Shared --fix path for both state.db corruption classes (FTS write health, malformed schema).""" + if kind == "structural": + from hermes_constants import profile_cli_selector + return f.manual_issues.append(_STATE_DB_STRUCTURAL_ISSUE.format( + profile_arg=profile_cli_selector(), db_path=state_db_path)) ok_label, not_fixed_label, failed_issue, fix_hint = _STATE_DB_REPAIRS[kind] if not should_fix: return f.issues.append(fix_hint) @@ -200,11 +213,15 @@ def _state_db_health(f: Finding, should_fix: bool, state_db_path: Path, _DHH: st check_ok(f"{_DHH}/state.db exists ({_session_count(state_db_path)} sessions)") # COUNT(*) succeeds even when the FTS index is corrupt and every write fails through the triggers; # _db_opens_cleanly drives a rolled-back write to surface that. - from hermes_state_repair import _db_opens_cleanly + from hermes_state_repair import _db_opens_cleanly, state_db_has_structural_damage # `_db_opens_cleanly` now drives a rolled-back write so this otherwise-silent corruption class is # surfaced (and repaired in place with --fix). See #50502. _write_reason = _db_opens_cleanly(state_db_path) if _write_reason is not None: + if state_db_has_structural_damage(state_db_path): + check_warn(f"{_DHH}/state.db has structural corruption (canonical tables/indexes damaged, " + "not the FTS index)", f"({_write_reason})") + return _repair_state_db(f, should_fix, state_db_path, "structural") check_warn(f"{_DHH}/state.db fails a write-health probe (FTS index may be corrupt)", f"({_write_reason})") _repair_state_db(f, should_fix, state_db_path, "fts") except Exception as e: diff --git a/hermes_constants.py b/hermes_constants.py index 96d8331244..1b3c99532c 100644 --- a/hermes_constants.py +++ b/hermes_constants.py @@ -771,6 +771,15 @@ def display_hermes_home() -> str: return str(home) +def profile_cli_selector() -> str: + """``-p `` (trailing space) pinning copy-pasteable ``hermes ...`` guidance to the + active NAMED profile, else ``""``: a bare ``hermes`` follows the sticky ``active_profile`` + file, which can name a different database than the one that failed (#105887). A custom + home outside the profile tree has no selector (only HERMES_HOME names it).""" + name = profile_name_for_home(get_hermes_home()) + return f"-p {name} " if name and name != "default" else "" + + def secure_parent_dir(path: Path) -> None: """Chmod ``0o700`` on *path*'s parent, refusing ``/`` and top-level dirs (misresolved HERMES_HOME).""" parent = path.parent.resolve() diff --git a/hermes_state_repair.py b/hermes_state_repair.py index 50cc8890b0..7062534366 100644 --- a/hermes_state_repair.py +++ b/hermes_state_repair.py @@ -12,6 +12,7 @@ import itertools import json import logging import os +import re import shutil import sqlite3 import stat @@ -684,6 +685,62 @@ def _schema_not_built(exc: BaseException) -> bool: return any(m in str(exc).lower() for m in ("no such table", "no such column")) +# Hermes-owned FTS5 objects: the virtual tables and their shadow b-trees. Full-matched, so a +# user-created lookalike (``archive_fts_data``) is not swept into the rebuildable set. +_FTS_OBJECT_RE = re.compile( + r"messages_fts(_trigram|_cjk)?(_data|_idx|_content|_docsize|_config|_segdir|_segments)?" +) +_INTEGRITY_TREE_RE = re.compile(r"\bTree (\d+)\b") +_INTEGRITY_MISSING_INDEX_RE = re.compile(r"missing from index (\S+)") + + +def integrity_damage_is_structural(integrity_lines, master_rows) -> bool: + """True when any damaged object named by ``PRAGMA integrity_check`` output lies outside + the FTS shadow set: a ``Tree N`` id mapped through ``sqlite_master.rootpage``, an index in + ``row N missing from index X``, or the file's own freelist. Unparseable lines are not + counted (the FTS wording stays, which is incomplete rather than wrong). #88587: an FTS + rebuild cannot repair a canonical b-tree, and the ``.malformed-backup`` it leaves behind + is a snapshot of the same damage.""" + name_by_rootpage = {int(rp): name for rp, _type, name in master_rows if rp} + for line in integrity_lines: + text = str(line) + if text.startswith("Freelist"): + return True + tree = _INTEGRITY_TREE_RE.search(text) + if tree: + name = name_by_rootpage.get(int(tree.group(1)), "") + if name and not _FTS_OBJECT_RE.fullmatch(name): + return True + missing = _INTEGRITY_MISSING_INDEX_RE.search(text) + if missing and not _FTS_OBJECT_RE.fullmatch(missing.group(1)): + return True + return False + + +def state_db_has_structural_damage(db_path: Path) -> bool: + """Read-only ``integrity_check`` + ``sqlite_master`` rootpage map on a fresh connection; + ``integrity_damage_is_structural`` over the result. A check that RAISES instead of + reporting (a torn page under the walk) is structural too: no FTS-only fixture does that + while ``messages``/``sessions`` read cleanly, and the FTS rebuild ladder cannot help. + Cannot-open / locked stays False so the caller keeps the FTS path.""" + try: + conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True, timeout=1.0) + except sqlite3.Error: + return False + try: + master_rows = [tuple(r) for r in conn.execute( + "SELECT rootpage, type, name FROM sqlite_master WHERE rootpage > 0").fetchall()] + lines = [str(r[0]) for r in conn.execute("PRAGMA integrity_check").fetchall()] + except sqlite3.OperationalError: + return False + except sqlite3.DatabaseError: + return True + finally: + conn.close() + return integrity_damage_is_structural( + itertools.chain.from_iterable(line.splitlines() for line in lines), master_rows) + + def _db_opens_cleanly(db_path: Path) -> Optional[str]: """Probe a DB on a fresh connection. Returns None if healthy, else a reason. diff --git a/tests/hermes_cli/test_doctor_structural_corruption.py b/tests/hermes_cli/test_doctor_structural_corruption.py new file mode 100644 index 0000000000..52d822e045 --- /dev/null +++ b/tests/hermes_cli/test_doctor_structural_corruption.py @@ -0,0 +1,93 @@ +"""#88587 — doctor must name STRUCTURAL state.db corruption honestly. + +The write-health probe's failure used to be reported as "FTS write corruption" unconditionally, +routing operators to `--fix` / `sessions repair` (FTS rebuilds that cannot repair canonical-table +damage) and to the .malformed-backup beside the DB (a snapshot of the same corrupt file). The +discriminator maps integrity_check damage through sqlite_master.rootpage and keeps the FTS path +only when every damaged object is a Hermes FTS shadow. +""" + +import contextlib +import io +import sqlite3 + +from hermes_cli.doctor_report import Finding +from hermes_cli.doctor_state import _state_db_health +from hermes_state import SessionDB +from hermes_state_repair import integrity_damage_is_structural, state_db_has_structural_damage + + +def test_integrity_damage_classifier_maps_tree_ids_through_rootpage(): + """Field mappings from #88587: tree 5 -> sessions and tree 15 -> gateway_routing are + structural; a damaged messages_fts shadow tree is not; a lookalike foreign object, + a canonical index named in a "missing from index" line, and the freelist are.""" + fts_only = [ + "Tree 12 page 9: btreeInitPage() returns error code 11", + "row 3 missing from index messages_fts_trigram_idx", + ] + master = [(12, "table", "messages_fts_data"), (5, "table", "sessions"), + (15, "table", "gateway_routing"), (40, "table", "archive_fts_data")] + assert integrity_damage_is_structural(fts_only, master) is False + assert integrity_damage_is_structural(["Tree 5 page 421385: btreeInitPage() returns error code 11"], master) + assert integrity_damage_is_structural(["Tree 15 page 15 cell 0: 2nd reference to page 5453"], master) + assert integrity_damage_is_structural(["Tree 40 page 40: btreeInitPage() returns error code 11"], master) + assert integrity_damage_is_structural(["row 1 missing from index sqlite_autoindex_delivery_obligations_1"], master) + assert integrity_damage_is_structural(["Freelist: invalid page number 167772160"], master) + # Unparseable / unknown-tree lines keep the FTS wording (incomplete, never wrong). + assert integrity_damage_is_structural(["Tree 999 page 1: garbage", "*** in database main ***"], master) is False + + +def _seed(tmp_path, rows=120): + db_path = tmp_path / "state.db" + db = SessionDB(db_path=db_path) + db.create_session("s1", source="cli") + for i in range(rows): + db.append_message("s1", "user", f"hello {i} alpha beta " + "lorem " * 30) + db.close() + raw = sqlite3.connect(db_path) + raw.execute("PRAGMA wal_checkpoint(TRUNCATE)") + page_size = raw.execute("PRAGMA page_size").fetchone()[0] + root = raw.execute("SELECT rootpage FROM sqlite_master WHERE name='sessions'").fetchone()[0] + raw.close() + return db_path, page_size, root + + +def _run_doctor(db_path, should_fix): + finding = Finding() + with contextlib.redirect_stdout(io.StringIO()): + _state_db_health(finding, should_fix, db_path, "~/x") + return finding + + +def test_doctor_routes_structural_damage_to_recover_not_fts_rebuild(tmp_path, monkeypatch): + """Real torn ``sessions`` b-tree: doctor --fix must not run the FTS repair ladder (no + .malformed-backup, nothing fixed) and must point at `hermes sessions recover` for THIS + database with the profile pinned; a real FTS-only stomp still takes the FTS path.""" + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + db_path, page_size, root = _seed(tmp_path) + with open(db_path, "r+b") as f: + f.seek((root - 1) * page_size + 8) + f.write(b"\xff\xff" * 8) + assert state_db_has_structural_damage(db_path) is True + + finding = _run_doctor(db_path, should_fix=True) + assert finding.fixed == 0 and finding.issues == [] + (issue,) = finding.manual_issues + assert "structural" in issue and "sessions recover" in issue and str(db_path) in issue + assert "FTS write corruption" not in issue and "restore from the backup" not in issue + assert not list(tmp_path.glob("state.db.malformed-backup-*")) + + fts_path = tmp_path / "fts" / "state.db" + fts_db = SessionDB(db_path=fts_path) + fts_db.create_session("s1", source="cli") + for i in range(40): + fts_db.append_message("s1", "user", f"hello {i} alpha") + fts_db.close() + raw = sqlite3.connect(fts_path) + raw.execute("UPDATE messages_fts_data SET block = X'DEADBEEFDEADBEEFDEADBEEFDEADBEEF'") + raw.commit() + raw.close() + assert state_db_has_structural_damage(fts_path) is False + fts_finding = _run_doctor(fts_path, should_fix=False) + assert fts_finding.manual_issues == [] + assert any("FTS" in i for i in fts_finding.issues) From 754ecff46614520778df3783064e41ab9e66208f Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 11 Sep 2026 03:07:11 -0700 Subject: [PATCH 11/98] fix(state): pin corrupt-session recovery guidance to the failing profile MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The recovery commands rendered on structural corruption — the turn explainer's `session_persistence_failed`/corrupt body, the gateway's home-channel state.db warning, and hermes_state_repair._persistent_repair_exhausted_error — already interpolate the active profile's state.db path, but every `hermes ...` verb in them was bare. A bare `hermes` follows the sticky `active_profile` file, so an operator running the pasted `hermes doctor --fix` (or `hermes sessions recover` with a relative source) from a named-profile incident could inspect or repair a different profile's database (#105887). hermes_constants.profile_cli_selector() renders `-p ` for a named profile home (default home and custom roots outside the profile tree render nothing: the default is what a bare `hermes` already means, and a custom root is only reachable via HERMES_HOME). Every command in the three guidance sites now carries it, and the new `fts_index` guidance inherits the same interpolation. Live check with HERMES_HOME=/profiles/research and active_profile=other: before `1. Run \`hermes doctor --fix\`` (targets "other"); after `1. Run \`hermes -p research doctor --fix\`` and `hermes -p research sessions recover --source /profiles/research/state.db --inspect-only`. Refs #105887 Reported-by: Cuttingwater --- agent/turn_explainers.py | 20 ++++----- gateway/run_notifications.py | 16 ++++--- hermes_state_repair.py | 9 ++-- .../test_corruption_recovery_guidance.py | 44 +++++++++++++++++++ 4 files changed, 69 insertions(+), 20 deletions(-) diff --git a/agent/turn_explainers.py b/agent/turn_explainers.py index feda8388cf..8d78279a31 100644 --- a/agent/turn_explainers.py +++ b/agent/turn_explainers.py @@ -139,10 +139,10 @@ _PERSISTENCE_CAUSE_EXPLANATIONS: Dict[str, str] = { "reported structural corruption (the transcript would " "have been lost on restart). Freeing disk space will " "not help. Recovery options:\n" - "1. Run `hermes doctor --fix`\n" + "1. Run `hermes {profile_arg}doctor --fix`\n" "2. Stop the gateway, then recover with:\n" - " hermes sessions recover --source {db_path} --inspect-only\n" - " (if it reports recoverable) hermes sessions recover " + " hermes {profile_arg}sessions recover --source {db_path} --inspect-only\n" + " (if it reports recoverable) hermes {profile_arg}sessions recover " "--source {db_path} --output recovered-state.db\n" " — recovery snapshots the damaged file first; do NOT " "run `sqlite3 ... \".recover\"` against the live " @@ -157,7 +157,7 @@ _PERSISTENCE_CAUSE_EXPLANATIONS: Dict[str, str] = { "the turn was stopped because the session search index (FTS5) " "is corrupt and could not be detached, so this message was not " "saved. The message store itself is not damaged: do not run " - "recovery tools or restore a backup. Run `hermes doctor --fix` " + "recovery tools or restore a backup. Run `hermes {profile_arg}doctor --fix` " "(or restart Hermes, which repairs the index on open), then " "send your message again." ), @@ -328,14 +328,14 @@ class TurnExplainersMixin: body = _PERSISTENCE_CAUSE_EXPLANATIONS.get( persistence_cause or "unknown", _PERSISTENCE_DEFAULT_EXPLANATION ) - if persistence_cause == "corrupt": - # Copy-pasteable, so name the store that actually failed: the agent's own - # SessionDB. A multi-profile backend (Desktop serve) hosts sessions whose - # state.db is NOT the process default, so the default would send the operator - # to inspect/repair the wrong profile's database (#105887). - from hermes_constants import get_default_hermes_root + if persistence_cause in ("corrupt", "fts_index"): + # Copy-pasteable, so name the store that actually failed and pin the profile: + # a multi-profile backend (Desktop serve) hosts sessions whose state.db is NOT + # the process default, and a bare `hermes` follows active_profile (#105887). + from hermes_constants import get_default_hermes_root, profile_cli_selector from hermes_state import _default_db_path + body = body.replace("{profile_arg}", profile_cli_selector()) body = body.replace("{db_path}", str(db_path or _default_db_path())) body = body.replace( "{backups_dir}", str(get_default_hermes_root() / "backups") diff --git a/gateway/run_notifications.py b/gateway/run_notifications.py index 65e89351e6..2af9cb78a5 100644 --- a/gateway/run_notifications.py +++ b/gateway/run_notifications.py @@ -833,27 +833,29 @@ class GatewayNotificationsMixin: if not error: logger.info("state.db recovered before the home-channel warning went out; not broadcasting") return - from hermes_constants import get_default_hermes_root + from hermes_constants import get_default_hermes_root, profile_cli_selector from hermes_state import _default_db_path, classify_persistence_error, format_session_db_unavailable cause = classify_persistence_error(error) + # Copy-pasteable, so name the real store and pin the profile: a bare `hermes` follows + # active_profile, which may be a different database (#105887). + profile_arg = profile_cli_selector() if cause == "corrupt": - # Copy-pasteable, so name the real store (profiles / HERMES_HOME do not live under ~/.hermes). db_path = _default_db_path() backups_dir = get_default_hermes_root() / "backups" message = ( "⚠️ Session database corruption detected. Messages may not be " "persisted. Recovery options:\n" - "1. Run `hermes doctor --fix`\n" + f"1. Run `hermes {profile_arg}doctor --fix`\n" "2. Stop the gateway, then recover with:\n" - f" hermes sessions recover --source {db_path} " + f" hermes {profile_arg}sessions recover --source {db_path} " "--inspect-only\n" - " (if it reports recoverable) hermes sessions recover " + f" (if it reports recoverable) hermes {profile_arg}sessions recover " f"--source {db_path} --output recovered-state.db\n" " — recovery snapshots the damaged file first; do NOT run " "`sqlite3 ... \".recover\"` against the live state.db, a " "vulnerable sqlite3 CLI can corrupt it further\n" f"3. Restore from a backup in {backups_dir}/\n" - "Run `hermes doctor` for sanitized diagnostics." + f"Run `hermes {profile_arg}doctor` for sanitized diagnostics." ) elif cause == "fts_index": # Index-scoped corruption: the message tables are not damaged, so the recover / @@ -861,7 +863,7 @@ class GatewayNotificationsMixin: message = ( "⚠️ Session database reported a corruption error confined to the search index " "(FTS5); the message tables are not damaged. Messages may not be persisted until " - "it is repaired: run `hermes doctor --fix`, then restart the gateway. Do not run " + f"it is repaired: run `hermes {profile_arg}doctor --fix`, then restart the gateway. Do not run " "recovery tools or restore a backup unless `hermes doctor` confirms damage." ) else: diff --git a/hermes_state_repair.py b/hermes_state_repair.py index 7062534366..e7918d0d40 100644 --- a/hermes_state_repair.py +++ b/hermes_state_repair.py @@ -391,11 +391,14 @@ def _persistent_repair_attempts_exhausted(db_path: Path) -> bool: def _persistent_repair_exhausted_error(db_path: Path) -> str: - """The stable operator-facing diagnostic for an exhausted repair budget.""" + """The stable operator-facing diagnostic for an exhausted repair budget. The ``hermes`` commands + carry the profile selector: a bare ``hermes`` follows ``active_profile`` (#105887).""" + from hermes_constants import profile_cli_selector + profile_arg = profile_cli_selector() return (f"automatic repair has already failed {_MAX_PERSISTENT_REPAIR_ATTEMPTS} times on this exact file — the " f"corruption is beyond the schema/FTS repair strategies (likely b-tree page damage). Manual recovery " - f"required: restore a backup, or salvage with `hermes sessions recover --source {db_path} " - f"--inspect-only`, then (if it reports recoverable) `hermes sessions recover --source {db_path} " + f"required: restore a backup, or salvage with `hermes {profile_arg}sessions recover --source {db_path} " + f"--inspect-only`, then (if it reports recoverable) `hermes {profile_arg}sessions recover --source {db_path} " f"--output recovered-state.db` (recovery snapshots the damaged file first, then runs the page-level " f"`.recover` lane on the copy; do NOT point a raw `sqlite3` shell at the live database). " f"Delete {_repair_ledger_path(db_path).name} to force another automatic attempt.") diff --git a/tests/run_agent/test_corruption_recovery_guidance.py b/tests/run_agent/test_corruption_recovery_guidance.py index e942bdfa6c..2204de8cd5 100644 --- a/tests/run_agent/test_corruption_recovery_guidance.py +++ b/tests/run_agent/test_corruption_recovery_guidance.py @@ -123,3 +123,47 @@ def test_format_turn_completion_locked_still_advises_retry(): ) assert "busy" in explanation assert "send it again" in explanation + + +def test_corrupt_guidance_pins_the_failing_profile(tmp_path, monkeypatch): + """#105887: every `hermes ...` command in the recovery guidance (turn explainer, gateway + home-channel notice, exhausted-repair diagnostic) carries the active profile selector and + names that profile's state.db. A bare `hermes` follows the sticky ``active_profile`` file, + so with another profile active the operator would repair the wrong database.""" + import asyncio + + import gateway.run as gateway_run + from hermes_state import _default_db_path + from hermes_state_repair import _persistent_repair_exhausted_error + from run_agent import AIAgent + + root = tmp_path / "hermes" + home = root / "profiles" / "research" + home.mkdir(parents=True) + (root / "config.yaml").write_text("") + (root / "active_profile").write_text("other\n") + monkeypatch.setenv("HERMES_HOME", str(home)) + + explanation = AIAgent._format_turn_completion_explanation("session_persistence_failed", "corrupt") + commands = [line.strip() for line in explanation.splitlines() if "hermes " in line] + assert commands and all("hermes -p research " in line for line in commands), commands + # The conftest pins hermes_state.DEFAULT_DB_PATH, so the store named is whatever the + # process resolves — the contract is "the same path the runtime would open". + assert f"--source {_default_db_path()} " in explanation + + runner = object.__new__(gateway_run.GatewayRunner) + runner._session_db_init_error = "database disk image is malformed" + sent = [] + monkeypatch.setattr(runner, "_home_channel_transports", lambda: [("telegram", {}, "home-chat", object())]) + + async def _capture_send(_platform, _home, _transport, message, _log_fmt): + sent.append(message) + + monkeypatch.setattr(runner, "_send_home_channel_message", _capture_send) + asyncio.run(runner._send_session_db_warning_notifications()) + notice_commands = [line.strip() for line in sent[0].splitlines() if "hermes " in line] + assert notice_commands and all("hermes -p research " in line for line in notice_commands), notice_commands + assert f"--source {_default_db_path()} " in sent[0] + + exhausted = _persistent_repair_exhausted_error(home / "state.db") + assert "`hermes -p research sessions recover --source" in exhausted From d8cf5da7ac78f7ac866b81a3bc1171f352b7eb29 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 11 Sep 2026 02:07:00 -0700 Subject: [PATCH 12/98] fix(state): make the FTS write-health probe flush segments and catch IntegrityError MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `_db_opens_cleanly` drove one probe row through the messages_fts* triggers and rolled back. FTS5 only buffers that row in an in-memory segment until commit, so the probe never wrote to `_idx`/`_data` and could not hit a stale `messages_fts_trigram_idx` row waiting at the next segid — the class where PRAGMA integrity_check, the FTS5 integrity-check command and MATCH all report clean while every committed append fails with `IntegrityError: constraint failed`. The probe also caught only OperationalError; IntegrityError is a DatabaseError sibling, so even a colliding probe would have escaped and been reported as healthy. Now the probe issues `INSERT INTO () VALUES('flush')` for every FTS family inside the rolled-back transaction (capability / not-built errors stay benign), catches sqlite3.DatabaseError, and always rolls back in a finally. `hermes doctor` and `hermes sessions repair --check-only` surface the corruption and `repair_state_db_schema` heals it via the FTS rebuild strategy (verified with a real stale-segid fixture). Refs #100227 Reported-by: #100227 --- hermes_state_repair.py | 24 ++++-- ...st_state_db_fts_segment_collision_probe.py | 76 +++++++++++++++++++ 2 files changed, 94 insertions(+), 6 deletions(-) create mode 100644 tests/test_state_db_fts_segment_collision_probe.py diff --git a/hermes_state_repair.py b/hermes_state_repair.py index e7918d0d40..94cf209f9d 100644 --- a/hermes_state_repair.py +++ b/hermes_state_repair.py @@ -788,7 +788,11 @@ def _db_opens_cleanly(db_path: Path) -> Optional[str]: # image is malformed") while reads of the FTS5 table itself parse fine. return f"fts5 read probe failed on {fts_table}: {exc}" # FTS write probe: drive a row through the messages_fts* triggers in a transaction that is always - # rolled back. + # rolled back. The trigger INSERT alone only buffers the row in FTS5's in-memory segment; the + # ``flush`` command writes that segment to ``_idx``/``_data`` exactly as a committed append + # would, so a stale ``_idx`` row at the next segid (IntegrityError "constraint failed", the #100227 + # class: integrity_check and MATCH both clean, every real append fails) is hit here rather than + # by the user's next message. probe_session_id = f"_hermes_fts_health_probe_{time.time_ns()}" try: conn.execute("BEGIN IMMEDIATE") @@ -796,16 +800,24 @@ def _db_opens_cleanly(db_path: Path) -> Optional[str]: (probe_session_id, "_health_probe", time.time())) conn.execute("INSERT INTO messages (session_id, role, content, timestamp) VALUES (?, ?, ?, ?)", (probe_session_id, "user", "_fts_health_probe", time.time())) - conn.execute("ROLLBACK") - except sqlite3.OperationalError as exc: - with contextlib.suppress(sqlite3.Error): - conn.execute("ROLLBACK") + for fts_table in _FTS_TABLES: + try: + conn.execute(f"INSERT INTO {fts_table}({fts_table}) VALUES('flush')") + except sqlite3.OperationalError as exc: + if not (SessionDB._is_fts5_unavailable_error(exc) or _schema_not_built(exc)): + raise + except sqlite3.DatabaseError as exc: + # IntegrityError is a DatabaseError sibling of OperationalError, not a child: catching only the + # latter let the trigram-segment collision report "healthy". # Missing messages/sessions tables = brand new file mid-init, not corruption. "no such tokenizer": # this process lacks the cjk extension the DB's index needs — capability gap; a tokenizer-less # SessionDB drops the triggers itself. if _schema_not_built(exc) or "no such tokenizer: cjk_unicode61" in str(exc).lower(): return None - return str(exc) + return f"fts5 write probe failed: {exc}" + finally: + with contextlib.suppress(sqlite3.Error): + conn.execute("ROLLBACK") return None except sqlite3.DatabaseError as exc: return str(exc) diff --git a/tests/test_state_db_fts_segment_collision_probe.py b/tests/test_state_db_fts_segment_collision_probe.py new file mode 100644 index 0000000000..cd4725c483 --- /dev/null +++ b/tests/test_state_db_fts_segment_collision_probe.py @@ -0,0 +1,76 @@ +"""Segment-dependent FTS5 trigram corruption (#100227). + +A stale ``messages_fts_trigram_idx`` row sitting at the segid FTS5 allocates next makes every +committed append fail with ``IntegrityError: constraint failed`` while ``PRAGMA integrity_check``, +the FTS5 ``integrity-check`` command and ``MATCH`` all report healthy. The write probe must +report it (and the repair must heal it) without any mocked detector. +""" +import sqlite3 +import time +import uuid +from pathlib import Path + +import pytest + +from hermes_state import SessionDB +from hermes_state_repair import _db_opens_cleanly, repair_state_db_schema + + +def _build_db_with_trigram(db_path: Path) -> str: + db = SessionDB(db_path=db_path) + if not db._trigram_available: + db.close() + pytest.skip("trigram tokenizer unavailable in this SQLite build") + sid = db.create_session(session_id=str(uuid.uuid4()), source="cli") + for i in range(60): + db.append_message(sid, role="user", content=f"quick brown fox {i} lorem ipsum dolor {i * 7}") + db.close() + return sid + + +def _plant_stale_trigram_segment(db_path: Path) -> None: + """Leave an index row at the next free segid: the shape an aborted segment write leaves behind.""" + conn = sqlite3.connect(str(db_path), isolation_level=None) + used = {r[0] for r in conn.execute("SELECT segid FROM messages_fts_trigram_idx")} + stale = next(s for s in range(1, 1 << 20) if s not in used) + conn.execute("INSERT INTO messages_fts_trigram_idx(segid, term, pgno) VALUES (?, X'', 2)", (stale,)) + conn.close() + + +def _real_append_fails(db_path: Path, sid: str) -> bool: + conn = sqlite3.connect(str(db_path), isolation_level=None) + try: + conn.execute("INSERT INTO messages (session_id, role, content, timestamp) VALUES (?, ?, ?, ?)", + (sid, "user", "zebra yak xylophone wombat", time.time())) + return False + except sqlite3.IntegrityError: + return True + finally: + conn.close() + + +def test_write_probe_reports_segment_collision_that_integrity_check_misses(tmp_path): + db_path = tmp_path / "state.db" + sid = _build_db_with_trigram(db_path) + _plant_stale_trigram_segment(db_path) + + assert sqlite3.connect(str(db_path)).execute("PRAGMA integrity_check").fetchall() == [("ok",)] + assert _real_append_fails(db_path, sid), "fixture must break real appends" + + reason = _db_opens_cleanly(db_path) + assert reason is not None and "constraint failed" in reason + # The probe rolls back: it must not have added rows or moved the FTS state. + assert sqlite3.connect(str(db_path)).execute("SELECT COUNT(*) FROM sessions").fetchone()[0] == 1 + + +def test_repair_heals_segment_collision_and_restores_appends(tmp_path): + db_path = tmp_path / "state.db" + sid = _build_db_with_trigram(db_path) + _plant_stale_trigram_segment(db_path) + + report = repair_state_db_schema(db_path, backup=False) + assert report.get("repaired"), report + assert _db_opens_cleanly(db_path) is None + assert not _real_append_fails(db_path, sid) + with SessionDB(db_path=db_path) as db: + assert db._conn.execute("SELECT COUNT(*) FROM messages WHERE session_id = ?", (sid,)).fetchone()[0] == 61 From 5a2f512390698d9b85c2106f19cf70567f7ff60b Mon Sep 17 00:00:00 2001 From: jango <91889514+jangomango76@users.noreply.github.com> Date: Fri, 11 Sep 2026 02:09:04 -0700 Subject: [PATCH 13/98] fix(cli): exit non-zero from `sessions repair --check-only` on an unhealthy store `hermes sessions repair --check-only` printed the corruption reason and exited 0, so scripts and the console wrapper gating on the status read a broken state.db as healthy. Return 1 from the CLI handler (main.py already sys.exits a truthy return) and from the console handler, where `_capture_output` turns the status into a ConsoleCommandError carrying the printed reason. Salvaged from PR #103321 (the check-only reporting part only; the probe rewrite and connection-tracking changes were not taken). Refs #63386. --- ...9514+jangomango76@users.noreply.github.com | 2 ++ hermes_cli/console_engine.py | 4 +-- hermes_cli/sessions_cmd.py | 2 +- .../test_sessions_repair_check_only_exit.py | 28 +++++++++++++++++++ 4 files changed, 33 insertions(+), 3 deletions(-) create mode 100644 contributors/emails/91889514+jangomango76@users.noreply.github.com create mode 100644 tests/hermes_cli/test_sessions_repair_check_only_exit.py diff --git a/contributors/emails/91889514+jangomango76@users.noreply.github.com b/contributors/emails/91889514+jangomango76@users.noreply.github.com new file mode 100644 index 0000000000..e75f7b7ef0 --- /dev/null +++ b/contributors/emails/91889514+jangomango76@users.noreply.github.com @@ -0,0 +1,2 @@ +jangomango76 +# PR #103321 salvage diff --git a/hermes_cli/console_engine.py b/hermes_cli/console_engine.py index 6d6b23380e..160e76e85e 100644 --- a/hermes_cli/console_engine.py +++ b/hermes_cli/console_engine.py @@ -705,7 +705,7 @@ def _sessions_optimize(_engine: HermesConsoleEngine, args: list[str]) -> None: @_captured -def _sessions_repair(_engine: HermesConsoleEngine, args: list[str]) -> None: +def _sessions_repair(_engine: HermesConsoleEngine, args: list[str]) -> int | None: ns = _parse( "sessions repair", args, (("--check-only",), dict(action="store_true")), (("--no-backup",), dict(action="store_true"))) @@ -721,7 +721,7 @@ def _sessions_repair(_engine: HermesConsoleEngine, args: list[str]) -> None: return print(f"{db_path} does not open cleanly: {reason}") if ns.check_only: - return + return 1 # _capture_output turns a non-zero status into a ConsoleCommandError carrying the printed reason report = repair_state_db_schema(db_path, backup=not ns.no_backup) if not report.get("repaired"): raise ConsoleCommandError(f"Repair failed: {report.get('error')}") diff --git a/hermes_cli/sessions_cmd.py b/hermes_cli/sessions_cmd.py index ba3b226e76..37b95fd7b6 100644 --- a/hermes_cli/sessions_cmd.py +++ b/hermes_cli/sessions_cmd.py @@ -97,7 +97,7 @@ def _cmd_repair(args): return print(f"✗ {db_path} does not open cleanly: {reason}") if getattr(args, "check_only", False): - return + return 1 print("Repairing (a backup copy is made first)…") report = repair_state_db_schema(db_path, backup=not getattr(args, "no_backup", False)) if report.get("repaired"): diff --git a/tests/hermes_cli/test_sessions_repair_check_only_exit.py b/tests/hermes_cli/test_sessions_repair_check_only_exit.py new file mode 100644 index 0000000000..834a40ea64 --- /dev/null +++ b/tests/hermes_cli/test_sessions_repair_check_only_exit.py @@ -0,0 +1,28 @@ +"""`hermes sessions repair --check-only` must fail (non-zero) when the store is unhealthy. + +Automation gates on the exit status; printing the reason and exiting 0 read as "healthy" +(#63386, PR #103321). +""" +import argparse +from pathlib import Path + +import hermes_state +from hermes_cli import sessions_cmd +from hermes_state import SessionDB + + +def test_check_only_exit_status_tracks_probe_verdict(tmp_path, monkeypatch): + db_path = tmp_path / "state.db" + SessionDB(db_path=db_path).close() + monkeypatch.setattr(hermes_state, "DEFAULT_DB_PATH", db_path) + args = argparse.Namespace(check_only=True, no_backup=False) + + assert not sessions_cmd._cmd_repair(args) + # Destroy the schema header so the probe fails on its first statement. + with open(db_path, "r+b") as f: + f.seek(100) + f.write(b"\xff" * (4096 - 100)) + for side in ("-wal", "-shm"): + Path(str(db_path) + side).unlink(missing_ok=True) + assert sessions_cmd._cmd_repair(args) == 1 + assert db_path.stat().st_size > 0, "check-only must not touch the file" From fdd32f81d45ddaf955267b2d969204aa26f81faf Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 11 Sep 2026 02:15:22 -0700 Subject: [PATCH 14/98] fix(web): report a corrupt state.db as a throttled 503 status instead of a traceback per poll MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The dashboard polls /api/analytics/usage and /api/analytics/models every few seconds. When state.db is malformed the read raised straight through the handler, so uvicorn logged a full traceback at ERROR on every poll — one fleet host wrote ~520K identical journal entries in 24 h. Wrap both analytics handlers in `corrupt_store_as_status`: a corrupt-image sqlite3.DatabaseError (is_malformed_db_error) becomes a 503 with an explicit `state_db_corrupt` payload pointing at `hermes doctor`, and the warning is gated per store path via `{path: monotonic}` (>=300 s), then debug. Busy/locked and every other error propagate unchanged, and the file is never renamed or quarantined from the dashboard — repair stays with `hermes doctor` / `hermes sessions repair`. `_session_db_path_for_profile` is split out of `_open_session_db_for_profile` so the router can name the store without opening it. Refs #96591 Reported-by: #96591 --- hermes_cli/web_routers/_common.py | 40 ++++++++++++- hermes_cli/web_routers/analytics.py | 8 ++- hermes_cli/web_server_sessions.py | 21 ++++--- .../test_web_analytics_corrupt_store.py | 57 +++++++++++++++++++ 4 files changed, 114 insertions(+), 12 deletions(-) create mode 100644 tests/hermes_cli/test_web_analytics_corrupt_store.py diff --git a/hermes_cli/web_routers/_common.py b/hermes_cli/web_routers/_common.py index 1301353f5b..a93974894d 100644 --- a/hermes_cli/web_routers/_common.py +++ b/hermes_cli/web_routers/_common.py @@ -7,7 +7,9 @@ from __future__ import annotations import asyncio import contextlib import logging -from typing import Any, Callable, Optional +import sqlite3 +import time +from typing import Any, Callable, Dict, Optional from fastapi import HTTPException @@ -76,3 +78,39 @@ def require(value: Optional[str], detail: str) -> str: if not stripped: raise HTTPException(status_code=400, detail=detail) return stripped + + +# Corrupt-store reporting for polled read endpoints. The dashboard polls analytics every few +# seconds; a persistently malformed state.db once produced ~520K identical tracebacks in 24 h +# (#96591). One WARNING per store per interval, then debug; the caller gets an explicit status +# instead of a 500. The file is never quarantined or renamed from here — that is `hermes doctor`'s job. +_CORRUPT_STORE_WARN_INTERVAL_S = 300.0 +_corrupt_store_warned_at: Dict[str, float] = {} # {db path: monotonic} + +CORRUPT_STORE_DETAIL = { + "error": "state_db_corrupt", + "message": "state.db corrupt — run `hermes doctor` (then `hermes doctor --fix` or `hermes sessions repair`).", +} + + +@contextlib.contextmanager +def corrupt_store_as_status(db_path): + """Map a corrupt-image ``sqlite3.DatabaseError`` from a state.db read to a 503 status + payload, warning once per store per :data:`_CORRUPT_STORE_WARN_INTERVAL_S`. + Busy/locked and every other error propagate unchanged.""" + from hermes_state_errors import is_malformed_db_error + + try: + yield + except sqlite3.DatabaseError as exc: + if not is_malformed_db_error(exc): + raise + key, now = str(db_path), time.monotonic() + last = _corrupt_store_warned_at.get(key) + if last is None or now - last >= _CORRUPT_STORE_WARN_INTERVAL_S: + _corrupt_store_warned_at[key] = now + log.warning("state.db at %s is corrupt (%s); dashboard reads return a status payload until it is " + "repaired — run `hermes doctor`", db_path, exc) + else: + log.debug("state.db at %s still corrupt: %s", db_path, exc) + raise HTTPException(status_code=503, detail={**CORRUPT_STORE_DETAIL, "path": key}) from exc diff --git a/hermes_cli/web_routers/analytics.py b/hermes_cli/web_routers/analytics.py index d3dac5f9bf..77d1131865 100644 --- a/hermes_cli/web_routers/analytics.py +++ b/hermes_cli/web_routers/analytics.py @@ -13,6 +13,7 @@ from fastapi import APIRouter, HTTPException, Query from hermes_cli.config import get_config_path, read_raw_config from hermes_cli.web_deps import late +from hermes_cli.web_routers._common import corrupt_store_as_status from hermes_cli.web_server_profiles import ( _approval_mode_of, _aux_task_summary, _aux_usage_rows, _broadcast_gateway_session_info, _is_other_profile, _merge_aux_into_by_model, ) @@ -22,6 +23,7 @@ router = APIRouter() # Late-bound so a test's monkeypatch on the owning module wins at call time. _open_session_db_for_profile = late("_open_session_db_for_profile", "hermes_cli.web_server_sessions") +_session_db_path_for_profile = late("_session_db_path_for_profile", "hermes_cli.web_server_sessions") _profile_scope = late("_profile_scope", "hermes_cli.web_server_profiles") save_config = late("save_config", "hermes_cli.config") @@ -147,7 +149,8 @@ async def get_usage_analytics( values would force expensive full-history SQL and InsightsEngine work, or produce empty/inverted time windows. The UI only offers 7/30/90-day presets.""" - return await asyncio.to_thread(_get_usage_analytics, days, profile) + with corrupt_store_as_status(_session_db_path_for_profile(profile)): + return await asyncio.to_thread(_get_usage_analytics, days, profile) _USAGE_KEYS = ( @@ -305,4 +308,5 @@ async def get_models_analytics( profile: Optional[str] = None, ): """Return model analytics without blocking the serving event loop.""" - return await asyncio.to_thread(_get_models_analytics, days, profile) + with corrupt_store_as_status(_session_db_path_for_profile(profile)): + return await asyncio.to_thread(_get_models_analytics, days, profile) diff --git a/hermes_cli/web_server_sessions.py b/hermes_cli/web_server_sessions.py index 106288207d..70bfbc914b 100644 --- a/hermes_cli/web_server_sessions.py +++ b/hermes_cli/web_server_sessions.py @@ -180,20 +180,23 @@ def _open_session_db_at_path(db_path: Path, *, read_only: bool): return _open_probed() -def _open_session_db_for_profile(profile: Optional[str], *, read_only: bool): - """Open a SessionDB for ``profile`` (None/empty = this process's own state.db). - - Access-mode semantics: see :func:`_open_session_db_at_path`. - """ +def _session_db_path_for_profile(profile: Optional[str]) -> Path: + """state.db path for ``profile`` (None/empty = this process's own).""" from hermes_cli.web_server_cron import _cron_profile_home from hermes_state import _default_db_path if profile: _name, home = _cron_profile_home(profile) - db_path = Path(home) / "state.db" - else: - db_path = Path(_default_db_path()) - return _open_session_db_at_path(db_path, read_only=read_only) + return Path(home) / "state.db" + return Path(_default_db_path()) + + +def _open_session_db_for_profile(profile: Optional[str], *, read_only: bool): + """Open a SessionDB for ``profile`` (None/empty = this process's own state.db). + + Access-mode semantics: see :func:`_open_session_db_at_path`. + """ + return _open_session_db_at_path(_session_db_path_for_profile(profile), read_only=read_only) # In-process throttle for the opportunistic auto-archive trigger, keyed by diff --git a/tests/hermes_cli/test_web_analytics_corrupt_store.py b/tests/hermes_cli/test_web_analytics_corrupt_store.py new file mode 100644 index 0000000000..d1cd877ed0 --- /dev/null +++ b/tests/hermes_cli/test_web_analytics_corrupt_store.py @@ -0,0 +1,57 @@ +"""A malformed state.db must not turn dashboard analytics polling into a traceback storm (#96591).""" +import logging +import sqlite3 +from pathlib import Path + +import pytest +from fastapi import FastAPI +from fastapi.testclient import TestClient + +from hermes_cli.web_routers import _common, analytics +from hermes_state import SessionDB + + +def _malformed_state_db(home: Path) -> Path: + db_path = home / "state.db" + db = SessionDB(db_path=db_path) + db.create_session("s1", source="cli", model="m") + db.close() + for side in ("-wal", "-shm"): + Path(str(db_path) + side).unlink(missing_ok=True) + with open(db_path, "r+b") as f: + f.seek(100) + f.write(b"\xff" * (4096 - 100)) + with pytest.raises(sqlite3.DatabaseError): + sqlite3.connect(str(db_path)).execute("SELECT count(*) FROM sqlite_master") + return db_path + + +def test_corrupt_store_polls_return_status_and_warn_once_per_interval(tmp_path, monkeypatch, caplog): + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + import hermes_state + db_path = _malformed_state_db(tmp_path) + monkeypatch.setattr(hermes_state, "_default_db_path", lambda: db_path) + monkeypatch.setattr(_common, "_corrupt_store_warned_at", {}) + app = FastAPI() + app.include_router(analytics.router) + client = TestClient(app) + + with caplog.at_level(logging.DEBUG, logger="hermes_cli.web_server"): + first = client.get("/api/analytics/usage?days=7") + second = client.get("/api/analytics/usage?days=7") + third = client.get("/api/analytics/models?days=7") + for resp in (first, second, third): + assert resp.status_code == 503 + assert resp.json()["detail"]["error"] == "state_db_corrupt" + assert "hermes doctor" in resp.json()["detail"]["message"] + warnings = [r for r in caplog.records if r.levelno >= logging.WARNING] + assert len(warnings) == 1, [r.getMessage() for r in warnings] + assert not any(r.exc_info for r in caplog.records), "no tracebacks for a known corrupt store" + assert db_path.exists() and db_path.stat().st_size > 0, "dashboard must never quarantine the file" + + # The gate re-arms once the interval has elapsed (aged, not slept). + _common._corrupt_store_warned_at[str(db_path)] -= _common._CORRUPT_STORE_WARN_INTERVAL_S + 1 + caplog.clear() + with caplog.at_level(logging.WARNING, logger="hermes_cli.web_server"): + assert client.get("/api/analytics/usage?days=7").status_code == 503 + assert sum(r.levelno >= logging.WARNING for r in caplog.records) == 1 From 04dd80a977f40b05e5b2054111747af07a61886a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 11 Sep 2026 03:33:19 -0700 Subject: [PATCH 15/98] test(web): count only the dashboard's own warning in the corrupt-store poll test hermes_state emits a once-per-process SQLite-version advisory on CI's linked 3.50.4, which caplog captured as a second WARNING. Scope the assertion to the hermes_cli.web_server logger the router actually writes to. --- tests/hermes_cli/test_web_analytics_corrupt_store.py | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/tests/hermes_cli/test_web_analytics_corrupt_store.py b/tests/hermes_cli/test_web_analytics_corrupt_store.py index d1cd877ed0..d2214be81f 100644 --- a/tests/hermes_cli/test_web_analytics_corrupt_store.py +++ b/tests/hermes_cli/test_web_analytics_corrupt_store.py @@ -44,7 +44,12 @@ def test_corrupt_store_polls_return_status_and_warn_once_per_interval(tmp_path, assert resp.status_code == 503 assert resp.json()["detail"]["error"] == "state_db_corrupt" assert "hermes doctor" in resp.json()["detail"]["message"] - warnings = [r for r in caplog.records if r.levelno >= logging.WARNING] + # Only the dashboard's own warning counts: hermes_state logs an unrelated + # once-per-process SQLite-version advisory on some interpreters (CI's 3.50.4). + warnings = [ + r for r in caplog.records + if r.levelno >= logging.WARNING and r.name.startswith("hermes_cli.web_server") + ] assert len(warnings) == 1, [r.getMessage() for r in warnings] assert not any(r.exc_info for r in caplog.records), "no tracebacks for a known corrupt store" assert db_path.exists() and db_path.stat().st_size > 0, "dashboard must never quarantine the file" @@ -54,4 +59,7 @@ def test_corrupt_store_polls_return_status_and_warn_once_per_interval(tmp_path, caplog.clear() with caplog.at_level(logging.WARNING, logger="hermes_cli.web_server"): assert client.get("/api/analytics/usage?days=7").status_code == 503 - assert sum(r.levelno >= logging.WARNING for r in caplog.records) == 1 + assert sum( + r.levelno >= logging.WARNING and r.name.startswith("hermes_cli.web_server") + for r in caplog.records + ) == 1 From 0b8daf30aae1d0b129ede9b857cac2158eb50324 Mon Sep 17 00:00:00 2001 From: KoNit-K <124019182+KoNit-K@users.noreply.github.com> Date: Thu, 10 Sep 2026 16:13:35 +0800 Subject: [PATCH 16/98] fix(bootstrap-installer): stamp setup app version from release semver Fixes #107177 Co-authored-by: Cursor --- apps/bootstrap-installer/package.json | 2 +- apps/bootstrap-installer/src-tauri/Cargo.toml | 2 +- .../src-tauri/tauri.conf.json | 2 +- scripts/release.py | 56 ++++++- ...est_release_bootstrap_installer_version.py | 144 ++++++++++++++++++ 5 files changed, 202 insertions(+), 4 deletions(-) create mode 100644 tests/scripts/test_release_bootstrap_installer_version.py diff --git a/apps/bootstrap-installer/package.json b/apps/bootstrap-installer/package.json index c385f9b108..ff55c62f54 100644 --- a/apps/bootstrap-installer/package.json +++ b/apps/bootstrap-installer/package.json @@ -1,7 +1,7 @@ { "name": "@hermes/bootstrap-installer", "private": true, - "version": "0.0.1", + "version": "0.21.1", "description": "Hermes Setup — signed installer that drives scripts/install.ps1 with a polished native UI.", "type": "module", "scripts": { diff --git a/apps/bootstrap-installer/src-tauri/Cargo.toml b/apps/bootstrap-installer/src-tauri/Cargo.toml index 78fc71e56c..2057af89bb 100644 --- a/apps/bootstrap-installer/src-tauri/Cargo.toml +++ b/apps/bootstrap-installer/src-tauri/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "hermes-bootstrap" -version = "0.0.1" +version = "0.21.1" description = "Hermes Setup — signed installer that drives scripts/install.ps1" authors = ["Nous Research "] edition = "2021" diff --git a/apps/bootstrap-installer/src-tauri/tauri.conf.json b/apps/bootstrap-installer/src-tauri/tauri.conf.json index a74bd105c3..c789e9524c 100644 --- a/apps/bootstrap-installer/src-tauri/tauri.conf.json +++ b/apps/bootstrap-installer/src-tauri/tauri.conf.json @@ -1,7 +1,7 @@ { "$schema": "https://schema.tauri.app/config/2", "productName": "Hermes", - "version": "0.0.1", + "version": "0.21.1", "identifier": "com.nousresearch.hermes.setup", "build": { "beforeDevCommand": "npm run dev", diff --git a/scripts/release.py b/scripts/release.py index 10eb846a63..5a07abf88a 100755 --- a/scripts/release.py +++ b/scripts/release.py @@ -2230,6 +2230,60 @@ def update_version_files(semver: str, calver_date: str): ) desktop_pkg.write_text(pkg_text, encoding="utf-8") + # Keep the bootstrap installer (Hermes-Setup.dmg CFBundleShortVersionString) + # in lockstep with the Python package version. Tauri reads `version` from + # package.json + tauri.conf.json; a hardcoded 0.0.1 ships in the DMG. + installer_pkg = REPO_ROOT / "apps" / "bootstrap-installer" / "package.json" + if installer_pkg.exists(): + pkg_text = installer_pkg.read_text(encoding="utf-8") + pkg_text = re.sub( + r'("version"\s*:\s*)"[^"]+"', + rf'\g<1>"{semver}"', + pkg_text, + count=1, + ) + installer_pkg.write_text(pkg_text, encoding="utf-8") + + installer_tauri = ( + REPO_ROOT / "apps" / "bootstrap-installer" / "src-tauri" / "tauri.conf.json" + ) + if installer_tauri.exists(): + pkg_text = installer_tauri.read_text(encoding="utf-8") + pkg_text = re.sub( + r'("version"\s*:\s*)"[^"]+"', + rf'\g<1>"{semver}"', + pkg_text, + count=1, + ) + installer_tauri.write_text(pkg_text, encoding="utf-8") + + installer_cargo = ( + REPO_ROOT / "apps" / "bootstrap-installer" / "src-tauri" / "Cargo.toml" + ) + if installer_cargo.exists(): + cargo_text = installer_cargo.read_text(encoding="utf-8") + cargo_text = re.sub( + r'^version\s*=\s*"[^"]+"', + f'version = "{semver}"', + cargo_text, + count=1, + flags=re.MULTILINE, + ) + installer_cargo.write_text(cargo_text, encoding="utf-8") + + +def version_files_to_stage() -> list[str]: + """Return version-bearing files that exist and should be `git add`ed after a bump.""" + candidates = [ + VERSION_FILE, + PYPROJECT_FILE, + REPO_ROOT / "apps" / "desktop" / "package.json", + REPO_ROOT / "apps" / "bootstrap-installer" / "package.json", + REPO_ROOT / "apps" / "bootstrap-installer" / "src-tauri" / "tauri.conf.json", + REPO_ROOT / "apps" / "bootstrap-installer" / "src-tauri" / "Cargo.toml", + ] + return [str(path) for path in candidates if path.exists()] + def resolve_author(name: str, email: str) -> str: """Resolve a git author to a GitHub @mention.""" @@ -2568,7 +2622,7 @@ def main(): print(f" ✓ Updated version files to v{new_version} ({calver_date})") # Commit version bump - add_files = [str(VERSION_FILE), str(PYPROJECT_FILE)] + add_files = version_files_to_stage() add_result = git_result("add", *add_files) if add_result.returncode != 0: print(f" ✗ Failed to stage version files: {add_result.stderr.strip()}") diff --git a/tests/scripts/test_release_bootstrap_installer_version.py b/tests/scripts/test_release_bootstrap_installer_version.py new file mode 100644 index 0000000000..a6e4ab25e5 --- /dev/null +++ b/tests/scripts/test_release_bootstrap_installer_version.py @@ -0,0 +1,144 @@ +"""release.py must stamp bootstrap-installer versions with the release semver. + +Tauri CFBundleShortVersionString is read from +apps/bootstrap-installer/src-tauri/tauri.conf.json (and the sibling +package.json). Those files were hardcoded 0.0.1 and omitted from +update_version_files / the --publish --bump git add list, so Hermes-Setup.dmg +always shipped 0.0.1. Same class as the desktop stamp (#68783 / PR #68796). +""" + +from __future__ import annotations + +import importlib.util +import json +from pathlib import Path + +SCRIPT = Path(__file__).resolve().parents[2] / "scripts" / "release.py" + + +def _load(): + spec = importlib.util.spec_from_file_location( + "release_bootstrap_installer_version", SCRIPT + ) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +release = _load() + + +def _json_version(path: Path) -> str: + return json.loads(path.read_text(encoding="utf-8"))["version"] + + +def _patch_repo(tmp_path, monkeypatch, *, with_installer: bool = True): + repo = tmp_path + init_py = repo / "hermes_cli" / "__init__.py" + init_py.parent.mkdir(parents=True) + init_py.write_text( + '__version__ = "0.0.1"\n__release_date__ = "2026.1.1"\n', + encoding="utf-8", + ) + pyproject = repo / "pyproject.toml" + pyproject.write_text('version = "0.0.1"\n', encoding="utf-8") + + desktop_pkg = repo / "apps" / "desktop" / "package.json" + desktop_pkg.parent.mkdir(parents=True) + desktop_pkg.write_text('{"version":"0.0.1"}\n', encoding="utf-8") + + installer_pkg = repo / "apps" / "bootstrap-installer" / "package.json" + tauri_conf = ( + repo / "apps" / "bootstrap-installer" / "src-tauri" / "tauri.conf.json" + ) + cargo_toml = repo / "apps" / "bootstrap-installer" / "src-tauri" / "Cargo.toml" + if with_installer: + tauri_conf.parent.mkdir(parents=True) + installer_pkg.write_text( + '{"name":"x","version":"0.0.1"}\n', encoding="utf-8" + ) + tauri_conf.write_text( + '{"productName":"Hermes","version":"0.0.1"}\n', encoding="utf-8" + ) + cargo_toml.write_text('[package]\nversion = "0.0.1"\n', encoding="utf-8") + + monkeypatch.setattr(release, "REPO_ROOT", repo) + monkeypatch.setattr(release, "VERSION_FILE", init_py) + monkeypatch.setattr(release, "PYPROJECT_FILE", pyproject) + return { + "repo": repo, + "init_py": init_py, + "pyproject": pyproject, + "desktop_pkg": desktop_pkg, + "installer_pkg": installer_pkg, + "tauri_conf": tauri_conf, + "cargo_toml": cargo_toml, + } + + +def test_update_version_files_stamps_bootstrap_installer(tmp_path, monkeypatch): + paths = _patch_repo(tmp_path, monkeypatch) + + release.update_version_files("0.21.1", "2026.9.10") + + assert _json_version(paths["installer_pkg"]) == "0.21.1" + assert _json_version(paths["tauri_conf"]) == "0.21.1" + assert 'version = "0.21.1"' in paths["cargo_toml"].read_text(encoding="utf-8") + + # CONTROL: existing desktop / Python stamps still happen. + assert _json_version(paths["desktop_pkg"]) == "0.21.1" + assert 'version = "0.21.1"' in paths["pyproject"].read_text(encoding="utf-8") + init_text = paths["init_py"].read_text(encoding="utf-8") + assert '__version__ = "0.21.1"' in init_text + assert '__release_date__ = "2026.9.10"' in init_text + + +def test_update_version_files_skips_missing_installer_dir(tmp_path, monkeypatch): + paths = _patch_repo(tmp_path, monkeypatch, with_installer=False) + + release.update_version_files("0.21.1", "2026.9.10") + + assert not paths["installer_pkg"].exists() + assert not paths["tauri_conf"].exists() + assert _json_version(paths["desktop_pkg"]) == "0.21.1" + assert 'version = "0.21.1"' in paths["pyproject"].read_text(encoding="utf-8") + assert '__version__ = "0.21.1"' in paths["init_py"].read_text(encoding="utf-8") + + +def test_update_version_files_does_not_invent_version_keys(tmp_path, monkeypatch): + paths = _patch_repo(tmp_path, monkeypatch) + original = '{"name":"x","productName":"Hermes"}\n' + paths["installer_pkg"].write_text(original, encoding="utf-8") + paths["tauri_conf"].write_text(original, encoding="utf-8") + + release.update_version_files("0.21.1", "2026.9.10") + + assert paths["installer_pkg"].read_text(encoding="utf-8") == original + assert paths["tauri_conf"].read_text(encoding="utf-8") == original + assert _json_version(paths["desktop_pkg"]) == "0.21.1" + + +def test_version_files_to_stage_includes_installer_when_present(tmp_path, monkeypatch): + paths = _patch_repo(tmp_path, monkeypatch) + + staged = release.version_files_to_stage() + + assert str(paths["init_py"]) in staged + assert str(paths["pyproject"]) in staged + assert str(paths["desktop_pkg"]) in staged + assert str(paths["installer_pkg"]) in staged + assert str(paths["tauri_conf"]) in staged + assert str(paths["cargo_toml"]) in staged + + +def test_version_files_to_stage_omits_missing_installer(tmp_path, monkeypatch): + paths = _patch_repo(tmp_path, monkeypatch, with_installer=False) + + staged = release.version_files_to_stage() + + assert str(paths["installer_pkg"]) not in staged + assert str(paths["tauri_conf"]) not in staged + assert str(paths["cargo_toml"]) not in staged + assert str(paths["desktop_pkg"]) in staged + assert str(paths["init_py"]) in staged + assert str(paths["pyproject"]) in staged From 3b45681c25a880477a2a806cdebe91d2f1bfe9ce Mon Sep 17 00:00:00 2001 From: KoNit-K <124019182+KoNit-K@users.noreply.github.com> Date: Thu, 10 Sep 2026 04:25:18 +0800 Subject: [PATCH 17/98] fix(desktop): qualify group (you) by connection, not bare profile name Same-named defaults on different connections were mislabeled as (you) in member room-delta prompts. Compare speaker/viewer with connection source identity so only the true self gets the suffix. Fixes #106851 Co-authored-by: Cursor --- .../hermes-bots/cross-connection-bots.test.ts | 44 +++++++++++++++++++ .../hermes-bots/group-round-members.ts | 4 +- .../plugins/hermes-bots/group-round-prompt.ts | 42 ++++++++++++++++-- 3 files changed, 85 insertions(+), 5 deletions(-) diff --git a/apps/desktop/src/plugins/hermes-bots/cross-connection-bots.test.ts b/apps/desktop/src/plugins/hermes-bots/cross-connection-bots.test.ts index b241c223a4..e767812a61 100644 --- a/apps/desktop/src/plugins/hermes-bots/cross-connection-bots.test.ts +++ b/apps/desktop/src/plugins/hermes-bots/cross-connection-bots.test.ts @@ -175,4 +175,48 @@ describe('a group room seats members from several machines', () => { }) ).toMatch(/@dixie \[on Mac Mini\]/) }) + + it('does not self-attribute a same-named default on another connection (#106851)', () => { + // Cross-connection same profile name must not self-attribute in model prompt lines. + const line = formatGroupChatLine( + { at: 1, from: { kind: 'member', name: 'default', source: 'Connection B' }, text: 'Remote reply' }, + { name: 'default' } + ) + + expect(line).not.toContain('(you)') + expect(line).toMatch(/Remote reply/) + + // CONTROL: same-connection self still gets (you) + expect( + formatGroupChatLine( + { at: 2, from: { kind: 'member', name: 'default' }, text: 'Mine' }, + { name: 'default' } + ) + ).toContain('(you)') + + // CONTROL: remote viewer seeing own remote line still gets (you) + expect( + formatGroupChatLine( + { at: 3, from: { kind: 'member', name: 'default', source: 'Connection B' }, text: 'Mine remote' }, + { name: 'default', remoteSource: true, connectionLabel: 'Connection B', connectionId: 'conn-b' } + ) + ).toContain('(you)') + + // String viewer API is local / unsourced — a sourced peer of the same + // name must not pick up (you) just because the names match. + expect( + formatGroupChatLine( + { at: 4, from: { kind: 'member', name: 'default', source: 'Connection B' }, text: 'Still remote' }, + 'default' + ) + ).not.toContain('(you)') + + // User lines stay on the (user) path even when the display name collides. + expect( + formatGroupChatLine({ at: 5, from: { kind: 'user', name: 'default' }, text: 'human' }, { name: 'default' }) + ).toContain('(user)') + expect( + formatGroupChatLine({ at: 5, from: { kind: 'user', name: 'default' }, text: 'human' }, { name: 'default' }) + ).not.toContain('(you)') + }) }) diff --git a/apps/desktop/src/plugins/hermes-bots/group-round-members.ts b/apps/desktop/src/plugins/hermes-bots/group-round-members.ts index ad62d8ec05..51037b227f 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-round-members.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-round-members.ts @@ -93,7 +93,7 @@ function prepareGroupRoundMember(context: GroupRoundMemberContext, member: Group groupName: context.group, members, viewer: member, - deltaLines: delta.slice(-GROUP_CHAT_HISTORY_LIMIT).map((e: GroupMessage) => formatGroupChatLine(e, member.name)) + deltaLines: delta.slice(-GROUP_CHAT_HISTORY_LIMIT).map((e: GroupMessage) => formatGroupChatLine(e, member)) }) // Images riding this delta (user attachments — member entries don't @@ -279,7 +279,7 @@ async function runGroupContinuationMember( // The continuation prompt centers on what the member missed: // everything since its watermark, which includes the reply // that cites it. - deltaLines: delta.slice(-GROUP_CHAT_HISTORY_LIMIT).map((e: GroupMessage) => formatGroupChatLine(e, member.name)) + deltaLines: delta.slice(-GROUP_CHAT_HISTORY_LIMIT).map((e: GroupMessage) => formatGroupChatLine(e, member)) }) let continuationReply: null | string = null diff --git a/apps/desktop/src/plugins/hermes-bots/group-round-prompt.ts b/apps/desktop/src/plugins/hermes-bots/group-round-prompt.ts index 1ee31a6fa1..eaa03c1665 100644 --- a/apps/desktop/src/plugins/hermes-bots/group-round-prompt.ts +++ b/apps/desktop/src/plugins/hermes-bots/group-round-prompt.ts @@ -1,11 +1,17 @@ import { botHandle } from './data' import { groupSpeakerLabel } from './group-chat' import { groupMemberKey } from './group-membership' -import type { GroupMember, GroupMessage } from './types' +import type { GroupMember, GroupMessage, GroupMessageAuthor } from './types' + +/** Viewer identity for a room-log line. A bare string is the local, unsourced + * profile name (legacy call sites and single-connection jobs). */ +export type GroupChatLineViewer = + | string + | (Pick & Partial>) /** Room-log line as a member sees it: `Name (user): …` / `Name: …` / * `Name (you): …`. */ -export function formatGroupChatLine(entry: GroupMessage, viewerName: string) { +export function formatGroupChatLine(entry: GroupMessage, viewer: GroupChatLineViewer) { // Attachments are staged into each member's session as real payloads; the // transcript line names them so the delta text and the bytes line up. const attached = @@ -23,7 +29,7 @@ export function formatGroupChatLine(entry: GroupMessage, viewerName: string) { return `${entry.from.name || 'User'} (user): ${entry.text}${attached}` } - const suffix = entry.from.name === viewerName ? ' (you)' : '' + const suffix = isGroupChatSelf(entry.from, viewer) ? ' (you)' : '' // Cross-connection speakers carry their device so same-named agents on // two machines stay tellable apart in every member's transcript. const source = entry.from.source ? ` [${entry.from.source}]` : '' @@ -31,6 +37,36 @@ export function formatGroupChatLine(entry: GroupMessage, viewerName: string) { return `${groupSpeakerLabel(entry.from.name)}${suffix}${source}: ${entry.text}${attached}` } +function viewerNameOf(viewer: GroupChatLineViewer): string { + return typeof viewer === 'string' ? viewer : viewer?.name || '' +} + +/** Remote members stamp `from.source` as `connectionLabel || connectionId`. + * Only a remoteSource viewer exposes those tokens; a string or local member + * is unsourced so same-name remote lines fail open (no `(you)`). */ +function viewerConnectionSources(viewer: GroupChatLineViewer): string[] { + if (typeof viewer === 'string' || !viewer?.remoteSource) { + return [] + } + + return [viewer.connectionLabel, viewer.connectionId].filter((token): token is string => Boolean(token)) +} + +function isGroupChatSelf(from: GroupMessageAuthor, viewer: GroupChatLineViewer): boolean { + if (!from.name || from.name !== viewerNameOf(viewer)) { + return false + } + + const speakerSource = from.source || '' + const viewerSources = viewerConnectionSources(viewer) + + if (!speakerSource && viewerSources.length === 0) { + return true + } + + return Boolean(speakerSource) && viewerSources.includes(speakerSource) +} + interface GroupChatTurnPromptInput { deltaLines: string[] groupName: string From 73a2597c813406d213c1904a669a596a31049632 Mon Sep 17 00:00:00 2001 From: xxxigm Date: Fri, 11 Sep 2026 22:08:28 +0700 Subject: [PATCH 18/98] fix(desktop): serve local hermes-media ranges so chat video can seek Electron's file:// loader ignores Range, so long clips stay unseekable even though media-range.ts already landed. Wire fetchLocal through that helper. --- apps/desktop/electron/main.ts | 12 +++++------- apps/desktop/electron/media-range.ts | 17 +++++++++++++++++ 2 files changed, 22 insertions(+), 7 deletions(-) diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index 6a19766ca0..24ca25e817 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -262,6 +262,7 @@ import { } from './managed-ssh-update' import { registerMcpOauthCallbackIpc } from './mcp-oauth-callback-ipc' import { createMediaProtocolHandler, MEDIA_PROTOCOL } from './media-protocol' +import { fetchLocalMedia } from './media-range' import { createNativeAccessTokenCoordinator, NativeAuthChangedError } from './native-access-token' import { oauthSessionIsLive, resolveJsonBody, resolveReadinessProbeAuth } from './native-auth-decisions' import { @@ -1366,13 +1367,10 @@ protocol.registerSchemesAsPrivileged([ function registerMediaProtocol() { const handler = createMediaProtocolHandler({ ensureRemoteBearer: baseUrl => ensureNativeAccessToken(baseUrl), - fetchLocal: (resolvedPath, headers, method) => - electronNet.fetch(pathToFileURL(resolvedPath).toString(), { - bypassCustomProtocolHandlers: true, - credentials: 'omit', - headers, - method - }), + // Answer local files ourselves: Electron's file:// loader ignores Range and + // returns the whole body as 200 without Accept-Ranges, which makes