From 6ba45b0e064629c94b45ba75ff41c359f1bea59d Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Sun, 20 Sep 2026 16:02:49 -0700 Subject: [PATCH] fix(sessions): storage maintenance refuses while a writer holds state.db; human-first retired-WAL guard text + recovery guide `hermes sessions optimize`, `optimize-storage` and `prune` now run the same fail-closed holder scan doctor and repair use before rewriting the store. While a gateway, Desktop, dashboard or cron process holds state.db (or a WAL sidecar) they print each holder as `PID N (command)` with the stop remedy and exit 1; `--force` overrides with a warning, `--dry-run` previews are never gated. The Desktop console's `sessions optimize` gets the same refusal. Why: a user ran `optimize-storage` under a fleet of eight live gateways and every agent answered every turn with the retired-WAL refusal until all writers were stopped by hand (#110054, maintainer follow-up 09-20). The DeletedWalGenerationError text is now two layers: a first sentence for the person reading a chat bubble or banner (what happened, nothing is lost, quit every Hermes process on the profile, `hermes doctor` names the holders, never `doctor --fix` or delete files while they run, docs link), then the operator detail. The classifier fingerprint "deleted state.db-wal or state.db-shm" is unchanged. The cause table (`hermes_state_user_copy`, feeding the CLI banner, TUI/Desktop RPC error and the gateway home-channel notice) and the chat explainer carry the same first steps; the gateway notice no longer hardcodes `doctor --fix` + `gateway restart` for every non-corrupt cause, which for a held retired generation is the second-writer trap. New user-guide page `session-storage-recovery.md` (registered in sidebars, linked from the guard text, the developer state-db-recovery page and the sessions guide): the three steps, the do-nots, why maintenance refuses, and what the files beside state.db are (retired-wal captures + manifest.json, pre-update-emergency backups, corrupt backups, snapshots). --- agent/turn_explainers.py | 13 ++- gateway/run_notifications.py | 5 +- hermes_cli/console_engine.py | 4 + hermes_cli/sessions_cmd.py | 11 ++ hermes_cli/subcommands/sessions.py | 8 +- hermes_state_errors.py | 18 ++- hermes_state_holders.py | 37 ++++++ hermes_state_user_copy.py | 12 +- .../test_sessions_held_store_gate.py | 88 ++++++++++++++ .../docs/developer-guide/state-db-recovery.md | 4 + .../user-guide/session-storage-recovery.md | 107 ++++++++++++++++++ website/docs/user-guide/sessions.md | 5 +- website/sidebars.ts | 1 + 13 files changed, 297 insertions(+), 16 deletions(-) create mode 100644 tests/hermes_cli/test_sessions_held_store_gate.py create mode 100644 website/docs/user-guide/session-storage-recovery.md diff --git a/agent/turn_explainers.py b/agent/turn_explainers.py index 16f767eff5..c07b28a5ef 100644 --- a/agent/turn_explainers.py +++ b/agent/turn_explainers.py @@ -124,10 +124,13 @@ _PERSISTENCE_CAUSE_EXPLANATIONS: Dict[str, str] = { "in the log." ), "deleted_wal": ( - "the session database was changed or replaced while Hermes was running, so this " - "message was not saved (a copy is kept in {home}/sessions/). Stop Hermes " - "(`hermes {profile_arg}gateway stop`), run `hermes {profile_arg}doctor`, then start " - "it again and send your message once more. Advanced recovery steps are in the log." + "another Hermes process still holds an old copy of the session database's write-ahead " + "log, so Hermes stopped writing to keep the file safe and this message was not saved (a " + "copy is kept in {home}/sessions/). Nothing is lost. Quit every Hermes process on this " + "profile (Desktop app, `hermes {profile_arg}gateway stop`, dashboard, cron), run " + "`hermes {profile_arg}doctor` — it names any process still holding the log — then start " + "Hermes again and send your message once more. Do not run `doctor --fix` or delete " + "any state.db files while they run. Guide: {recovery_docs}" ), "corrupt": ( "the turn was stopped because the state database " @@ -371,6 +374,7 @@ class TurnExplainersMixin: body = body.format(model=model or "The model") if body is None and reason == "session_persistence_failed": from hermes_constants import display_hermes_home, profile_cli_selector + from hermes_state_errors import STORAGE_RECOVERY_DOCS_URL # Copy-pasteable, so pin every `hermes` command to the profile whose store failed: # a multi-profile backend (Desktop serve) hosts sessions whose state.db is NOT the @@ -381,6 +385,7 @@ class TurnExplainersMixin: ) .replace("{home}", display_hermes_home()) .replace("{profile_arg}", profile_cli_selector()) + .replace("{recovery_docs}", STORAGE_RECOVERY_DOCS_URL) ) if persistence_cause in ("corrupt", "fts_index"): from hermes_constants import get_default_hermes_root diff --git a/gateway/run_notifications.py b/gateway/run_notifications.py index 412667a2fb..9d7c3fbd64 100644 --- a/gateway/run_notifications.py +++ b/gateway/run_notifications.py @@ -943,10 +943,11 @@ class GatewayNotificationsMixin: else: from hermes_state_user_copy import describe_storage_failure failure = describe_storage_failure(error) + # The cause table owns the remedy: for a held retired-WAL generation a bare `doctor --fix` + # is the second-writer trap this notice used to send users into (#110054). message = ( "⚠️ Session database unavailable — messages may not be saved and /resume will be " - f"empty. Cause: {failure.gloss}. Run `hermes {profile_arg}doctor --fix` on the " - f"gateway machine, then `hermes {profile_arg}gateway restart`." + f"empty. Cause: {failure.gloss}. {failure.action}" ) logger.warning("Broadcasting state.db failure warning to home channels: %s", error) from gateway.warning_notifications import present_notification diff --git a/hermes_cli/console_engine.py b/hermes_cli/console_engine.py index 89c598ff88..58ca3248e5 100644 --- a/hermes_cli/console_engine.py +++ b/hermes_cli/console_engine.py @@ -708,6 +708,10 @@ def _sessions_rename(_engine: HermesConsoleEngine, args: list[str]) -> None: def _sessions_optimize(_engine: HermesConsoleEngine, args: list[str]) -> None: _expect_no_args(args, "sessions optimize") with _session_db(read_only=False) as db: + from hermes_state_holders import held_store_refusal + refusal = held_store_refusal(db.db_path, command="optimize", force_hint="`hermes sessions optimize --force`") + if refusal: + raise ConsoleCommandError(refusal) print(f"Optimized {db.vacuum()} FTS index(es).") diff --git a/hermes_cli/sessions_cmd.py b/hermes_cli/sessions_cmd.py index 958d42dc91..b5222020e0 100644 --- a/hermes_cli/sessions_cmd.py +++ b/hermes_cli/sessions_cmd.py @@ -1002,6 +1002,11 @@ def _print_empty_store(action: str, args) -> None: print("No sessions found.") +# VACUUM, the FTS-layout rebuild and bulk deletes rewrite the store; underneath a live gateway/Desktop/cron +# writer that is the second-writer class behind the retired-WAL refusal (#110054). `--force` is the override. +_HELD_STORE_ACTIONS = frozenset({"optimize", "optimize-storage", "prune"}) + + def cmd_sessions(args, sessions_parser=None): action = args.sessions_action pre = _PRE_DB_HANDLERS.get(action) @@ -1024,6 +1029,12 @@ def cmd_sessions(args, sessions_parser=None): if handler is None: sessions_parser.print_help() return + if action in _HELD_STORE_ACTIONS and not getattr(args, "dry_run", False) and not getattr(args, "force", False): + from hermes_state_holders import held_store_refusal + refusal = held_store_refusal(db.db_path, command=action) + if refusal: + print(refusal) + return 1 try: return handler(db, args) except sqlite3.OperationalError as e: diff --git a/hermes_cli/subcommands/sessions.py b/hermes_cli/subcommands/sessions.py index d3d26390d1..33306507f3 100644 --- a/hermes_cli/subcommands/sessions.py +++ b/hermes_cli/subcommands/sessions.py @@ -117,6 +117,8 @@ def build_sessions_parser(subparsers, *, cmd_sessions: Callable) -> None: "opened and never used (no messages, tokens, tool calls or title) " "and are older than AGE (default 30 days). Ordinary prune can " "never reach these — it only ever selects ended sessions") + _flag(sessions_prune, "--force", + help="Run even while another Hermes process (gateway, Desktop, dashboard, cron) holds state.db — rewriting the store under a live writer can leave every agent refusing turns until all writers are stopped") sessions_archive = sessions_subparsers.add_parser( "archive", help="Bulk-archive (soft-hide) sessions matching filters — no deletion") @@ -124,8 +126,10 @@ def build_sessions_parser(subparsers, *, cmd_sessions: Callable) -> None: sessions_archive, "Only archive sessions older than AGE (duration like '5h'/'2d', " "bare number of days, or ISO timestamp)") - sessions_subparsers.add_parser( + sessions_optimize = sessions_subparsers.add_parser( "optimize", help="Reclaim disk space: merge FTS5 segments + VACUUM (no data change)") + _flag(sessions_optimize, "--force", + help="Run even while another Hermes process (gateway, Desktop, dashboard, cron) holds state.db — rewriting the store under a live writer can leave every agent refusing turns until all writers are stopped") sessions_clean_markers = sessions_subparsers.add_parser("clean-markers", help="Permanently clear stale tool-call marker content left by sessions from before #78148", @@ -156,6 +160,8 @@ def build_sessions_parser(subparsers, *, cmd_sessions: Callable) -> None: help="Skip the final VACUUM (index is rebuilt but freed pages aren't returned to the OS until a later VACUUM)") _flag(sessions_optimize_storage, "--yes", "-y", default=False, help="Skip the disk-space confirmation prompt") + _flag(sessions_optimize_storage, "--force", + help="Run even while another Hermes process (gateway, Desktop, dashboard, cron) holds state.db — rewriting the store under a live writer can leave every agent refusing turns until all writers are stopped") sessions_repair = sessions_subparsers.add_parser( "repair", help="Repair a malformed state.db schema so hidden sessions reappear", diff --git a/hermes_state_errors.py b/hermes_state_errors.py index 6856610439..0ea726ba08 100644 --- a/hermes_state_errors.py +++ b/hermes_state_errors.py @@ -173,12 +173,20 @@ _STATE_DB_REPLACED_MSG = ( "writes to this file. Divert transcripts to sessions/.jsonl (and the " "gateway pending_messages spool) and restore or reopen after operator intervention." ) +STORAGE_RECOVERY_DOCS_URL = "https://hermes-agent.nousresearch.com/docs/user-guide/session-storage-recovery" + +# Two layers (#110054): the first sentence is for the person reading a chat bubble or a banner (what +# happened, nothing is lost, the one thing to do); the rest is the operator detail. The phrase +# "deleted state.db-wal or state.db-shm" is the classifier's RPC-wrapped fingerprint — keep it. _DELETED_WAL_GENERATION_MSG = ( - "FATAL: a live process holds a deleted state.db-wal or state.db-shm " - "inode while the path names a different (or missing) generation. " - "Refusing to open or write so a second WAL cannot be minted. " - "Stop the gateway, dashboard, and cron writers that hold the deleted " - "sidecar, then reopen. Do not delete the WAL yourself. " + "FATAL: session storage stopped writing because another Hermes process still holds a deleted " + "state.db-wal or state.db-shm inode (an old copy of the write-ahead log). Nothing is lost: quit " + "every Hermes process on this profile (Desktop app, gateway, dashboard, cron), run `hermes doctor` " + "(it names the processes still holding the log), then start Hermes again. Do not delete the WAL " + "yourself and do not run `hermes doctor --fix` while they are running. " + f"Guide: {STORAGE_RECOVERY_DOCS_URL} " + "Detail: the path names a different (or missing) generation than the one this process holds " + "open; opening or writing through it would mint a second WAL (split-brain). " "database.journal_mode: delete is operator containment, not a new default." ) diff --git a/hermes_state_holders.py b/hermes_state_holders.py index 1f1fe4c246..bb739004f3 100644 --- a/hermes_state_holders.py +++ b/hermes_state_holders.py @@ -335,6 +335,43 @@ def foreign_state_db_holders(db_path: Path) -> List[Tuple[int, str]]: return holders +def held_store_refusal(db_path: Path, *, command: str, force_hint: Optional[str] = "--force") -> Optional[str]: + """Operator-facing refusal for structural maintenance (VACUUM, index rebuild, bulk delete) while another + process holds ``db_path`` or a WAL sidecar; ``None`` when the store is provably quiet. + + Running ``hermes sessions optimize-storage`` underneath a fleet of live gateways put every agent into + the retired-WAL refusal until all writers were stopped (#110054). Same fail-closed scan doctor and + repair use: an incomplete scan refuses too, it never reads as an all-clear. + """ + holders = foreign_state_db_holders(db_path) + if not holders: + return None + from hermes_constants import profile_cli_selector + from hermes_state_errors import STORAGE_RECOVERY_DOCS_URL + + by_pid: dict[int, Set[str]] = {} + unknown: List[str] = [] + for pid, target in holders: + if pid <= 0 or target.startswith("uninspectable"): + unknown.append(target) + else: + by_pid.setdefault(pid, set()).add(Path(target.removesuffix(" (deleted)")).name) + lines = [f"Refusing `hermes sessions {command}`: another process is using {db_path}."] + lines += [f" {describe_holder_pid(pid)}: {', '.join(sorted(by_pid[pid]))}" for pid in sorted(by_pid)] + if unknown: + lines.append(f" cannot prove the database is quiet (holder scan incomplete: {unknown[0][:120]})") + profile_arg = profile_cli_selector() + lines += [ + "Rewriting the database under a live writer is how every agent ends up refusing turns with the " + "retired state.db-wal error. Nothing is lost.", + f"Stop them first (`hermes {profile_arg}gateway stop`, quit the Desktop app, pause cron), then re-run.", + ] + if force_hint: + lines.append(f"Override with {force_hint} if you accept the risk.") + lines.append(f"Recovery guide: {STORAGE_RECOVERY_DOCS_URL}") + return "\n".join(lines) + + def live_writer_holds_db( db_path: Path, *, diff --git a/hermes_state_user_copy.py b/hermes_state_user_copy.py index 6aa8923314..2fecd87939 100644 --- a/hermes_state_user_copy.py +++ b/hermes_state_user_copy.py @@ -9,7 +9,7 @@ from __future__ import annotations from dataclasses import dataclass -from hermes_state_errors import classify_persistence_error, is_disk_full_error +from hermes_state_errors import STORAGE_RECOVERY_DOCS_URL, classify_persistence_error, is_disk_full_error @dataclass(frozen=True) @@ -55,10 +55,16 @@ _STORAGE_FAILURES: dict[str, tuple[str, str, str]] = { "the session database file was replaced while Hermes was running", "Stop Hermes (`hermes {profile_arg}gateway stop`), run `hermes {profile_arg}doctor`, then start it again.", ), + # Code stays `storage_replaced` (GUI clients key on it); the copy names the real remedy: every writer + # on the profile must stop, doctor names the ones still holding the retired log (#110054). "deleted_wal": ( "storage_replaced", - "the session database file was changed or replaced while Hermes was running", - "Stop Hermes (`hermes {profile_arg}gateway stop`), run `hermes {profile_arg}doctor`, then start it again.", + "another Hermes process still holds an old copy of the session database's write-ahead log, " + "so Hermes stopped writing to keep the file safe", + "Nothing is lost. Quit every Hermes process on this profile (Desktop app, " + "`hermes {profile_arg}gateway stop`, dashboard, cron), run `hermes {profile_arg}doctor` — it names " + "any process still holding the log — then start Hermes again. Do not run `doctor --fix` or delete " + "any state.db files while they run. Guide: " + STORAGE_RECOVERY_DOCS_URL, ), "compression": ( "storage_busy", diff --git a/tests/hermes_cli/test_sessions_held_store_gate.py b/tests/hermes_cli/test_sessions_held_store_gate.py new file mode 100644 index 0000000000..8cf6388ca2 --- /dev/null +++ b/tests/hermes_cli/test_sessions_held_store_gate.py @@ -0,0 +1,88 @@ +"""`hermes sessions optimize|optimize-storage|prune` refuse while another process holds state.db (#110054). + +Running the storage rewrite underneath a fleet of live gateways put every agent into the retired-WAL +refusal; the command now runs the same fail-closed holder scan doctor/repair use, names each holder as +``PID N (command)`` and exits non-zero, with ``--force`` as the operator override. Driven through the +production entry point ``cmd_sessions`` with a REAL second process holding the store. +""" + +import subprocess +import sys +from argparse import Namespace + +import pytest + +import hermes_cli.sessions_cmd as sessions_cmd + +pytestmark = pytest.mark.skipif(sys.platform == "win32", reason="holder scan is unavailable on Windows") + + +_HOLDER = ( + "import sqlite3, sys, time\n" + "conn = sqlite3.connect(sys.argv[1])\n" + "conn.execute('SELECT count(*) FROM sqlite_master')\n" + "print('ready', flush=True)\n" + "sys.stdin.readline()\n" +) + + +@pytest.fixture +def state_db(monkeypatch, tmp_path): + import hermes_state + from hermes_state import SessionDB + + db_path = tmp_path / "state.db" + monkeypatch.setattr(hermes_state, "_default_db_path", lambda: db_path) + seed = SessionDB(db_path=db_path) + seed.create_session("seed", "cli") + seed.append_message("seed", "user", "hello") + seed.close() + return db_path + + +@pytest.fixture +def foreign_holder(state_db): + proc = subprocess.Popen( + [sys.executable, "-c", _HOLDER, str(state_db)], + stdin=subprocess.PIPE, stdout=subprocess.PIPE, text=True, + ) + assert proc.stdout.readline().strip() == "ready" + try: + yield proc + finally: + if proc.poll() is None: + proc.stdin.close() + proc.wait(timeout=10) + + +def _optimize(force=False): + return Namespace(sessions_action="optimize", force=force) + + +def test_optimize_refuses_and_names_the_holder_until_forced(state_db, foreign_holder, capsys): + assert sessions_cmd.cmd_sessions(_optimize()) == 1 + out = capsys.readouterr().out + assert f"PID {foreign_holder.pid} (" in out + assert "Refusing `hermes sessions optimize`" in out and "--force" in out + assert "Optimized" not in out + + assert sessions_cmd.cmd_sessions(_optimize(force=True)) is None + assert "Optimized" in capsys.readouterr().out + + +def test_prune_preview_passes_the_delete_waits_for_a_quiet_store(state_db, foreign_holder, capsys): + # A preview never rewrites anything, so it is answered even while the holder lives. + prune_preview = Namespace(sessions_action="prune", dry_run=True, yes=False, force=False, never_active=False, + include_archived=False, include_pinned=False, + **{name: None for name in sessions_cmd._FILTER_ARGS}) + assert sessions_cmd.cmd_sessions(prune_preview) is None + assert "Refusing" not in capsys.readouterr().out + prune_preview.dry_run = False + prune_preview.yes = True + assert sessions_cmd.cmd_sessions(prune_preview) == 1 + assert "Refusing `hermes sessions prune`" in capsys.readouterr().out + # Control: once the holder exits the same command runs. + foreign_holder.stdin.close() + foreign_holder.wait(timeout=10) + assert sessions_cmd.cmd_sessions(prune_preview) is None + assert "Refusing" not in capsys.readouterr().out diff --git a/website/docs/developer-guide/state-db-recovery.md b/website/docs/developer-guide/state-db-recovery.md index 7a7af15bd6..c33820b546 100644 --- a/website/docs/developer-guide/state-db-recovery.md +++ b/website/docs/developer-guide/state-db-recovery.md @@ -5,6 +5,10 @@ description: "How Hermes recovers state.db when the FTS index or the file itself # State database and FTS recovery +For the user-facing walkthrough (what to stop, what the files beside `state.db` are, why +maintenance commands refuse while a writer is live) see +[Session storage recovery](../user-guide/session-storage-recovery.md). + `state.db` stores two different data classes: - `sessions` and `messages` are the canonical transcript. diff --git a/website/docs/user-guide/session-storage-recovery.md b/website/docs/user-guide/session-storage-recovery.md new file mode 100644 index 0000000000..3c890a7ede --- /dev/null +++ b/website/docs/user-guide/session-storage-recovery.md @@ -0,0 +1,107 @@ +--- +title: "Session Storage Recovery" +description: "What to do when Hermes says another process holds an old copy of the session database's write-ahead log, and what the files beside state.db are" +--- + +# Session storage recovery + +Hermes keeps every conversation in one SQLite file per profile, `state.db`, with two +sidecar files SQLite manages itself: `state.db-wal` (the write-ahead log) and `state.db-shm`. +Several Hermes processes can share that file safely — the gateway, the Desktop app, the +dashboard, cron, and CLI commands all write through SQLite's own locking. + +One thing is not safe: **rewriting the store while another process is writing to it**. +When that happens, the processes still holding the *old* copy of the log stop writing on +purpose and every turn answers with a message like: + +> another Hermes process still holds an old copy of the session database's write-ahead log, +> so Hermes stopped writing to keep the file safe … + +This page is the guide that message links to. Nothing is lost when you see it; the +refusal exists precisely so nothing gets lost. + +## The fix in three steps + +1. **Quit every Hermes process on that profile.** Desktop app, gateway, dashboard, cron: + + ```bash + hermes gateway stop # add -p for a named profile + ``` + + then quit the Desktop app from its menu and stop any dashboard (`hermes dashboard --stop`) + or custom service you run. Restarting *one* of them is not enough — a single process left + holding the old log keeps every new one refusing. + +2. **Ask doctor who is still holding the log.** + + ```bash + hermes doctor # add -p for a named profile + ``` + + While anything still holds the retired log, doctor prints each holder as + `PID N (command)` with the same remedy, and skips its health probes and any `--fix` work + so it cannot become another writer. Stop the listed processes and run it again until the + line is gone. + +3. **Start Hermes again** (one process first — the gateway or the Desktop app) and send your + message once more. Your conversation resumes from where it stopped. + +## Do not + +- **Do not run `hermes doctor --fix` while the processes are running.** Doctor refuses the + checkpoint while it can see a holder, but on a host where it cannot inspect processes the + fix path is exactly the second writer that caused the problem. +- **Do not delete `state.db-wal` or `state.db-shm`.** The log holds committed conversations + that are not yet in `state.db`. Deleting it is the one action that turns a refusal into + real data loss. +- **Do not copy `state.db` alone.** The three files are one image. Use a snapshot + (`hermes backup`) or `hermes sessions recover`, never `cp state.db somewhere/`. +- **Do not ask the agent to fix it.** The agent's own session is in the same store; it will + hit the same refusal. + +## Maintenance commands refuse while someone is writing + +`hermes sessions optimize`, `hermes sessions optimize-storage` and `hermes sessions prune` +rewrite the store (VACUUM, a full-text index rebuild, bulk deletes). Running one of them under +a live gateway is how a fleet of agents ends up in the refusal above, so they now check first +and refuse while another process holds the database: + +```text +Refusing `hermes sessions optimize-storage`: another process is using ~/.hermes/state.db. + PID 41230 (hermes gateway run): state.db, state.db-shm, state.db-wal + PID 41355 (hermes serve --profile work): state.db-wal +Rewriting the database under a live writer is how every agent ends up refusing turns with the +retired state.db-wal error. Nothing is lost. +Stop them first (`hermes gateway stop`, quit the Desktop app, pause cron), then re-run. +Override with --force if you accept the risk. +``` + +`--dry-run` previews are never blocked. `--force` runs anyway — use it only when you know +the listed processes are idle (a reader you started yourself, for example). The same check +runs when you type `sessions optimize` in the Desktop console. + +## Files you may find beside `state.db` + +| File or directory | What it is | What to do | +|---|---|---| +| `state.db-wal`, `state.db-shm` | SQLite's live write-ahead log and its shared-memory index. A large `-wal` is normal while the gateway or Desktop is running. | Leave them alone. They shrink on their own at the next checkpoint. | +| `state.db.retired-wal--/` | A capture Hermes made of the log copy a process was still holding when it refused to write, plus a `manifest.json` describing it. Forensic evidence, not a backup you restore blindly. | Keep it. If conversations from just before the incident are missing after recovery, attach the directory to a bug report; a maintainer can tell from `manifest.json` whether those frames belong on top of the current file. | +| `state.db.pre-update-emergency-.bak` | A snapshot the Desktop updater takes before it touches the store. | Keep it until you have used the updated app for a while. Restore only with every Hermes process stopped: `hermes sessions recover --source --inspect-only` first. | +| `state.db.corrupt..bak`, `*.malformed-backup` | Copies of a file Hermes found damaged before it repaired or quarantined it. | Do not restore them over `state.db` — they are the same damage. Keep for a report; safe to delete once you are back to normal. | +| `state-snapshots/` | Quick snapshots `hermes update` and `hermes backup` take. | Restore with every Hermes process stopped; see [`hermes backup`](../reference/cli-commands.md#hermes-backup). | + +## When the three steps do not work + +If every Hermes process is stopped, `hermes doctor` no longer lists a holder, and the +gateway still refuses to write when you start it, the file itself may be damaged. Stop +everything again and inspect without writing: + +```bash +hermes sessions recover --source ~/.hermes/state.db --inspect-only +``` + +`--inspect-only` never modifies the file. If it reports the store as recoverable, follow the +command it prints, or restore the newest snapshot from `state-snapshots/`. The mechanics +behind all of this are in the developer guide: +[State DB recovery](../developer-guide/state-db-recovery.md) and +[Session storage](../developer-guide/session-storage.md). diff --git a/website/docs/user-guide/sessions.md b/website/docs/user-guide/sessions.md index efc8e0e508..08a7f887ea 100644 --- a/website/docs/user-guide/sessions.md +++ b/website/docs/user-guide/sessions.md @@ -62,7 +62,10 @@ Use `/compress` when a session gets long, `/new` for a fresh thread, and `hermes sessions prune` only when you want to delete old ended sessions from storage. If `state.db` has simply grown large, start with the non-destructive option first: `hermes sessions optimize` merges FTS5 index segments and -VACUUMs the database without touching any session data. Compression reduces the active context; it is not a privacy delete. +VACUUMs the database without touching any session data. Both `optimize` and `prune` refuse +while another Hermes process (gateway, Desktop, dashboard, cron) holds `state.db` — stop it +first, or pass `--force`; see [Session storage recovery](session-storage-recovery.md). +Compression reduces the active context; it is not a privacy delete. Pass a name to `/new` (e.g. `/new payments-refactor`) to set the new session's initial title up front — useful for finding it later with `/resume ` or in the `/sessions` picker. diff --git a/website/sidebars.ts b/website/sidebars.ts index e48d8b8a01..efc6b2405a 100644 --- a/website/sidebars.ts +++ b/website/sidebars.ts @@ -52,6 +52,7 @@ const sidebars: SidebarsConfig = { ], }, 'user-guide/sessions', + 'user-guide/session-storage-recovery', 'user-guide/profiles', 'user-guide/profile-distributions', 'user-guide/multi-profile-gateways',