fix(state): pin corrupt-session recovery guidance to the failing profile

The recovery commands rendered on structural corruption — the turn explainer's
`session_persistence_failed`/corrupt body, the gateway's home-channel state.db
warning, and hermes_state_repair._persistent_repair_exhausted_error — already
interpolate the active profile's state.db path, but every `hermes ...` verb in them
was bare. A bare `hermes` follows the sticky `active_profile` file, so an operator
running the pasted `hermes doctor --fix` (or `hermes sessions recover` with a
relative source) from a named-profile incident could inspect or repair a different
profile's database (#105887).

hermes_constants.profile_cli_selector() renders `-p <name> ` for a named profile
home (default home and custom roots outside the profile tree render nothing: the
default is what a bare `hermes` already means, and a custom root is only reachable
via HERMES_HOME). Every command in the three guidance sites now carries it, and
the new `fts_index` guidance inherits the same interpolation.

Live check with HERMES_HOME=<root>/profiles/research and active_profile=other:
before `1. Run \`hermes doctor --fix\`` (targets "other"); after
`1. Run \`hermes -p research doctor --fix\`` and `hermes -p research sessions
recover --source <root>/profiles/research/state.db --inspect-only`.

Refs #105887
Reported-by: Cuttingwater
This commit is contained in:
Teknium
2026-09-11 03:07:11 -07:00
parent 3d167a8271
commit 754ecff466
4 changed files with 69 additions and 20 deletions

View File

@@ -139,10 +139,10 @@ _PERSISTENCE_CAUSE_EXPLANATIONS: Dict[str, str] = {
"reported structural corruption (the transcript would "
"have been lost on restart). Freeing disk space will "
"not help. Recovery options:\n"
"1. Run `hermes doctor --fix`\n"
"1. Run `hermes {profile_arg}doctor --fix`\n"
"2. Stop the gateway, then recover with:\n"
" hermes sessions recover --source {db_path} --inspect-only\n"
" (if it reports recoverable) hermes sessions recover "
" hermes {profile_arg}sessions recover --source {db_path} --inspect-only\n"
" (if it reports recoverable) hermes {profile_arg}sessions recover "
"--source {db_path} --output recovered-state.db\n"
" — recovery snapshots the damaged file first; do NOT "
"run `sqlite3 ... \".recover\"` against the live "
@@ -157,7 +157,7 @@ _PERSISTENCE_CAUSE_EXPLANATIONS: Dict[str, str] = {
"the turn was stopped because the session search index (FTS5) "
"is corrupt and could not be detached, so this message was not "
"saved. The message store itself is not damaged: do not run "
"recovery tools or restore a backup. Run `hermes doctor --fix` "
"recovery tools or restore a backup. Run `hermes {profile_arg}doctor --fix` "
"(or restart Hermes, which repairs the index on open), then "
"send your message again."
),
@@ -328,14 +328,14 @@ class TurnExplainersMixin:
body = _PERSISTENCE_CAUSE_EXPLANATIONS.get(
persistence_cause or "unknown", _PERSISTENCE_DEFAULT_EXPLANATION
)
if persistence_cause == "corrupt":
# Copy-pasteable, so name the store that actually failed: the agent's own
# SessionDB. A multi-profile backend (Desktop serve) hosts sessions whose
# state.db is NOT the process default, so the default would send the operator
# to inspect/repair the wrong profile's database (#105887).
from hermes_constants import get_default_hermes_root
if persistence_cause in ("corrupt", "fts_index"):
# Copy-pasteable, so name the store that actually failed and pin the profile:
# a multi-profile backend (Desktop serve) hosts sessions whose state.db is NOT
# the process default, and a bare `hermes` follows active_profile (#105887).
from hermes_constants import get_default_hermes_root, profile_cli_selector
from hermes_state import _default_db_path
body = body.replace("{profile_arg}", profile_cli_selector())
body = body.replace("{db_path}", str(db_path or _default_db_path()))
body = body.replace(
"{backups_dir}", str(get_default_hermes_root() / "backups")

View File

@@ -833,27 +833,29 @@ class GatewayNotificationsMixin:
if not error:
logger.info("state.db recovered before the home-channel warning went out; not broadcasting")
return
from hermes_constants import get_default_hermes_root
from hermes_constants import get_default_hermes_root, profile_cli_selector
from hermes_state import _default_db_path, classify_persistence_error, format_session_db_unavailable
cause = classify_persistence_error(error)
# Copy-pasteable, so name the real store and pin the profile: a bare `hermes` follows
# active_profile, which may be a different database (#105887).
profile_arg = profile_cli_selector()
if cause == "corrupt":
# Copy-pasteable, so name the real store (profiles / HERMES_HOME do not live under ~/.hermes).
db_path = _default_db_path()
backups_dir = get_default_hermes_root() / "backups"
message = (
"⚠️ Session database corruption detected. Messages may not be "
"persisted. Recovery options:\n"
"1. Run `hermes doctor --fix`\n"
f"1. Run `hermes {profile_arg}doctor --fix`\n"
"2. Stop the gateway, then recover with:\n"
f" hermes sessions recover --source {db_path} "
f" hermes {profile_arg}sessions recover --source {db_path} "
"--inspect-only\n"
" (if it reports recoverable) hermes sessions recover "
f" (if it reports recoverable) hermes {profile_arg}sessions recover "
f"--source {db_path} --output recovered-state.db\n"
" — recovery snapshots the damaged file first; do NOT run "
"`sqlite3 ... \".recover\"` against the live state.db, a "
"vulnerable sqlite3 CLI can corrupt it further\n"
f"3. Restore from a backup in {backups_dir}/\n"
"Run `hermes doctor` for sanitized diagnostics."
f"Run `hermes {profile_arg}doctor` for sanitized diagnostics."
)
elif cause == "fts_index":
# Index-scoped corruption: the message tables are not damaged, so the recover /
@@ -861,7 +863,7 @@ class GatewayNotificationsMixin:
message = (
"⚠️ Session database reported a corruption error confined to the search index "
"(FTS5); the message tables are not damaged. Messages may not be persisted until "
"it is repaired: run `hermes doctor --fix`, then restart the gateway. Do not run "
f"it is repaired: run `hermes {profile_arg}doctor --fix`, then restart the gateway. Do not run "
"recovery tools or restore a backup unless `hermes doctor` confirms damage."
)
else:

View File

@@ -391,11 +391,14 @@ def _persistent_repair_attempts_exhausted(db_path: Path) -> bool:
def _persistent_repair_exhausted_error(db_path: Path) -> str:
"""The stable operator-facing diagnostic for an exhausted repair budget."""
"""The stable operator-facing diagnostic for an exhausted repair budget. The ``hermes`` commands
carry the profile selector: a bare ``hermes`` follows ``active_profile`` (#105887)."""
from hermes_constants import profile_cli_selector
profile_arg = profile_cli_selector()
return (f"automatic repair has already failed {_MAX_PERSISTENT_REPAIR_ATTEMPTS} times on this exact file — the "
f"corruption is beyond the schema/FTS repair strategies (likely b-tree page damage). Manual recovery "
f"required: restore a backup, or salvage with `hermes sessions recover --source {db_path} "
f"--inspect-only`, then (if it reports recoverable) `hermes sessions recover --source {db_path} "
f"required: restore a backup, or salvage with `hermes {profile_arg}sessions recover --source {db_path} "
f"--inspect-only`, then (if it reports recoverable) `hermes {profile_arg}sessions recover --source {db_path} "
f"--output recovered-state.db` (recovery snapshots the damaged file first, then runs the page-level "
f"`.recover` lane on the copy; do NOT point a raw `sqlite3` shell at the live database). "
f"Delete {_repair_ledger_path(db_path).name} to force another automatic attempt.")

View File

@@ -123,3 +123,47 @@ def test_format_turn_completion_locked_still_advises_retry():
)
assert "busy" in explanation
assert "send it again" in explanation
def test_corrupt_guidance_pins_the_failing_profile(tmp_path, monkeypatch):
"""#105887: every `hermes ...` command in the recovery guidance (turn explainer, gateway
home-channel notice, exhausted-repair diagnostic) carries the active profile selector and
names that profile's state.db. A bare `hermes` follows the sticky ``active_profile`` file,
so with another profile active the operator would repair the wrong database."""
import asyncio
import gateway.run as gateway_run
from hermes_state import _default_db_path
from hermes_state_repair import _persistent_repair_exhausted_error
from run_agent import AIAgent
root = tmp_path / "hermes"
home = root / "profiles" / "research"
home.mkdir(parents=True)
(root / "config.yaml").write_text("")
(root / "active_profile").write_text("other\n")
monkeypatch.setenv("HERMES_HOME", str(home))
explanation = AIAgent._format_turn_completion_explanation("session_persistence_failed", "corrupt")
commands = [line.strip() for line in explanation.splitlines() if "hermes " in line]
assert commands and all("hermes -p research " in line for line in commands), commands
# The conftest pins hermes_state.DEFAULT_DB_PATH, so the store named is whatever the
# process resolves — the contract is "the same path the runtime would open".
assert f"--source {_default_db_path()} " in explanation
runner = object.__new__(gateway_run.GatewayRunner)
runner._session_db_init_error = "database disk image is malformed"
sent = []
monkeypatch.setattr(runner, "_home_channel_transports", lambda: [("telegram", {}, "home-chat", object())])
async def _capture_send(_platform, _home, _transport, message, _log_fmt):
sent.append(message)
monkeypatch.setattr(runner, "_send_home_channel_message", _capture_send)
asyncio.run(runner._send_session_db_warning_notifications())
notice_commands = [line.strip() for line in sent[0].splitlines() if "hermes " in line]
assert notice_commands and all("hermes -p research " in line for line in notice_commands), notice_commands
assert f"--source {_default_db_path()} " in sent[0]
exhausted = _persistent_repair_exhausted_error(home / "state.db")
assert "`hermes -p research sessions recover --source" in exhausted