Activation reaches plugin discovery before the application dependencies exist. Give PM its own locked Python project and runtime so it can install or repair the application without importing that dependency tree. Keep PM outside the application workspace. A shared uv workspace resolves the application graph and cannot provide this isolation. Route mutations through an isolated worker and preserve transaction callbacks, cancellation, custom package registrations, and correlated receipts. Use the same runtime builder for source installs and packaged payloads. Keep offline wheelhouse support in that builder. Nix builds the independent PM lock as a separate derivation. Refuse lazy-disabled bootstrap before installing tools or dependencies. Move first-party YAML readers and writers to ruamel. Keep the application lock's transitive PyYAML requirements for third-party packages. Verification: - Focused canonical Python suite: 177 passed, 1 host-gated skip. - Electron backend probes: 12 passed. Electron typecheck passed. - Both uv locks, scoped lint, Bash syntax, and whitespace checks passed. - Cold activation, corrupt-app repair, offline staging, and relocation ran. - Built and exercised the Nix PM runtime and standalone YAML merge script. Six broader caller test files retain the same 24 failing test IDs as an archive of HEAD. The existing real-home guard blocks those tests before they can exercise the affected paths. No full-suite pass is claimed. Native Windows signing and full Bionic package execution remain unverified.
2019 lines
83 KiB
Python
2019 lines
83 KiB
Python
"""Backup and import commands for hermes CLI."""
|
|
|
|
import json
|
|
import logging
|
|
import os
|
|
import shutil
|
|
import sqlite3
|
|
import stat
|
|
import sys
|
|
import tempfile
|
|
import threading
|
|
import time
|
|
import zipfile
|
|
from contextlib import contextmanager, suppress
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
from typing import Any, Dict, List, Optional, Tuple
|
|
|
|
from hermes_constants import (
|
|
_get_platform_default_hermes_home, get_default_hermes_root, get_hermes_home, display_hermes_home,
|
|
)
|
|
from hermes_state_dbfile import RETIRED_GENERATION_DIR_SUFFIX
|
|
from utils import (
|
|
_preserve_file_mode, _preserve_file_owner, _restore_file_mode, _restore_file_owner, atomic_replace,
|
|
)
|
|
|
|
from hermes_cli.archive_safe import normalize_archive_parts
|
|
from hermes_cli.home_data_layout import PM_RUNTIME_ROOT_DIRS, profile_root_entry
|
|
from hermes_cli.sizefmt import format_bytes as _format_size
|
|
|
|
from hermes_cli.backup_restore import (
|
|
_count_session_rows,
|
|
_default_new_file_mode,
|
|
_detect_prefix,
|
|
_extract_member_atomically,
|
|
_import_db_member,
|
|
_safe_restore_db,
|
|
_validate_backup_zip,
|
|
)
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# --- Exclusion rules ---
|
|
|
|
# Where ``hermes backup --quick`` / ``/snapshot`` / the pre-update safety net write state
|
|
# snapshots (see ``create_quick_snapshot``); defined here because the exclusion set needs it.
|
|
_QUICK_SNAPSHOTS_DIR = "state-snapshots"
|
|
|
|
# Directory names to skip (matched against each path component). ``hermes-agent`` only matches at
|
|
# the root (``_should_exclude``) so skill dirs like ``skills/.../hermes-agent/`` survive. The
|
|
# dependency/cache entries matter: one plugin venv or pip/uv cache under HERMES_HOME walked
|
|
# file-by-file balloons a backup to hundreds of thousands of entries ("backup stuck for days").
|
|
# Mostly mirrors ``agent.skill_utils.EXCLUDED_SKILL_DIRS``; ``.cache`` is backup-only. ``.archive``
|
|
# is deliberately NOT excluded: the curator's ``skills/.archive/`` holds restorable user skills.
|
|
_EXCLUDED_DIRS = {
|
|
"hermes-agent", # the codebase repo — re-clone instead
|
|
"__pycache__", # bytecode caches — regenerated on import
|
|
".git", # nested git dirs (profiles shouldn't have these, but safety)
|
|
"node_modules", # js deps — reinstalled on demand
|
|
"backups", # prior auto-backups — don't nest backups exponentially
|
|
_QUICK_SNAPSHOTS_DIR, # each holds a full state.db copy — same reason as ``backups``
|
|
"checkpoints", # session-hash-keyed trajectory caches — regenerated, don't port
|
|
# Live CDP browser profiles: Chromium holds their SQLite DBs exclusively locked while running
|
|
# and sqlite3.backup() retries SQLITE_BUSY forever, hanging the backup. Regenerable anyway.
|
|
"browser-profiles",
|
|
# Real-profile browsing snapshot (browser.use_real_profile): copies of the user's Cookies /
|
|
# Login Data — a credential store that must NOT enter an archive. Regenerated on next launch.
|
|
"browser-profile",
|
|
# Python dependency trees (plugin / MCP-server venvs) — regenerated by reinstalling.
|
|
".venv", "venv", "site-packages",
|
|
# Tool / build caches — all regeneratable.
|
|
".cache", ".tox", ".nox", ".pytest_cache", ".mypy_cache", ".ruff_cache",
|
|
}
|
|
|
|
# Hermes-managed runtime downloads (GGUF models, llama.cpp runtimes, managed Node): re-downloaded
|
|
# on demand and routinely tens to hundreds of GB. Matched ONLY at the root of HERMES_HOME and at
|
|
# ``profiles/<name>/`` — a deeper dir of the same name (a skill's ``models/``) is user data.
|
|
_EXCLUDED_ROOT_DIRS = {"models", "runtimes", "node"} | PM_RUNTIME_ROOT_DIRS
|
|
|
|
|
|
def _in_excluded_root_dir(rel_path: Path) -> bool:
|
|
"""True when *rel_path* is, or sits inside, a managed runtime tree at a profile-home root."""
|
|
return profile_root_entry(rel_path.parts) in _EXCLUDED_ROOT_DIRS
|
|
|
|
|
|
# SQLite sidecars are excluded because ``*.db`` is snapshotted via ``sqlite3.backup()``:
|
|
# shipping the live WAL/SHM/journal alongside would pair a fresh snapshot with stale sidecar
|
|
# state and produce a torn restore on next open. They are regenerated on first connection.
|
|
_SQLITE_SIDECAR_SUFFIXES = (".db-wal", ".db-shm", ".db-journal")
|
|
_EXCLUDED_SUFFIXES = (".pyc", ".pyo", *_SQLITE_SIDECAR_SUFFIXES)
|
|
|
|
# File names to skip (runtime state that's meaningless on another machine)
|
|
_EXCLUDED_NAMES = {".backup.lock", "gateway.pid", "cron.pid"}
|
|
|
|
# The desktop updater's pre-flight drops ``state.db.pre-update-emergency-<ts>.bak`` at the root
|
|
# — a backup artifact like ``backups/``. Prefix-matched because the name carries a timestamp;
|
|
# a plain ``.bak`` suffix rule would drop user files.
|
|
# Retired-WAL capture dirs (``<name>.retired-wal-<ts>-<pid>/``) are excluded whole: a
|
|
# ``sqlite3.backup()`` snapshot of the live db paired with the captured ``-wal`` is exactly the
|
|
# torn-restore hazard the sidecar exclusion below exists to prevent, and the capture is an
|
|
# operator-recovery artifact that must move as a unit (manifest + image + WAL), never partially.
|
|
_EXCLUDED_PREFIXES = (
|
|
"state.db.pre-update-emergency-",
|
|
f"state.db{RETIRED_GENERATION_DIR_SUFFIX}",
|
|
)
|
|
|
|
# Files ``hermes import`` must never overwrite, matched by basename so root and named profiles are
|
|
# both covered. They hold runtime state namespaced to the SOURCE machine: ``gateway_state.json``
|
|
# drives the container-boot reconciler (a foreign value leaves the gateway stuck "starting" and
|
|
# disconnected from the Nous portal); PID/lock/registry files reference source PIDs. Mirrors
|
|
# ``container_boot._STALE_RUNTIME_FILES``; import filters too because older backups predate the
|
|
# backup-side exclusions.
|
|
_IMPORT_SKIP_NAMES = {"gateway_state.json", "gateway.pid", "cron.pid", "gateway.lock", "processes.json"}
|
|
|
|
# zipfile.open() drops Unix mode bits on extract; restore tightens these to 0600.
|
|
# vault.key / vault.json.enc: the local credential vault (agent/vault_store.py)
|
|
# IS included in backups (user-entered secrets, not regenerable — unlike the
|
|
# excluded browser-profile/ snapshot) but must come back owner-only.
|
|
_SECRET_FILE_NAMES = {".env", "auth.json", "state.db", "vault.key", "vault.json.enc"}
|
|
|
|
# Reserved archive subtree for memory-provider state OUTSIDE HERMES_HOME (e.g. ~/.honcho, via
|
|
# MemoryProvider.backup_paths()), stored and restored relative to the user's home; paths not
|
|
# under home are skipped.
|
|
_EXTERNAL_PREFIX = "_external/"
|
|
|
|
|
|
class BackupInProgressError(RuntimeError):
|
|
"""Raised when another process already owns the Hermes backup slot."""
|
|
|
|
|
|
class _SQLiteSnapshotError(RuntimeError):
|
|
pass
|
|
|
|
|
|
class _SQLiteBackupTimeout(RuntimeError):
|
|
"""Raised when a SQLite snapshot remains busy past its deadline."""
|
|
|
|
|
|
@contextmanager
|
|
def _backup_operation_lock(hermes_home: Path, timeout_seconds: float = 0.25):
|
|
"""Acquire one cross-process backup slot for full and quick snapshots."""
|
|
lock_path = hermes_home / ".backup.lock"
|
|
lock_path.parent.mkdir(parents=True, exist_ok=True)
|
|
handle = lock_path.open("a+b")
|
|
acquired = False
|
|
deadline = time.monotonic() + max(0.0, timeout_seconds)
|
|
try:
|
|
if os.name == "nt":
|
|
import msvcrt
|
|
if lock_path.stat().st_size == 0:
|
|
handle.write(b" ")
|
|
handle.flush()
|
|
def _lock_op(flag: int) -> None:
|
|
handle.seek(0)
|
|
msvcrt.locking(handle.fileno(), flag, 1)
|
|
lock_flag, unlock_flag = msvcrt.LK_NBLCK, msvcrt.LK_UNLCK
|
|
else:
|
|
import fcntl
|
|
def _lock_op(flag: int) -> None:
|
|
fcntl.flock(handle.fileno(), flag)
|
|
lock_flag, unlock_flag = fcntl.LOCK_EX | fcntl.LOCK_NB, fcntl.LOCK_UN
|
|
while not acquired:
|
|
try:
|
|
_lock_op(lock_flag)
|
|
acquired = True
|
|
except OSError:
|
|
if time.monotonic() >= deadline:
|
|
raise BackupInProgressError("another Hermes backup is already running")
|
|
time.sleep(0.05)
|
|
yield
|
|
finally:
|
|
if acquired:
|
|
with suppress(OSError):
|
|
_lock_op(unlock_flag)
|
|
handle.close()
|
|
|
|
|
|
@contextmanager
|
|
def _atomic_output_path(final_path: Path):
|
|
"""Yield a hidden sibling path and publish it only after a clean close."""
|
|
partial_path = final_path.with_name(f".{final_path.name}.{os.getpid()}-{threading.get_ident()}.partial")
|
|
partial_path.unlink(missing_ok=True)
|
|
try:
|
|
yield partial_path
|
|
os.replace(partial_path, final_path)
|
|
except BaseException:
|
|
partial_path.unlink(missing_ok=True)
|
|
raise
|
|
|
|
|
|
def _collect_memory_provider_external_paths() -> List[Path]:
|
|
"""Existing paths the active memory provider declares via ``backup_paths()``; ``[]`` on any
|
|
provider failure (backup must never fail because of a flaky plugin)."""
|
|
try:
|
|
from plugins.memory import _get_active_memory_provider, load_memory_provider
|
|
active = _get_active_memory_provider()
|
|
provider = load_memory_provider(active) if active else None
|
|
except Exception:
|
|
return []
|
|
if provider is None:
|
|
return []
|
|
try:
|
|
declared = provider.backup_paths() or []
|
|
except Exception as exc:
|
|
logger.warning("backup_paths() failed for memory provider %r: %s", active, exc)
|
|
return []
|
|
out: Dict[Path, Path] = {} # resolved -> first declared spelling
|
|
for raw in declared:
|
|
try:
|
|
p = Path(raw).expanduser()
|
|
except Exception:
|
|
continue
|
|
if not p.exists():
|
|
continue
|
|
try:
|
|
resolved = p.resolve()
|
|
except (OSError, ValueError):
|
|
continue
|
|
out.setdefault(resolved, p)
|
|
return list(out.values())
|
|
|
|
|
|
def _iter_external_files(base: Path) -> List[Path]:
|
|
"""Regular files under *base* (a file or a directory), skipping symlinks, caches, and pyc."""
|
|
if base.is_file() and not base.is_symlink():
|
|
return [base]
|
|
if not base.is_dir():
|
|
return []
|
|
files: List[Path] = []
|
|
for dirpath, dirnames, filenames in os.walk(base, followlinks=False):
|
|
dirnames[:] = [d for d in dirnames if d not in _EXCLUDED_DIRS]
|
|
files.extend(fp for fp in (Path(dirpath) / f for f in filenames)
|
|
if not (fp.is_symlink() or fp.name in _EXCLUDED_NAMES
|
|
or fp.name.endswith(_EXCLUDED_SUFFIXES)))
|
|
return files
|
|
|
|
|
|
def _should_exclude(rel_path: Path) -> bool:
|
|
"""Return True if *rel_path* (relative to hermes root) should be skipped."""
|
|
parts = rel_path.parts
|
|
if _in_excluded_root_dir(rel_path):
|
|
return True
|
|
# ``hermes-agent`` only matches at the root level; nested same-named dirs are preserved.
|
|
if any(p in _EXCLUDED_DIRS and (p != "hermes-agent" or p == parts[0]) for p in parts):
|
|
return True
|
|
name = rel_path.name
|
|
return name in _EXCLUDED_NAMES or name.startswith(_EXCLUDED_PREFIXES) or name.endswith(_EXCLUDED_SUFFIXES)
|
|
|
|
|
|
def _iter_backup_files(hermes_root: Path, out_path: Path, skipped_dirs: Optional[set] = None):
|
|
"""Yield ``(abs_path, rel_path)`` for every file a full backup should hold.
|
|
|
|
The one owner of the walk policy (directory pruning so os.walk never descends a multi-GB
|
|
excluded tree, the root-only ``hermes-agent`` carve-out, root runtime trees, per-file rules),
|
|
shared by ``hermes backup`` and the pre-update / pre-migration path so they can never drift.
|
|
"""
|
|
for dirpath, dirnames, filenames in os.walk(hermes_root, followlinks=False):
|
|
rel_dir = Path(dirpath).relative_to(hermes_root)
|
|
is_root = rel_dir == Path(".")
|
|
kept = [
|
|
d for d in dirnames
|
|
if (d not in _EXCLUDED_DIRS or (d == "hermes-agent" and not is_root))
|
|
and not _in_excluded_root_dir(rel_dir / d)]
|
|
if skipped_dirs is not None:
|
|
skipped_dirs.update(str(rel_dir / d) for d in set(dirnames) - set(kept))
|
|
dirnames[:] = kept
|
|
for fname in filenames:
|
|
rel = rel_dir / fname
|
|
fpath = hermes_root / rel
|
|
# zipfile.write() follows file symlinks, so skip links before any archive write can
|
|
# copy data from outside HERMES_HOME; never archive the output zip into itself.
|
|
if _should_exclude(rel) or fpath.is_symlink():
|
|
continue
|
|
with suppress(OSError, ValueError):
|
|
if fpath.resolve() == out_path.resolve():
|
|
continue
|
|
yield fpath, rel
|
|
|
|
|
|
# --- SQLite safe copy ---
|
|
|
|
def _close_quietly(conn: Optional[sqlite3.Connection]) -> None:
|
|
if conn is not None:
|
|
with suppress(Exception):
|
|
conn.close()
|
|
|
|
|
|
def _query_ro_sqlite(path: Path, fn):
|
|
"""Run ``fn(conn)`` on a read-only connection to *path*; return ``(value, None)`` or ``(None, exc)``."""
|
|
conn = None
|
|
try:
|
|
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True, timeout=1.0)
|
|
return fn(conn), None
|
|
except Exception as exc:
|
|
return None, exc
|
|
finally:
|
|
_close_quietly(conn)
|
|
|
|
|
|
def _safe_copy_db(src: Path, dst: Path, *, timeout_seconds: float = 10.0) -> bool:
|
|
"""Copy a SQLite database with the backup() API (WAL-safe consistent snapshot).
|
|
|
|
Fails closed when no consistent snapshot can be made: copying only the main file loses WAL data.
|
|
"""
|
|
conn = backup_conn = None
|
|
try:
|
|
# timeout=0.0 disables sqlite3's implicit busy wait so the progress callback owns the
|
|
# full locked-source deadline instead of adding the default timeout before each callback.
|
|
conn = sqlite3.connect(f"file:{src}?mode=ro", uri=True, timeout=0.0)
|
|
backup_conn = sqlite3.connect(str(dst))
|
|
busy_deadline = time.monotonic() + max(0.0, timeout_seconds)
|
|
|
|
def _check_backup_progress(status: int, _remaining: int, _total: int) -> None:
|
|
nonlocal busy_deadline
|
|
now = time.monotonic()
|
|
if status in (sqlite3.SQLITE_BUSY, sqlite3.SQLITE_LOCKED):
|
|
if now >= busy_deadline:
|
|
raise _SQLiteBackupTimeout(f"database remained locked for {timeout_seconds:g} seconds")
|
|
else:
|
|
busy_deadline = now + max(0.0, timeout_seconds)
|
|
|
|
conn.backup(backup_conn, pages=256, progress=_check_backup_progress, sleep=0.1)
|
|
return True
|
|
except Exception as exc:
|
|
logger.warning("SQLite safe copy failed for %s: %s", src, exc)
|
|
# Windows won't remove the partial destination while SQLite still has it open.
|
|
_close_quietly(backup_conn)
|
|
backup_conn = None
|
|
with suppress(OSError):
|
|
dst.unlink(missing_ok=True)
|
|
return False
|
|
finally:
|
|
_close_quietly(backup_conn)
|
|
_close_quietly(conn)
|
|
|
|
|
|
def is_zeroed_sqlite_file(path: Path, *, probe_bytes: int = 100, force: bool = False) -> bool:
|
|
"""True when *path* looks like the #68474 zeroed-state.db signature.
|
|
|
|
Only regular files qualify: probing a FIFO/device/socket could block indefinitely.
|
|
|
|
Signature: no ``SQLite format 3`` header and no data — either empty (size 0, the total-loss case,
|
|
#97568) or first *probe_bytes* all NUL. Used at SessionDB open and for snapshot diagnostics so a silent
|
|
all-zero file becomes a guided recovery instead of a generic failure.
|
|
"""
|
|
try:
|
|
if not path.is_file():
|
|
return False
|
|
except OSError:
|
|
return False
|
|
from hermes_cli.sqlite_safe_read import has_live_connection, read_header_bytes_preopen
|
|
if not force and has_live_connection(path):
|
|
return False
|
|
|
|
head = read_header_bytes_preopen(path, length=max(16, probe_bytes), force=force)
|
|
# Empty or all-NUL header => zeroed; a real header (or unreadable) => not.
|
|
return head is not None and not head.startswith(b"SQLite format 3") and not any(head)
|
|
|
|
|
|
# --- SQLite integrity verification ---
|
|
|
|
_SQLITE_HEADER = b"SQLite format 3\0"
|
|
|
|
# Above this size ``PRAGMA integrity_check`` (walks every b-tree page — minutes of pegged CPU on a
|
|
# 30 GB state.db, reading as a hung ``hermes update``) is replaced by the O(1) header+schema probe.
|
|
# Default ceiling above which ``PRAGMA integrity_check`` is skipped in favour of the (O(1)) header +
|
|
# structural probe. Sessions databases in the tens of GB are normal for heavy users, so the size-unbounded
|
|
# check is never an acceptable default on the update path. See #70553.
|
|
DEFAULT_INTEGRITY_CHECK_MAX_BYTES = 2 << 30 # 2 GiB
|
|
|
|
|
|
def verify_sqlite_integrity(
|
|
path: Path, *, check_header: bool = True, run_pragma: bool = True,
|
|
max_bytes: int = DEFAULT_INTEGRITY_CHECK_MAX_BYTES) -> dict:
|
|
"""Verify a SQLite database: existence + minimum size, header magic, then a read-only
|
|
``PRAGMA integrity_check`` (or a cheap structural probe above ``max_bytes``)."""
|
|
def _done(message: str, valid: bool = False, size: Optional[int] = None) -> dict:
|
|
return {"valid": valid, "message": message, "size": size}
|
|
try:
|
|
st = path.stat()
|
|
except FileNotFoundError:
|
|
return _done(f"not found: {path}")
|
|
except OSError as exc:
|
|
return _done(f"cannot stat: {exc}")
|
|
size = st.st_size
|
|
if size < 100: # SQLite minimum viable size (header + 1 page)
|
|
return _done(f"too small ({size} bytes) to be a valid SQLite database", size=size)
|
|
if check_header:
|
|
# Refused when a live connection exists (close() would cancel this process's POSIX locks
|
|
# — see sqlite_safe_read); verification targets offline snapshots/backup artifacts anyway.
|
|
from hermes_cli.sqlite_safe_read import read_header_bytes_preopen
|
|
head = read_header_bytes_preopen(path, length=len(_SQLITE_HEADER))
|
|
if head is None:
|
|
return _done("cannot read header", size=size)
|
|
if head != _SQLITE_HEADER:
|
|
return _done(f"missing SQLite header magic (got {head[:16].hex()!r})", size=size)
|
|
if max_bytes > 0 and size > max_bytes:
|
|
# O(1) probe: the header check caught the zeroed signature; reading sqlite_master + page
|
|
# geometry catches malformed-schema and truncated-header-page classes without a data walk.
|
|
_, exc = _query_ro_sqlite(path, lambda c: (
|
|
c.execute("PRAGMA schema_version").fetchone(),
|
|
c.execute("SELECT count(*) FROM sqlite_master").fetchone()))
|
|
if exc is not None:
|
|
kind = "failed" if isinstance(exc, sqlite3.DatabaseError) else "error"
|
|
return _done(f"schema probe {kind}: {exc}", size=size)
|
|
return _done(
|
|
f"size {size:,} bytes exceeds max_bytes {max_bytes:,}; "
|
|
"skipped PRAGMA integrity_check (header + schema probe passed)",
|
|
valid=True, size=size)
|
|
if run_pragma:
|
|
rows, exc = _query_ro_sqlite(
|
|
path, lambda c: [str(r[0]) for r in c.execute("PRAGMA integrity_check")])
|
|
if exc is not None:
|
|
kind = "cannot open database" if isinstance(exc, sqlite3.DatabaseError) else "integrity check error"
|
|
return _done(f"{kind}: {exc}", size=size)
|
|
if rows == ["ok"]:
|
|
return _done("integrity check passed", valid=True, size=size)
|
|
return _done(f"integrity check failed: {'; '.join(rows[:5])}", size=size)
|
|
return _done("header check passed", valid=True, size=size)
|
|
|
|
|
|
# --- Backup ---
|
|
|
|
_RUN_BACKUP_PREFIX = "hermes-backup-"
|
|
|
|
|
|
# --- Import ---
|
|
|
|
|
|
# --- Quick state snapshots (used by /snapshot slash command and hermes backup --quick) ---
|
|
|
|
# Critical state files (relative to HERMES_HOME) for quick snapshots; everything else is
|
|
# regeneratable or managed separately (skills, repo, sessions/). Entries may be files OR
|
|
# directories (recursive); missing entries are skipped. Pairing data lives in platform JSON blobs
|
|
# outside state.db, so it is listed explicitly — ``hermes update`` snapshots this set (#15733).
|
|
_QUICK_STATE_FILES = (
|
|
"state.db", "config.yaml", ".env", "auth.json", "cron/jobs.json", "cron/executions.db",
|
|
"gateway_state.json", "channel_directory.json", "channel_aliases.json", "processes.json",
|
|
"gateway/discord_message_recovery.db", # Discord reconnect replay ledger
|
|
# Per-profile user stores, destroyed if the update flow replaces the file and the post-update
|
|
# schema-init re-creates an empty one (#52889). Skipped when outside HERMES_HOME.
|
|
"projects.db", # per-profile project store
|
|
"response_store.db", # gateway conversation history / tool payloads
|
|
"memory_store.db", # holographic memory facts/entities
|
|
"verification_evidence.db", # agent verification audit trail
|
|
"kanban.db", # default board (back-compat <root>/kanban.db)
|
|
"kanban/boards", # non-default boards (workspaces/ + attachments/ skipped as regenerable)
|
|
# Pairing stores (generic + per-platform JSONs outside state.db)
|
|
"pairing", # legacy location (gateway/pairing.py)
|
|
"platforms/pairing", # new location (gateway/pairing.py)
|
|
"feishu_comment_pairing.json", # Feishu comment subscription pairings
|
|
)
|
|
|
|
_QUICK_DEFAULT_KEEP = 20
|
|
|
|
|
|
def _quick_snapshot_root(hermes_home: Optional[Path] = None) -> Path:
|
|
home = hermes_home or get_hermes_home()
|
|
return home / _QUICK_SNAPSHOTS_DIR
|
|
|
|
|
|
def _newest_first(root: Path, keep_entry) -> List[Path]:
|
|
"""Entries of *root* passing ``keep_entry``, newest (by name) first; ``[]`` if *root* is missing."""
|
|
if not root.exists():
|
|
return []
|
|
return sorted(filter(keep_entry, root.iterdir()), key=lambda p: p.name, reverse=True)
|
|
|
|
|
|
# Kept in sync with ``_QUICK_STATE_FILES`` and ``cron/jobs.py``'s ``JOBS_FILE``.
|
|
_CRON_JOBS_REL = "cron/jobs.json"
|
|
|
|
|
|
# Config paths the update flow must never change (#64160): model routing and the MoA section are
|
|
# consumed machine-wide, so an update/repair cycle that rewrites them silently redirects paid
|
|
# inference. Dotted paths into raw config.yaml; a single-element tuple protects a whole section.
|
|
_PROTECTED_CONFIG_PATHS: Tuple[Tuple[str, ...], ...] = (
|
|
("model", "provider"), ("model", "default"), ("model", "base_url"), ("model", "api_key"),
|
|
("moa",))
|
|
|
|
|
|
def _prune_oldest(newest_first: List[Path], keep: int, remove, what: str) -> int:
|
|
"""``remove(path)`` every entry past the first *keep*; return how many succeeded."""
|
|
deleted = 0
|
|
for p in newest_first[keep:]:
|
|
try:
|
|
remove(p)
|
|
deleted += 1
|
|
except OSError as exc:
|
|
logger.warning("Failed to prune %s %s: %s", what, p.name, exc)
|
|
return deleted
|
|
|
|
|
|
# --- Shared full-zip backup helper ---
|
|
|
|
|
|
# --- Pre-update / pre-migration auto-backups ---
|
|
|
|
_PRE_UPDATE_BACKUPS_DIR = "backups"
|
|
_PRE_UPDATE_PREFIX = "pre-update-"
|
|
_PRE_UPDATE_DEFAULT_KEEP = 5
|
|
_PRE_MIGRATION_PREFIX = "pre-migration-"
|
|
_PRE_MIGRATION_DEFAULT_KEEP = 5
|
|
|
|
|
|
def _prune_prefixed_zips(backup_dir: Path, prefix: str, keep: int, what: str) -> int:
|
|
"""Remove oldest ``<prefix>*.zip`` in *backup_dir* beyond *keep*; return count deleted.
|
|
|
|
Only prefix-matched files are touched, so hand-made zips or other backup kinds survive.
|
|
"""
|
|
backups = _newest_first(backup_dir, lambda p: p.is_file() and p.name.startswith(prefix)
|
|
and p.suffix.lower() == ".zip")
|
|
return _prune_oldest(backups, keep, Path.unlink, what)
|
|
|
|
|
|
def _create_prefixed_full_backup(
|
|
hermes_home: Optional[Path], prefix: str, keep: int, what: str, prune_what: str) -> Optional[Path]:
|
|
"""Write ``<HERMES_HOME>/backups/<prefix><timestamp>.zip`` and prune older same-prefix zips.
|
|
Returns the path, or ``None`` if nothing to back up or the write failed. Never raises."""
|
|
hermes_root = hermes_home or get_default_hermes_root()
|
|
if not hermes_root.is_dir():
|
|
return None
|
|
backup_dir = hermes_root / _PRE_UPDATE_BACKUPS_DIR
|
|
try:
|
|
backup_dir.mkdir(parents=True, exist_ok=True)
|
|
except OSError as exc:
|
|
logger.warning("Could not create %s backup dir %s: %s", what, backup_dir, exc)
|
|
return None
|
|
out_path = backup_dir / f"{prefix}{datetime.now().strftime('%Y-%m-%d-%H%M%S')}.zip"
|
|
if _write_full_zip_backup(out_path, hermes_root) is None:
|
|
return None
|
|
_prune_prefixed_zips(backup_dir, prefix, keep, prune_what)
|
|
return out_path
|
|
|
|
|
|
def create_pre_update_backup(
|
|
hermes_home: Optional[Path] = None, keep: int = _PRE_UPDATE_DEFAULT_KEEP) -> Optional[Path]:
|
|
"""Full zip backup to ``backups/pre-update-<timestamp>.zip``, auto-pruned; ``None`` if nothing
|
|
was found or the backup failed. Never raises — ``hermes update`` continues anyway."""
|
|
return _create_prefixed_full_backup(hermes_home, _PRE_UPDATE_PREFIX, max(keep, 1), "pre-update", "backup")
|
|
|
|
|
|
def create_pre_migration_backup(
|
|
hermes_home: Optional[Path] = None, keep: int = _PRE_MIGRATION_DEFAULT_KEEP) -> Optional[Path]:
|
|
"""Full zip backup to ``backups/pre-migration-<timestamp>.zip`` before ``hermes claw migrate``
|
|
(same dir as update backups so listings/``hermes import`` find it); ``None`` if nothing was
|
|
found or the write failed. Never raises."""
|
|
return _create_prefixed_full_backup(
|
|
hermes_home, _PRE_MIGRATION_PREFIX, max(keep, 0), "pre-migration", "pre-migration backup")
|
|
|
|
|
|
# ---- BEGIN PLUGIN-COMPAT (revert-scheduled; see COMPAT_MANIFEST.md) ----
|
|
# Names external plugins imported from this module before the Sep 2026 decomposition.
|
|
# Internal code MUST NOT use these (scripts/check_compat_pointers.py fails CI if it does).
|
|
# The whole block is removed by reverting the commit that added it.
|
|
|
|
def copy_db_and_verify(src: Path, dst: Path) -> bool:
|
|
"""Like :func:`_safe_copy_db` but verifies the destination after copy.
|
|
|
|
Returns True only when the copy succeeded AND the destination is valid
|
|
SQLite (header + integrity check). Verification honours the default
|
|
size ceiling — a multi-GB destination gets the header + schema probe
|
|
rather than a full ``PRAGMA integrity_check`` that would page through
|
|
the whole file.
|
|
"""
|
|
if not _safe_copy_db(src, dst):
|
|
return False
|
|
integrity = verify_sqlite_integrity(dst, run_pragma=True)
|
|
if not integrity.get("valid"):
|
|
try:
|
|
dst.unlink(missing_ok=True)
|
|
except OSError:
|
|
pass
|
|
logger.warning("Backup of %s failed integrity verification: %s", src, integrity.get("message"))
|
|
return False
|
|
return True
|
|
|
|
|
|
# ---- END PLUGIN-COMPAT ----
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Backup
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def run_backup(args) -> None:
|
|
"""Create a zip backup of the Hermes home directory."""
|
|
hermes_root = get_default_hermes_root()
|
|
|
|
if not hermes_root.is_dir():
|
|
print(f"Error: Hermes home directory not found at {hermes_root}")
|
|
sys.exit(1)
|
|
|
|
try:
|
|
with _backup_operation_lock(hermes_root):
|
|
_run_backup_locked(args, hermes_root)
|
|
except BackupInProgressError as exc:
|
|
print(f"Error: {exc}")
|
|
raise SystemExit(2) from exc
|
|
|
|
|
|
def _run_backup_locked(args, hermes_root: Path) -> None:
|
|
"""Write a full backup while the cross-process backup slot is held."""
|
|
|
|
# Determine output path
|
|
out_path = None
|
|
try:
|
|
if args.output:
|
|
out_path = Path(args.output).expanduser().resolve()
|
|
# If user gave a directory, put the zip inside it
|
|
if out_path.is_dir():
|
|
stamp = datetime.now().strftime("%Y-%m-%d-%H%M%S")
|
|
out_path = out_path / f"{_RUN_BACKUP_PREFIX}{stamp}.zip"
|
|
else:
|
|
stamp = datetime.now().strftime("%Y-%m-%d-%H%M%S")
|
|
out_path = Path.home() / f"{_RUN_BACKUP_PREFIX}{stamp}.zip"
|
|
|
|
# Ensure the suffix is .zip
|
|
if out_path.suffix.lower() != ".zip":
|
|
out_path = out_path.with_suffix(out_path.suffix + ".zip")
|
|
|
|
# Ensure parent directory exists
|
|
out_path.parent.mkdir(parents=True, exist_ok=True)
|
|
except OSError as exc:
|
|
# A bad/unwritable output path (permission denied, unreadable parent,
|
|
# etc.) should give a clean one-line error, not a raw traceback
|
|
# (round-3 QA SUB-01). is_dir() and mkdir() both hit the filesystem.
|
|
print(f"Error: cannot write backup to {args.output or out_path}: {exc}")
|
|
raise SystemExit(1) from exc
|
|
|
|
# Collect files
|
|
scan_started = time.monotonic()
|
|
logger.info("backup phase=scan status=started")
|
|
print(f"Scanning {display_hermes_home()} ...")
|
|
skipped_dirs: set = set()
|
|
files_to_add: list[tuple[Path, Path]] = list(
|
|
_iter_backup_files(hermes_root, out_path, skipped_dirs)
|
|
)
|
|
|
|
# External memory-provider state (e.g. ~/.honcho, ~/.hindsight) lives
|
|
# outside HERMES_HOME, so the walk above never sees it. Ask the active
|
|
# provider for its declared paths and stage them under the reserved
|
|
# ``_external/`` arc prefix, encoded relative to the user's home dir.
|
|
# Only paths under home are captured (security + portability); anything
|
|
# else is skipped with a note.
|
|
home_dir = Path.home().resolve()
|
|
external_to_add: list[tuple[Path, str]] = [] # (absolute, arcname)
|
|
skipped_external: list[str] = []
|
|
for base in _collect_memory_provider_external_paths():
|
|
try:
|
|
base_resolved = base.resolve()
|
|
base_resolved.relative_to(home_dir)
|
|
except (ValueError, OSError):
|
|
skipped_external.append(str(base))
|
|
continue
|
|
for fpath in _iter_external_files(base):
|
|
try:
|
|
rel_to_home = fpath.resolve().relative_to(home_dir)
|
|
except (ValueError, OSError):
|
|
continue
|
|
arcname = _EXTERNAL_PREFIX + rel_to_home.as_posix()
|
|
external_to_add.append((fpath, arcname))
|
|
|
|
if not files_to_add and not external_to_add:
|
|
logger.info(
|
|
"backup phase=scan status=empty duration_ms=%.1f",
|
|
(time.monotonic() - scan_started) * 1000,
|
|
)
|
|
print("No files to back up.")
|
|
return
|
|
|
|
# Create the zip
|
|
file_count = len(files_to_add) + len(external_to_add)
|
|
logger.info(
|
|
"backup phase=scan status=complete duration_ms=%.1f files=%d",
|
|
(time.monotonic() - scan_started) * 1000,
|
|
file_count,
|
|
)
|
|
logger.info("backup phase=archive status=started files=%d", file_count)
|
|
print(f"Backing up {file_count} files ...")
|
|
|
|
total_bytes = 0
|
|
errors = []
|
|
t0 = time.monotonic()
|
|
|
|
with _atomic_output_path(out_path) as archive_path, zipfile.ZipFile(
|
|
archive_path, "w", zipfile.ZIP_DEFLATED, compresslevel=6
|
|
) as zf:
|
|
for i, (abs_path, rel_path) in enumerate(files_to_add, 1):
|
|
try:
|
|
# Safe copy for SQLite databases (handles WAL mode)
|
|
if abs_path.suffix == ".db":
|
|
# Stage the snapshot alongside the output zip so that the
|
|
# temp file lives on the same filesystem. The system
|
|
# default (/tmp) may be a small tmpfs that cannot hold
|
|
# large databases, causing silent backup incompleteness.
|
|
with tempfile.NamedTemporaryFile(
|
|
suffix=".db", delete=False, dir=str(out_path.parent)
|
|
) as tmp:
|
|
tmp_db = Path(tmp.name)
|
|
if _safe_copy_db(abs_path, tmp_db):
|
|
zf.write(tmp_db, arcname=str(rel_path))
|
|
total_bytes += tmp_db.stat().st_size
|
|
tmp_db.unlink(missing_ok=True)
|
|
else:
|
|
tmp_db.unlink(missing_ok=True)
|
|
errors.append(f" {rel_path}: SQLite safe copy failed")
|
|
continue
|
|
else:
|
|
zf.write(abs_path, arcname=str(rel_path))
|
|
total_bytes += abs_path.stat().st_size
|
|
except (PermissionError, OSError, ValueError) as exc:
|
|
errors.append(f" {rel_path}: {exc}")
|
|
continue
|
|
|
|
# Progress every 500 files
|
|
if i % 500 == 0:
|
|
print(f" {i}/{file_count} files ...")
|
|
logger.info(
|
|
"backup phase=archive status=progress completed=%d total=%d",
|
|
i,
|
|
file_count,
|
|
)
|
|
|
|
# External memory-provider state, stored under the ``_external/`` arc
|
|
# prefix. These never include ``.db`` files in practice (config/env
|
|
# blobs), so a straight zf.write is fine.
|
|
for abs_path, arcname in external_to_add:
|
|
try:
|
|
zf.write(abs_path, arcname=arcname)
|
|
total_bytes += abs_path.stat().st_size
|
|
except (PermissionError, OSError, ValueError) as exc:
|
|
errors.append(f" {arcname}: {exc}")
|
|
continue
|
|
|
|
elapsed = time.monotonic() - t0
|
|
zip_size = out_path.stat().st_size
|
|
logger.info(
|
|
"backup phase=archive status=complete duration_ms=%.1f files=%d errors=%d bytes=%d",
|
|
elapsed * 1000,
|
|
file_count,
|
|
len(errors),
|
|
zip_size,
|
|
)
|
|
|
|
# Summary
|
|
print()
|
|
if errors:
|
|
print(f"Backup incomplete: {out_path}")
|
|
else:
|
|
print(f"Backup complete: {out_path}")
|
|
print(f" Files: {file_count}")
|
|
print(f" Original: {_format_size(total_bytes)}")
|
|
print(f" Compressed: {_format_size(zip_size)}")
|
|
print(f" Time: {elapsed:.1f}s")
|
|
|
|
if external_to_add:
|
|
print(
|
|
f"\n Included {len(external_to_add)} memory-provider file(s) "
|
|
f"stored outside {display_hermes_home()}."
|
|
)
|
|
|
|
if skipped_external:
|
|
print(
|
|
f"\n Skipped {len(skipped_external)} memory-provider path(s) "
|
|
f"outside your home directory (not portable):"
|
|
)
|
|
for p in sorted(skipped_external)[:10]:
|
|
print(f" {p}")
|
|
|
|
if skipped_dirs:
|
|
print("\n Excluded directories:")
|
|
for d in sorted(skipped_dirs):
|
|
print(f" {d}/")
|
|
|
|
if errors:
|
|
print(f"\n Warnings ({len(errors)} files skipped):")
|
|
for e in errors[:10]:
|
|
print(e)
|
|
if len(errors) > 10:
|
|
print(f" ... and {len(errors) - 10} more")
|
|
|
|
if not errors:
|
|
print(f"\nRestore with: hermes import {out_path.name}")
|
|
keep = getattr(args, "keep", 0)
|
|
if keep and out_path.name.startswith(_RUN_BACKUP_PREFIX):
|
|
pruned = _prune_prefixed_zips(out_path.parent, _RUN_BACKUP_PREFIX, keep, "backup")
|
|
if pruned:
|
|
print(f" Pruned {pruned} older {_RUN_BACKUP_PREFIX}*.zip (keeping {keep}).")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Import
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def run_import(args) -> None:
|
|
"""Restore a Hermes backup from a zip file."""
|
|
zip_path = Path(args.zipfile).expanduser().resolve()
|
|
|
|
if not zip_path.is_file():
|
|
print(f"Error: File not found: {zip_path}")
|
|
sys.exit(1)
|
|
|
|
if not zipfile.is_zipfile(zip_path):
|
|
print(f"Error: Not a valid zip file: {zip_path}")
|
|
sys.exit(1)
|
|
|
|
# The restore target must be the home the command operates under — the
|
|
# same path printed as "Target:" via display_hermes_home(). Resolving
|
|
# through get_default_hermes_root() instead maps a profile home
|
|
# (<root>/profiles/<name>) back to <root>, silently retargeting the
|
|
# restore at the live root while the profile directory stays empty.
|
|
hermes_root = get_hermes_home()
|
|
|
|
with zipfile.ZipFile(zip_path, "r") as zf:
|
|
# Validate
|
|
ok, reason = _validate_backup_zip(zf)
|
|
if not ok:
|
|
print(f"Error: {reason}")
|
|
sys.exit(1)
|
|
|
|
prefix = _detect_prefix(zf)
|
|
members = [n for n in zf.namelist() if not n.endswith("/")]
|
|
file_count = len(members)
|
|
|
|
print(f"Backup contains {file_count} files")
|
|
print(f"Target: {display_hermes_home()}")
|
|
|
|
if prefix:
|
|
print(f"Detected archive prefix: {prefix!r} (will be stripped)")
|
|
|
|
# Check for existing installation
|
|
has_config = (hermes_root / "config.yaml").exists()
|
|
has_env = (hermes_root / ".env").exists()
|
|
|
|
if (has_config or has_env) and not args.force:
|
|
print()
|
|
print("Warning: Target directory already has Hermes configuration.")
|
|
print("Importing will overwrite existing files with backup contents.")
|
|
print()
|
|
try:
|
|
answer = input("Continue? [y/N] ").strip().lower()
|
|
except (EOFError, KeyboardInterrupt):
|
|
print("\nAborted.")
|
|
sys.exit(1)
|
|
if answer not in {"y", "yes"}:
|
|
print("Aborted.")
|
|
return
|
|
|
|
# Extract
|
|
print(f"\nImporting {file_count} files ...")
|
|
hermes_root.mkdir(parents=True, exist_ok=True)
|
|
|
|
errors = []
|
|
restored = 0
|
|
restored_external = 0
|
|
skipped_runtime: list[str] = []
|
|
# (rel, live_counts, imported_counts) for every session database the
|
|
# import replaced with one holding fewer rows. A restore is allowed to
|
|
# do that — it just must not do it silently (issue #100960).
|
|
db_shrunk: list[tuple[str, tuple[int, int], tuple[int, int]]] = []
|
|
home_dir = Path.home().resolve()
|
|
# Resolved once: every member is published via a temp file, and mkstemp
|
|
# would otherwise create newly restored files as 0600.
|
|
new_file_mode = _default_new_file_mode()
|
|
t0 = time.monotonic()
|
|
|
|
for member in members:
|
|
# External memory-provider state captured under the reserved
|
|
# ``_external/`` arc prefix restores to its original home-relative
|
|
# location (e.g. ~/.honcho/config.json), NOT under HERMES_HOME.
|
|
if member.startswith(_EXTERNAL_PREFIX):
|
|
ext_rel = member[len(_EXTERNAL_PREFIX):]
|
|
if not ext_rel:
|
|
continue
|
|
target = home_dir / ext_rel
|
|
# Security: the resolved target must stay under the home dir.
|
|
try:
|
|
target.resolve().relative_to(home_dir)
|
|
except ValueError:
|
|
errors.append(f" {member}: path traversal blocked")
|
|
continue
|
|
try:
|
|
target.parent.mkdir(parents=True, exist_ok=True)
|
|
_extract_member_atomically(zf, member, target, new_file_mode)
|
|
# External provider configs commonly hold credentials.
|
|
if target.suffix in {".json", ".env", ".conf"} or target.name in _SECRET_FILE_NAMES:
|
|
try:
|
|
os.chmod(target, 0o600)
|
|
except OSError:
|
|
pass
|
|
restored += 1
|
|
restored_external += 1
|
|
except (PermissionError, OSError) as exc:
|
|
errors.append(f" {member}: {exc}")
|
|
if restored % 500 == 0:
|
|
print(f" {restored}/{file_count} files ...")
|
|
continue
|
|
|
|
# Strip prefix if detected
|
|
if prefix and member.startswith(prefix):
|
|
rel = member[len(prefix):]
|
|
else:
|
|
rel = member
|
|
|
|
if not rel:
|
|
continue
|
|
|
|
try:
|
|
parts = tuple(normalize_archive_parts(rel))
|
|
except ValueError:
|
|
errors.append(f" {rel}: path traversal blocked")
|
|
continue
|
|
|
|
# Never overwrite volatile gateway/process runtime state. These are
|
|
# namespaced to the machine/container the backup was taken on;
|
|
# clobbering them (especially gateway_state.json) breaks the gateway
|
|
# reconciler on the target and disconnects hosted instances from the
|
|
# Nous portal. Matched by basename so both the root profile and
|
|
# named profiles (profiles/<name>/gateway_state.json) are covered.
|
|
if parts[-1] in _IMPORT_SKIP_NAMES:
|
|
skipped_runtime.append(rel)
|
|
continue
|
|
|
|
# Older archives may contain PM selections pointing at another machine.
|
|
# Match their home-root paths; a plugin's own facts.json is user data.
|
|
if profile_root_entry(parts) in PM_RUNTIME_ROOT_DIRS:
|
|
skipped_runtime.append(rel)
|
|
continue
|
|
|
|
# A ``.db`` member is page-restored into the live file below; a
|
|
# WAL/SHM/journal member from the archive describes a different
|
|
# database image, and installing it beside the restored file (over
|
|
# a live sidecar, via os.replace) would replay a foreign WAL on
|
|
# the next open. Current backups never ship these
|
|
# (_EXCLUDED_SUFFIXES); older or hand-built archives might.
|
|
if rel.endswith(_SQLITE_SIDECAR_SUFFIXES):
|
|
skipped_runtime.append(rel)
|
|
continue
|
|
|
|
target = hermes_root.joinpath(*parts)
|
|
|
|
# Security: reject absolute paths and traversals
|
|
try:
|
|
target.resolve().relative_to(hermes_root.resolve())
|
|
except ValueError:
|
|
errors.append(f" {rel}: path traversal blocked")
|
|
continue
|
|
|
|
try:
|
|
target.parent.mkdir(parents=True, exist_ok=True)
|
|
if target.suffix == ".db":
|
|
# Count before the write: afterwards the rows this import
|
|
# drops are gone and there is nothing left to compare.
|
|
before = _count_session_rows(target)
|
|
_import_db_member(zf, member, target, new_file_mode)
|
|
after = _count_session_rows(target)
|
|
if before and after and after[1] < before[1]:
|
|
db_shrunk.append((rel, before, after))
|
|
else:
|
|
_extract_member_atomically(zf, member, target, new_file_mode)
|
|
if target.name in _SECRET_FILE_NAMES:
|
|
os.chmod(target, 0o600)
|
|
restored += 1
|
|
except (PermissionError, OSError) as exc:
|
|
errors.append(f" {rel}: {exc}")
|
|
|
|
if restored % 500 == 0:
|
|
print(f" {restored}/{file_count} files ...")
|
|
|
|
elapsed = time.monotonic() - t0
|
|
|
|
# Summary
|
|
print()
|
|
print(f"Import complete: {restored} files restored in {elapsed:.1f}s")
|
|
print(f" Target: {display_hermes_home()}")
|
|
|
|
if restored_external:
|
|
print(
|
|
f"\n Restored {restored_external} memory-provider file(s) to "
|
|
f"their original location(s) outside {display_hermes_home()}."
|
|
)
|
|
|
|
if errors:
|
|
print(f"\n Warnings ({len(errors)} files skipped):")
|
|
for e in errors[:10]:
|
|
print(e)
|
|
if len(errors) > 10:
|
|
print(f" ... and {len(errors) - 10} more")
|
|
|
|
if db_shrunk:
|
|
# The backup predates work that is now overwritten. Say so: the
|
|
# reported incident was twelve sessions disappearing with nothing
|
|
# logged anywhere (issue #100960).
|
|
print("\n ⚠ Session data replaced by older backup contents:")
|
|
for rel, before, after in db_shrunk:
|
|
print(
|
|
f" {rel}: {before[0]} session(s) / {before[1]} message(s)"
|
|
f" -> {after[0]} / {after[1]}"
|
|
)
|
|
print(
|
|
" Anything recorded after the backup was taken is not in it. "
|
|
"Recover from a newer backup or snapshot: hermes snapshot list"
|
|
)
|
|
|
|
if skipped_runtime:
|
|
print(
|
|
f"\n Preserved {len(skipped_runtime)} runtime state "
|
|
f"file(s) (kept this machine's, not the backup's):"
|
|
)
|
|
for rel in sorted(skipped_runtime)[:10]:
|
|
print(f" {rel}")
|
|
if len(skipped_runtime) > 10:
|
|
print(f" ... and {len(skipped_runtime) - 10} more")
|
|
|
|
# Post-import: restore profile wrapper scripts
|
|
profiles_dir = hermes_root / "profiles"
|
|
restored_profiles = []
|
|
if profiles_dir.is_dir():
|
|
try:
|
|
from hermes_cli.profiles import (
|
|
create_wrapper_script, check_alias_collision,
|
|
_is_wrapper_dir_in_path, _get_wrapper_dir,
|
|
)
|
|
for entry in sorted(profiles_dir.iterdir()):
|
|
if not entry.is_dir():
|
|
continue
|
|
profile_name = entry.name
|
|
# Only create wrappers for directories with config
|
|
if not (entry / "config.yaml").exists() and not (entry / ".env").exists():
|
|
continue
|
|
collision = check_alias_collision(profile_name)
|
|
if collision:
|
|
print(f" Skipped alias '{profile_name}': {collision}")
|
|
restored_profiles.append((profile_name, False))
|
|
else:
|
|
wrapper = create_wrapper_script(profile_name)
|
|
restored_profiles.append((profile_name, wrapper is not None))
|
|
|
|
if restored_profiles:
|
|
created = [n for n, ok in restored_profiles if ok]
|
|
skipped = [n for n, ok in restored_profiles if not ok]
|
|
if created:
|
|
print(f"\n Profile aliases restored: {', '.join(created)}")
|
|
if skipped:
|
|
print(f" Profile aliases skipped: {', '.join(skipped)}")
|
|
if not _is_wrapper_dir_in_path():
|
|
print(f"\n Note: {_get_wrapper_dir()} is not in your PATH.")
|
|
print(' Add to your shell config (~/.bashrc or ~/.zshrc):')
|
|
print(' export PATH="$HOME/.local/bin:$PATH"')
|
|
except ImportError:
|
|
# hermes_cli.profiles might not be available (fresh install)
|
|
if any(profiles_dir.iterdir()):
|
|
print("\n Profiles detected but aliases could not be created.")
|
|
print(" Run: hermes profile list (after installing hermes)")
|
|
|
|
# Guidance
|
|
print()
|
|
if not (hermes_root / "hermes-agent").is_dir():
|
|
print("Note: The hermes-agent codebase was not included in the backup.")
|
|
print(" If this is a fresh install, run: hermes update")
|
|
|
|
if restored_profiles:
|
|
gw_profiles = [n for n, _ in restored_profiles]
|
|
print("\nTo re-enable gateway services for profiles:")
|
|
for pname in gw_profiles:
|
|
print(f" hermes -p {pname} gateway install")
|
|
|
|
# Bring the restored install to life: the backup may contain bot
|
|
# tokens and registered cron jobs, but they're inert without a
|
|
# gateway process. Install/start the service automatically (a
|
|
# platform-less gateway is a supported mode, so this is safe even
|
|
# for backups with no messaging config). Best-effort and prompt-free;
|
|
# failures print a manual fallback and never fail the import.
|
|
native_default = _get_platform_default_hermes_home()
|
|
default_has_install = any(
|
|
(native_default / marker).exists()
|
|
for marker in ("config.yaml", ".env", "state.db")
|
|
)
|
|
# A restore into a sandbox or profile home must not silently install
|
|
# a second gateway pointed at it — on the default service name that
|
|
# would shadow or hijack the machine's primary install. Only revive
|
|
# the service automatically when the restore landed in the default
|
|
# home, or when no other install exists on this machine.
|
|
if hermes_root != native_default and default_has_install:
|
|
print(
|
|
"\nRestored into a non-default home; leaving the gateway service "
|
|
"alone to avoid clashing with the install at "
|
|
f"{native_default}."
|
|
)
|
|
print("To start a gateway for this home, run: hermes gateway install")
|
|
else:
|
|
try:
|
|
from hermes_cli.gateway import ensure_gateway_service, _is_service_running
|
|
|
|
if not _is_service_running():
|
|
print()
|
|
ensure_gateway_service(context="import")
|
|
except Exception:
|
|
print("\nStart the gateway to activate cron jobs and messaging:")
|
|
print(" hermes gateway install")
|
|
|
|
print("Done. Your Hermes configuration has been restored.")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Quick state snapshots (used by /snapshot slash command and hermes backup --quick)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
# Critical state files to include in quick snapshots (relative to HERMES_HOME).
|
|
# Everything else is either regeneratable (logs, cache) or managed separately
|
|
# (skills, repo, sessions/).
|
|
#
|
|
# Entries may be individual files OR directories. Directories are captured
|
|
# recursively; missing entries are silently skipped. Pairing data lives in
|
|
# platform-specific JSON blobs outside state.db, so it's listed here explicitly
|
|
# — `hermes update` snapshots this set before pulling so approved-user lists
|
|
# are recoverable if anything goes wrong (issue #15733).
|
|
_QUICK_STATE_FILES = (
|
|
"state.db",
|
|
"config.yaml",
|
|
".env",
|
|
"auth.json",
|
|
"cron/jobs.json",
|
|
"cron/executions.db",
|
|
"gateway_state.json",
|
|
"channel_directory.json",
|
|
"channel_aliases.json",
|
|
"processes.json",
|
|
"gateway/discord_message_recovery.db", # Discord reconnect replay ledger
|
|
# Per-profile user-created stores that live outside the git checkout and
|
|
# are therefore destroyed if the update flow removes/replaces the file and
|
|
# the post-update schema-init re-creates an empty one (issue #52889). All
|
|
# are at $HERMES_HOME/<name> for the default/root profile; on non-root
|
|
# profiles the real path is outside HERMES_HOME and the entry is silently
|
|
# skipped (best-effort, same as the pairing stores). SQLite DBs are copied
|
|
# WAL-safely via _safe_copy_db.
|
|
"projects.db", # per-profile project store
|
|
"response_store.db", # gateway conversation history / tool payloads
|
|
"memory_store.db", # holographic memory facts/entities
|
|
"verification_evidence.db", # agent verification audit trail
|
|
"kanban.db", # default board (back-compat <root>/kanban.db)
|
|
"kanban/boards", # non-default boards: each <slug>/kanban.db + board metadata (workspaces/ + attachments/ are skipped as regenerable)
|
|
# Pairing stores (generic + per-platform JSONs outside state.db)
|
|
"pairing", # legacy location (gateway/pairing.py)
|
|
"platforms/pairing", # new location (gateway/pairing.py)
|
|
"feishu_comment_pairing.json", # Feishu comment subscription pairings
|
|
)
|
|
|
|
# ``_QUICK_SNAPSHOTS_DIR`` lives with the exclusion rules at the top of the module.
|
|
_QUICK_DEFAULT_KEEP = 20
|
|
|
|
|
|
def create_quick_snapshot(
|
|
label: Optional[str] = None,
|
|
hermes_home: Optional[Path] = None,
|
|
keep: Optional[int] = None,
|
|
max_file_size: Optional[int] = None,
|
|
) -> Optional[str]:
|
|
"""Create one atomic quick snapshot while holding the shared backup slot."""
|
|
home = hermes_home or get_hermes_home()
|
|
with _backup_operation_lock(home):
|
|
return _create_quick_snapshot_locked(
|
|
label=label,
|
|
hermes_home=home,
|
|
keep=keep,
|
|
max_file_size=max_file_size,
|
|
)
|
|
|
|
|
|
def _create_quick_snapshot_locked(
|
|
label: Optional[str] = None,
|
|
hermes_home: Optional[Path] = None,
|
|
keep: Optional[int] = None,
|
|
max_file_size: Optional[int] = None,
|
|
) -> Optional[str]:
|
|
"""Create a quick state snapshot of critical files.
|
|
|
|
Copies STATE_FILES to a timestamped directory under state-snapshots/.
|
|
Auto-prunes old snapshots beyond the keep limit.
|
|
|
|
Args:
|
|
max_file_size: When set, individual files larger than this many bytes
|
|
are skipped (with a printed warning) instead of copied. Used by
|
|
the pre-update safety snapshot so a multi-GB ``state.db`` can
|
|
never stall ``hermes update`` or silently eat disk — the small
|
|
pairing/cron/config files the snapshot exists to protect are
|
|
always captured. ``None`` (default) copies everything, which
|
|
preserves manual ``/snapshot`` and ``hermes backup --quick``
|
|
behavior.
|
|
|
|
Returns:
|
|
Snapshot ID (timestamp-based), or None if no files found.
|
|
"""
|
|
home = hermes_home or get_hermes_home()
|
|
root = _quick_snapshot_root(home)
|
|
|
|
def _too_large(path: Path, rel_name: str) -> bool:
|
|
"""True (and warn) when ``path`` exceeds the max_file_size cap."""
|
|
if max_file_size is None:
|
|
return False
|
|
try:
|
|
size = path.stat().st_size
|
|
except OSError:
|
|
return False
|
|
if size <= max_file_size:
|
|
return False
|
|
print(
|
|
f" ⚠ Snapshot: skipping {rel_name} "
|
|
f"({_format_size(size)} exceeds {_format_size(max_file_size)} limit)"
|
|
)
|
|
logger.warning(
|
|
"Quick snapshot skipped %s: %d bytes exceeds %d byte limit",
|
|
rel_name,
|
|
size,
|
|
max_file_size,
|
|
)
|
|
return True
|
|
|
|
ts = datetime.now(timezone.utc).strftime("%Y%m%d-%H%M%S")
|
|
base_snap_id = f"{ts}-{label}" if label else ts
|
|
snap_id = base_snap_id
|
|
suffix = 2
|
|
while (root / snap_id).exists():
|
|
snap_id = f"{base_snap_id}-{suffix}"
|
|
suffix += 1
|
|
snap_dir = root / snap_id
|
|
staging_dir = root / f".{snap_id}.{os.getpid()}.partial"
|
|
shutil.rmtree(staging_dir, ignore_errors=True)
|
|
staging_dir.mkdir(parents=True, exist_ok=False)
|
|
logger.info("quick snapshot phase=copy status=started id=%s", snap_id)
|
|
|
|
manifest: Dict[str, int] = {} # rel_path -> file size
|
|
failed_dbs: list[str] = [] # present *.db that could not be snapshotted
|
|
# #68805: track protected DB files skipped for size — they are snapshot
|
|
# incompleteness just like a failed copy, so pruning must be suppressed
|
|
# to preserve the older complete snapshot that may contain the only
|
|
# recoverable database.
|
|
oversized_skipped: list[str] = []
|
|
|
|
for rel in _QUICK_STATE_FILES:
|
|
src = home / rel
|
|
if not src.exists():
|
|
continue
|
|
|
|
if src.is_dir():
|
|
# Walk the directory and record each file individually in the
|
|
# manifest so restore can treat them uniformly. Empty dirs are
|
|
# skipped (nothing to snapshot).
|
|
for sub in src.rglob("*"):
|
|
if not sub.is_file():
|
|
continue
|
|
sub_rel = sub.relative_to(home).as_posix()
|
|
# Skip heavy, regenerable per-board subtrees (scratch
|
|
# workspaces and task attachments can be large); we only need
|
|
# the board databases + their metadata to restore a board.
|
|
if "/workspaces/" in f"/{sub_rel}/" or "/attachments/" in f"/{sub_rel}/":
|
|
continue
|
|
if _too_large(sub, sub_rel):
|
|
if sub.suffix == ".db":
|
|
oversized_skipped.append(sub_rel)
|
|
continue
|
|
dst = staging_dir / sub_rel
|
|
dst.parent.mkdir(parents=True, exist_ok=True)
|
|
try:
|
|
# Route SQLite DBs through the WAL-safe backup() path so a
|
|
# board DB with an open WAL (the gateway may hold it at
|
|
# snapshot time) is captured consistently.
|
|
if sub.suffix == ".db":
|
|
if not _safe_copy_db(sub, dst):
|
|
failed_dbs.append(sub_rel)
|
|
print(
|
|
f" ⚠ Snapshot: SQLite safe copy FAILED for {sub_rel} "
|
|
f"— file may be locked or corrupted"
|
|
)
|
|
if is_zeroed_sqlite_file(sub):
|
|
print(
|
|
f" ⚠ Snapshot: {sub_rel} looks ZEROED "
|
|
f"(no SQLite header; {sub.stat().st_size} bytes of NULs?)"
|
|
)
|
|
continue
|
|
else:
|
|
shutil.copy2(sub, dst)
|
|
manifest[sub_rel] = dst.stat().st_size
|
|
except (OSError, PermissionError) as exc:
|
|
logger.warning("Could not snapshot %s: %s", sub_rel, exc)
|
|
continue
|
|
|
|
if not src.is_file():
|
|
continue
|
|
|
|
if _too_large(src, rel):
|
|
if src.suffix == ".db":
|
|
oversized_skipped.append(rel)
|
|
continue
|
|
|
|
dst = staging_dir / rel
|
|
dst.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
try:
|
|
if src.suffix == ".db":
|
|
if not _safe_copy_db(src, dst):
|
|
failed_dbs.append(rel)
|
|
print(
|
|
f" ⚠ Snapshot: SQLite safe copy FAILED for {rel} "
|
|
f"— file may be locked or corrupted"
|
|
)
|
|
if is_zeroed_sqlite_file(src):
|
|
print(
|
|
f" ⚠ Snapshot: {rel} looks ZEROED "
|
|
f"(no SQLite header; {src.stat().st_size} bytes)"
|
|
)
|
|
continue
|
|
else:
|
|
shutil.copy2(src, dst)
|
|
manifest[rel] = dst.stat().st_size
|
|
except (OSError, PermissionError) as exc:
|
|
logger.warning("Could not snapshot %s: %s", rel, exc)
|
|
|
|
if failed_dbs:
|
|
# Critical: update path used to log-and-continue with exit 0, so a
|
|
# missing state.db backup looked like a successful pre-update snapshot
|
|
# (#68474). Surface this on stdout where operators actually look.
|
|
print(
|
|
" ⚠ CRITICAL: could not snapshot DB file(s): "
|
|
+ ", ".join(failed_dbs)
|
|
)
|
|
print(
|
|
" ⚠ If sessions disappear after update, check "
|
|
f"{root} and run: hermes snapshot list"
|
|
)
|
|
logger.error(
|
|
"Quick snapshot failed to capture DB file(s): %s",
|
|
", ".join(failed_dbs),
|
|
)
|
|
|
|
if not manifest:
|
|
shutil.rmtree(staging_dir, ignore_errors=True)
|
|
if failed_dbs:
|
|
# Distinguish "nothing to snapshot" from "state.db present but unreadable"
|
|
print(
|
|
" ⚠ Snapshot aborted: no files captured "
|
|
f"(failed DBs: {', '.join(failed_dbs)})"
|
|
)
|
|
return None
|
|
|
|
# Write manifest
|
|
meta = {
|
|
"id": snap_id,
|
|
"timestamp": ts,
|
|
"label": label,
|
|
"file_count": len(manifest),
|
|
"total_size": sum(manifest.values()),
|
|
"files": manifest,
|
|
"failed_dbs": failed_dbs,
|
|
"oversized_skipped": oversized_skipped,
|
|
}
|
|
with open(staging_dir / "manifest.json", "w", encoding="utf-8") as f:
|
|
json.dump(meta, f, indent=2)
|
|
|
|
os.replace(staging_dir, snap_dir)
|
|
|
|
# Auto-prune. Defaults preserve historical manual /snapshot behavior; callers
|
|
# with known high-churn safety snapshots (for example pre-update) can pass a
|
|
# smaller keep value so large state.db copies do not accumulate indefinitely.
|
|
# #68805 review: skip pruning when a present DB failed to capture OR was
|
|
# skipped for size — either way the snapshot is incomplete and the older
|
|
# snapshot may contain the only recoverable database.
|
|
incomplete = failed_dbs or oversized_skipped
|
|
if not incomplete:
|
|
_prune_quick_snapshots(root, keep=_QUICK_DEFAULT_KEEP if keep is None else keep)
|
|
else:
|
|
if oversized_skipped:
|
|
print(
|
|
" ⚠ Skipping snapshot prune: DB file(s) skipped for size: "
|
|
+ ", ".join(oversized_skipped)
|
|
)
|
|
logger.warning(
|
|
"Quick snapshot skipped oversized DB file(s): %s",
|
|
", ".join(oversized_skipped),
|
|
)
|
|
logger.warning(
|
|
"Skipping snapshot prune because %d DB(s) failed to capture "
|
|
"and/or %d were oversized — preserving older snapshots as "
|
|
"recovery source",
|
|
len(failed_dbs), len(oversized_skipped),
|
|
)
|
|
|
|
logger.info(
|
|
"quick snapshot phase=copy status=complete id=%s files=%d bytes=%d",
|
|
snap_id,
|
|
len(manifest),
|
|
sum(manifest.values()),
|
|
)
|
|
return snap_id
|
|
|
|
|
|
def list_quick_snapshots(
|
|
limit: int = 20,
|
|
hermes_home: Optional[Path] = None,
|
|
) -> List[Dict[str, Any]]:
|
|
"""List existing quick state snapshots, most recent first."""
|
|
root = _quick_snapshot_root(hermes_home)
|
|
if not root.exists():
|
|
return []
|
|
|
|
results = []
|
|
for d in sorted(root.iterdir(), reverse=True):
|
|
if not d.is_dir() or d.name.startswith(".") or d.name.endswith(".partial"):
|
|
continue
|
|
manifest_path = d / "manifest.json"
|
|
if manifest_path.exists():
|
|
try:
|
|
with open(manifest_path, encoding="utf-8-sig") as f:
|
|
results.append(json.load(f))
|
|
except (json.JSONDecodeError, OSError):
|
|
results.append({"id": d.name, "file_count": 0, "total_size": 0})
|
|
if len(results) >= limit:
|
|
break
|
|
|
|
return results
|
|
|
|
|
|
def restore_quick_snapshot(
|
|
snapshot_id: str,
|
|
hermes_home: Optional[Path] = None,
|
|
) -> bool:
|
|
"""Restore state from a quick snapshot.
|
|
|
|
Overwrites current state files with the snapshot's copies.
|
|
Returns True if at least one file was restored.
|
|
"""
|
|
home = hermes_home or get_hermes_home()
|
|
root = _quick_snapshot_root(home)
|
|
|
|
# Security: reject snapshot_id values that contain path separators or
|
|
# traversal sequences so that `root / snapshot_id` stays inside root.
|
|
if not snapshot_id or "/" in snapshot_id or "\\" in snapshot_id or snapshot_id in (".", ".."):
|
|
logger.error("Invalid snapshot_id: %s", snapshot_id)
|
|
return False
|
|
|
|
snap_dir = root / snapshot_id
|
|
|
|
# Confirm the resolved path is still inside root (handles symlinks etc.)
|
|
try:
|
|
snap_dir.resolve().relative_to(root.resolve())
|
|
except ValueError:
|
|
logger.error("Snapshot path traversal blocked for id: %s", snapshot_id)
|
|
return False
|
|
|
|
if not snap_dir.is_dir():
|
|
return False
|
|
|
|
manifest_path = snap_dir / "manifest.json"
|
|
if not manifest_path.exists():
|
|
return False
|
|
|
|
with open(manifest_path, encoding="utf-8-sig") as f:
|
|
meta = json.load(f)
|
|
|
|
restored = 0
|
|
for rel in meta.get("files", {}):
|
|
# Security: reject absolute paths and traversals in manifest entries
|
|
src = snap_dir / rel
|
|
try:
|
|
src.resolve().relative_to(snap_dir.resolve())
|
|
except ValueError:
|
|
logger.error("Manifest path traversal blocked: %s", rel)
|
|
continue
|
|
|
|
dst = home / rel
|
|
try:
|
|
dst.resolve().relative_to(home.resolve())
|
|
except ValueError:
|
|
logger.error("Manifest path traversal blocked: %s", rel)
|
|
continue
|
|
|
|
if not src.exists():
|
|
continue
|
|
|
|
dst.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
try:
|
|
if dst.suffix == ".db":
|
|
# Restore through SQLite backup API so live connections
|
|
# (gateway, dashboard, another CLI session) see the
|
|
# restored data instead of continuing to serve stale
|
|
# cached pages from a replaced inode (issue #65942).
|
|
if not _safe_restore_db(src, dst):
|
|
# Refused (live holder) or failed: the destination was
|
|
# left as it was. Count it as a failure, not a restore.
|
|
logger.error("Failed to restore %s: live-safe restore refused", rel)
|
|
continue
|
|
else:
|
|
shutil.copy2(src, dst)
|
|
restored += 1
|
|
except (OSError, PermissionError) as exc:
|
|
logger.error("Failed to restore %s: %s", rel, exc)
|
|
|
|
logger.info("Restored %d files from snapshot %s", restored, snapshot_id)
|
|
return restored > 0
|
|
|
|
|
|
# Relative path of the cron job database inside HERMES_HOME. Kept in sync with
|
|
# the entry in ``_QUICK_STATE_FILES`` and with ``cron/jobs.py``'s ``JOBS_FILE``.
|
|
_CRON_JOBS_REL = "cron/jobs.json"
|
|
|
|
|
|
def _count_cron_jobs(path: Path) -> Optional[int]:
|
|
"""Return the number of cron jobs stored in ``path``.
|
|
|
|
The canonical on-disk shape is ``{"jobs": [...]}`` (see ``cron/jobs.py``).
|
|
A legacy bare-list shape (``[...]``) is also honoured.
|
|
|
|
Returns:
|
|
The job count for any *valid, readable* JSON document, or ``None`` if
|
|
the file is missing or cannot be parsed. ``None`` means "unknown" —
|
|
callers must not treat it as "zero jobs", because acting on an
|
|
unreadable file could mask a real corruption the user needs to see.
|
|
"""
|
|
if not path.is_file():
|
|
return None
|
|
try:
|
|
# utf-8-sig: same dialect as cron/jobs.load_jobs — Windows editors
|
|
# may leave a UTF-8 BOM that plain utf-8 json.load rejects. Without
|
|
# it a BOM'd jobs.json counts as "unreadable" (None) and the
|
|
# post-update cron-loss auto-restore safety net silently disables.
|
|
with open(path, "r", encoding="utf-8-sig") as f:
|
|
data = json.load(f)
|
|
except (OSError, json.JSONDecodeError):
|
|
return None
|
|
if isinstance(data, dict):
|
|
jobs = data.get("jobs", [])
|
|
return len(jobs) if isinstance(jobs, list) else None
|
|
if isinstance(data, list):
|
|
return len(data)
|
|
return None
|
|
|
|
|
|
def restore_cron_jobs_if_emptied(
|
|
snapshot_id: str,
|
|
hermes_home: Optional[Path] = None,
|
|
) -> Optional[Dict[str, Any]]:
|
|
"""Safety net for silent cron-job loss across ``hermes update``.
|
|
|
|
Config-version migrations have been observed to leave ``cron/jobs.json``
|
|
valid-but-empty after an update, silently dropping every scheduled job
|
|
(issue #34600). The desktop scheduler can also overwrite the file with its
|
|
own small set of internally-tracked crons, causing partial loss (issue
|
|
#52144).
|
|
|
|
This compares the *current* job count against the pre-update snapshot. If
|
|
the live file now has **fewer** jobs than the snapshot, the snapshot copy
|
|
of ``cron/jobs.json`` is restored in place.
|
|
|
|
The check is deliberately conservative — it only ever restores when there
|
|
is unambiguous evidence of loss (snapshot had more jobs than live file),
|
|
so a user who genuinely deleted jobs during/after the update is never
|
|
second-guessed, and an unreadable live file (count ``None``) is left
|
|
untouched so real corruption still surfaces.
|
|
|
|
Args:
|
|
snapshot_id: The pre-update quick-snapshot id (from
|
|
:func:`create_quick_snapshot`).
|
|
hermes_home: Override for the Hermes home directory (tests).
|
|
|
|
Returns:
|
|
``None`` when no action was taken (the common, healthy path). On a
|
|
successful restore, a dict ``{"restored": True, "job_count": N,
|
|
"snapshot_id": ...}`` so the caller can warn the user.
|
|
"""
|
|
if not snapshot_id:
|
|
return None
|
|
|
|
home = hermes_home or get_hermes_home()
|
|
live_path = home / _CRON_JOBS_REL
|
|
|
|
live_count = _count_cron_jobs(live_path)
|
|
# ``None`` (missing or unparseable) is intentionally left alone — that's a
|
|
# different failure mode the user should see rather than have papered over.
|
|
if live_count is None:
|
|
return None
|
|
|
|
snap_path = _quick_snapshot_root(home) / snapshot_id / _CRON_JOBS_REL
|
|
snap_count = _count_cron_jobs(snap_path)
|
|
if not snap_count: # None or 0 — nothing worth restoring
|
|
return None
|
|
|
|
# Restore when live has FEWER jobs than the pre-update snapshot.
|
|
# Catches both total loss (0 vs N) and partial loss (1 vs 19) — the
|
|
# desktop scheduler can overwrite jobs.json with its own small set of
|
|
# internally-tracked crons after an update/restart.
|
|
if live_count >= snap_count:
|
|
return None
|
|
|
|
try:
|
|
live_path.parent.mkdir(parents=True, exist_ok=True)
|
|
shutil.copy2(snap_path, live_path)
|
|
except (OSError, PermissionError) as exc:
|
|
logger.error(
|
|
"Cron jobs were emptied during update but auto-restore failed: %s", exc
|
|
)
|
|
return None
|
|
|
|
logger.warning(
|
|
"Restored %d cron job(s) from pre-update snapshot %s "
|
|
"(live file had %d job(s), snapshot had %d — jobs were lost during migration)",
|
|
snap_count,
|
|
snapshot_id,
|
|
live_count,
|
|
snap_count,
|
|
)
|
|
return {"restored": True, "job_count": snap_count, "snapshot_id": snapshot_id}
|
|
|
|
|
|
def _sibling_profile_homes(invoking_home: Path) -> list[tuple[str, Path]]:
|
|
"""(name, home) for every OTHER profile on this install. Never raises.
|
|
|
|
The update's code swap and gateway fleet restart touch every profile,
|
|
so the pre-update snapshot must too (#66140). The invoking profile is
|
|
excluded — its snapshot is taken by the existing call.
|
|
"""
|
|
homes: list[tuple[str, Path]] = []
|
|
try:
|
|
from hermes_cli.profiles import (
|
|
_get_default_hermes_home,
|
|
_get_profiles_root,
|
|
_PROFILE_ID_RE,
|
|
)
|
|
|
|
invoking = invoking_home.resolve()
|
|
default_home = _get_default_hermes_home()
|
|
if default_home.is_dir() and default_home.resolve() != invoking:
|
|
homes.append(("default", default_home))
|
|
root = _get_profiles_root()
|
|
if root.is_dir():
|
|
for entry in sorted(root.iterdir()):
|
|
if (
|
|
entry.is_dir()
|
|
and entry.name != "default"
|
|
and _PROFILE_ID_RE.match(entry.name)
|
|
and entry.resolve() != invoking
|
|
):
|
|
homes.append((entry.name, entry))
|
|
except Exception as exc:
|
|
logger.debug("Sibling profile enumeration failed: %s", exc)
|
|
return homes
|
|
|
|
|
|
def create_pre_update_snapshots_all_profiles(
|
|
invoking_home: Optional[Path] = None,
|
|
keep: Optional[int] = None,
|
|
max_file_size: Optional[int] = None,
|
|
) -> Dict[str, str]:
|
|
"""Pre-update quick snapshots for every SIBLING profile (#66140).
|
|
|
|
Same snapshot set, same per-file size cap, same keep policy as the
|
|
invoking profile's snapshot — identical semantics per profile, no
|
|
partial-tier coherence class. Each sibling's snapshot lands under its
|
|
OWN ``<home>/state-snapshots/`` so per-profile restore tooling finds
|
|
it where it expects. Returns ``{profile_name: snapshot_id}`` for the
|
|
siblings that snapshotted successfully. Never raises.
|
|
"""
|
|
results: Dict[str, str] = {}
|
|
home = invoking_home or get_hermes_home()
|
|
for name, profile_home in _sibling_profile_homes(home):
|
|
try:
|
|
snap_id = create_quick_snapshot(
|
|
label="pre-update",
|
|
hermes_home=profile_home,
|
|
keep=keep,
|
|
max_file_size=max_file_size,
|
|
)
|
|
if snap_id:
|
|
results[name] = snap_id
|
|
except Exception as exc:
|
|
logger.debug("Pre-update snapshot for profile %s failed: %s", name, exc)
|
|
return results
|
|
|
|
|
|
# Config paths that the update flow must never change (#64160): the model
|
|
# routing keys and the Mixture-of-Agents section are consumed machine-wide
|
|
# (gateway, cron, desktop), so an update/repair cycle that rewrites them
|
|
# silently redirects paid inference. Each entry is a dotted path into the raw
|
|
# config.yaml document; a single-element tuple protects the whole section.
|
|
_PROTECTED_CONFIG_PATHS: Tuple[Tuple[str, ...], ...] = (
|
|
("model", "provider"),
|
|
("model", "default"),
|
|
("model", "base_url"),
|
|
("model", "api_key"),
|
|
("moa",),
|
|
)
|
|
|
|
|
|
def _read_raw_yaml_dict(path: Path) -> Optional[Dict[str, Any]]:
|
|
"""Parse ``path`` as a YAML mapping. ``None`` = missing/unreadable/non-dict."""
|
|
if not path.is_file():
|
|
return None
|
|
try:
|
|
import hermes_yaml as yaml
|
|
|
|
with open(path, "r", encoding="utf-8-sig") as f:
|
|
data = yaml.safe_load(f)
|
|
except Exception:
|
|
return None
|
|
return data if isinstance(data, dict) else None
|
|
|
|
|
|
def _get_config_path_value(data: Dict[str, Any], dotted: Tuple[str, ...]) -> Any:
|
|
node: Any = data
|
|
for key in dotted:
|
|
if not isinstance(node, dict):
|
|
return None
|
|
node = node.get(key)
|
|
return node
|
|
|
|
|
|
def _set_config_path_value(data: Dict[str, Any], dotted: Tuple[str, ...], value: Any) -> None:
|
|
node = data
|
|
for key in dotted[:-1]:
|
|
child = node.get(key)
|
|
if not isinstance(child, dict):
|
|
child = {}
|
|
node[key] = child
|
|
node = child
|
|
node[dotted[-1]] = value
|
|
|
|
|
|
def restore_config_model_settings_if_rewritten(
|
|
snapshot_id: str,
|
|
hermes_home: Optional[Path] = None,
|
|
) -> Optional[Dict[str, Any]]:
|
|
"""Safety net for silent config.yaml model/MoA loss across ``hermes update``.
|
|
|
|
Desktop update/repair cycles have been observed to rewrite user-set
|
|
``model.provider``/``model.default`` and drop the ``moa:`` section
|
|
entirely (issue #64160; the macOS repair/relaunch variant rewrote a
|
|
pinned ``model.default`` to a transient composer pick). These keys are
|
|
consumed by the gateway and unattended cron jobs too, so a rewrite
|
|
silently changes paid inference behavior machine-wide.
|
|
|
|
Mirrors :func:`restore_cron_jobs_if_emptied`: compare the *current*
|
|
config against the pre-update snapshot taken minutes earlier by this
|
|
same update run, and restore only the protected keys — never the whole
|
|
file — when a value the user had set was changed or dropped. Everything
|
|
the update legitimately wrote (version stamps, new sections) is left in
|
|
place.
|
|
|
|
Args:
|
|
snapshot_id: The pre-update quick-snapshot id (from
|
|
:func:`create_quick_snapshot`).
|
|
hermes_home: Override for the Hermes home directory (tests/siblings).
|
|
|
|
Returns:
|
|
``None`` when no action was taken (the common, healthy path). On a
|
|
successful restore, ``{"restored": True, "keys": [...],
|
|
"snapshot_id": ...}`` so the caller can warn the user.
|
|
"""
|
|
if not snapshot_id:
|
|
return None
|
|
|
|
home = hermes_home or get_hermes_home()
|
|
live_path = home / "config.yaml"
|
|
snap_path = _quick_snapshot_root(home) / snapshot_id / "config.yaml"
|
|
|
|
snap = _read_raw_yaml_dict(snap_path)
|
|
if not snap:
|
|
return None # no snapshot copy — nothing to compare against
|
|
live = _read_raw_yaml_dict(live_path)
|
|
if live is None:
|
|
# Missing or unparseable live config is a different failure mode the
|
|
# user should see rather than have papered over (matches the cron net).
|
|
return None
|
|
|
|
restored_keys: list[str] = []
|
|
for dotted in _PROTECTED_CONFIG_PATHS:
|
|
snap_val = _get_config_path_value(snap, dotted)
|
|
if snap_val in (None, "", {}, []):
|
|
continue # user never set it — nothing to protect
|
|
live_val = _get_config_path_value(live, dotted)
|
|
if live_val == snap_val:
|
|
continue
|
|
_set_config_path_value(live, dotted, snap_val)
|
|
restored_keys.append(".".join(dotted))
|
|
|
|
if not restored_keys:
|
|
return None
|
|
|
|
try:
|
|
from utils import atomic_yaml_write
|
|
|
|
atomic_yaml_write(live_path, live)
|
|
except (OSError, PermissionError) as exc:
|
|
logger.error(
|
|
"config.yaml model settings were rewritten during update but "
|
|
"auto-restore failed: %s",
|
|
exc,
|
|
)
|
|
return None
|
|
|
|
logger.warning(
|
|
"Restored user config value(s) %s from pre-update snapshot %s — "
|
|
"the update flow rewrote them (#64160)",
|
|
", ".join(restored_keys),
|
|
snapshot_id,
|
|
)
|
|
return {"restored": True, "keys": restored_keys, "snapshot_id": snapshot_id}
|
|
|
|
|
|
def restore_config_model_settings_all_profiles(
|
|
profile_snapshots: Dict[str, str],
|
|
invoking_home: Optional[Path] = None,
|
|
) -> list[Dict[str, Any]]:
|
|
"""Run the config model-settings safety net for every sibling profile.
|
|
|
|
Same contract as :func:`restore_cron_jobs_all_profiles`: each profile's
|
|
live ``config.yaml`` is compared against ITS OWN same-generation
|
|
pre-update snapshot. Returns one result dict per restored profile, each
|
|
with a ``profile`` key added. Never raises.
|
|
"""
|
|
restored: list[Dict[str, Any]] = []
|
|
if not profile_snapshots:
|
|
return restored
|
|
home = invoking_home or get_hermes_home()
|
|
by_name = dict(_sibling_profile_homes(home))
|
|
for name, snap_id in profile_snapshots.items():
|
|
profile_home = by_name.get(name)
|
|
if profile_home is None:
|
|
continue
|
|
try:
|
|
result = restore_config_model_settings_if_rewritten(
|
|
snap_id, hermes_home=profile_home
|
|
)
|
|
except Exception as exc:
|
|
logger.debug(
|
|
"Config model-settings restore check for profile %s failed: %s",
|
|
name,
|
|
exc,
|
|
)
|
|
continue
|
|
if result:
|
|
result["profile"] = name
|
|
restored.append(result)
|
|
return restored
|
|
|
|
|
|
def restore_cron_jobs_all_profiles(
|
|
profile_snapshots: Dict[str, str],
|
|
invoking_home: Optional[Path] = None,
|
|
) -> list[Dict[str, Any]]:
|
|
"""Run the cron-jobs safety net for every sibling profile (#66140).
|
|
|
|
``profile_snapshots`` is the map returned by
|
|
:func:`create_pre_update_snapshots_all_profiles`. Each profile's live
|
|
``cron/jobs.json`` is compared against ITS OWN snapshot — restores are
|
|
same-generation by construction (the snapshot was taken minutes ago by
|
|
this update run). Returns one result dict per restored profile, each
|
|
with a ``profile`` key added. Never raises.
|
|
"""
|
|
restored: list[Dict[str, Any]] = []
|
|
if not profile_snapshots:
|
|
return restored
|
|
home = invoking_home or get_hermes_home()
|
|
by_name = dict(_sibling_profile_homes(home))
|
|
for name, snap_id in profile_snapshots.items():
|
|
profile_home = by_name.get(name)
|
|
if profile_home is None:
|
|
continue
|
|
try:
|
|
result = restore_cron_jobs_if_emptied(snap_id, hermes_home=profile_home)
|
|
except Exception as exc:
|
|
logger.debug("Cron restore check for profile %s failed: %s", name, exc)
|
|
continue
|
|
if result:
|
|
result["profile"] = name
|
|
restored.append(result)
|
|
return restored
|
|
|
|
|
|
def _prune_quick_snapshots(root: Path, keep: int = _QUICK_DEFAULT_KEEP) -> int:
|
|
"""Remove oldest quick snapshots beyond the keep limit. Returns count deleted."""
|
|
if not root.exists():
|
|
return 0
|
|
|
|
dirs = sorted(
|
|
(
|
|
d
|
|
for d in root.iterdir()
|
|
if d.is_dir() and not d.name.startswith(".") and not d.name.endswith(".partial")
|
|
),
|
|
key=lambda d: d.name,
|
|
reverse=True,
|
|
)
|
|
|
|
deleted = 0
|
|
for d in dirs[keep:]:
|
|
try:
|
|
shutil.rmtree(d)
|
|
deleted += 1
|
|
except OSError as exc:
|
|
logger.warning("Failed to prune snapshot %s: %s", d.name, exc)
|
|
|
|
return deleted
|
|
|
|
|
|
def prune_quick_snapshots(
|
|
keep: int = _QUICK_DEFAULT_KEEP,
|
|
hermes_home: Optional[Path] = None,
|
|
) -> int:
|
|
"""Manually prune quick snapshots. Returns count deleted."""
|
|
return _prune_quick_snapshots(_quick_snapshot_root(hermes_home), keep=keep)
|
|
|
|
|
|
def run_quick_backup(args) -> None:
|
|
"""CLI entry point for hermes backup --quick."""
|
|
label = getattr(args, "label", None)
|
|
snap_id = create_quick_snapshot(label=label)
|
|
if snap_id:
|
|
print(f"State snapshot created: {snap_id}")
|
|
snaps = list_quick_snapshots()
|
|
print(f" {len(snaps)} snapshot(s) stored in {display_hermes_home()}/state-snapshots/")
|
|
print(f" Restore with: /snapshot restore {snap_id}")
|
|
else:
|
|
print("No state files found to snapshot.")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Shared full-zip backup helper
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def _write_full_zip_backup(out_path: Path, hermes_root: Path) -> Optional[Path]:
|
|
"""Single-flight wrapper for automatic full zip backups."""
|
|
try:
|
|
with _backup_operation_lock(hermes_root):
|
|
return _write_full_zip_backup_locked(out_path, hermes_root)
|
|
except BackupInProgressError as exc:
|
|
logger.warning("Full-zip backup skipped: %s", exc)
|
|
return None
|
|
|
|
|
|
def _write_full_zip_backup_locked(out_path: Path, hermes_root: Path) -> Optional[Path]:
|
|
"""Write a full zip snapshot of ``hermes_root`` to ``out_path``.
|
|
|
|
Uses the same exclusion rules and SQLite safe-copy as :func:`run_backup`.
|
|
Returns the output path on success, None on failure (nothing to back up,
|
|
or write error — caller should surface the outcome but not raise).
|
|
"""
|
|
scan_started = time.monotonic()
|
|
logger.info("automatic backup phase=scan status=started")
|
|
try:
|
|
files_to_add = list(_iter_backup_files(hermes_root, out_path))
|
|
except OSError as exc:
|
|
logger.warning("Full-zip backup: walk failed: %s", exc)
|
|
return None
|
|
|
|
if not files_to_add:
|
|
return None
|
|
|
|
logger.info(
|
|
"automatic backup phase=scan status=complete duration_ms=%.1f files=%d",
|
|
(time.monotonic() - scan_started) * 1000,
|
|
len(files_to_add),
|
|
)
|
|
|
|
archive_started = time.monotonic()
|
|
try:
|
|
with _atomic_output_path(out_path) as archive_path, zipfile.ZipFile(
|
|
archive_path, "w", zipfile.ZIP_DEFLATED, compresslevel=6
|
|
) as zf:
|
|
for index, (abs_path, rel_path) in enumerate(files_to_add, 1):
|
|
try:
|
|
if abs_path.suffix == ".db":
|
|
# Stage the snapshot alongside the output zip so that the
|
|
# temp file lives on the same filesystem. The system
|
|
# default (/tmp) may be a small tmpfs that cannot hold
|
|
# large databases, causing silent backup incompleteness.
|
|
with tempfile.NamedTemporaryFile(
|
|
suffix=".db", delete=False, dir=str(out_path.parent)
|
|
) as tmp:
|
|
tmp_db = Path(tmp.name)
|
|
try:
|
|
if not _safe_copy_db(abs_path, tmp_db):
|
|
logger.warning(
|
|
"Full-zip backup aborted: SQLite snapshot failed for %s",
|
|
rel_path,
|
|
)
|
|
raise _SQLiteSnapshotError(str(rel_path))
|
|
zf.write(tmp_db, arcname=str(rel_path))
|
|
finally:
|
|
tmp_db.unlink(missing_ok=True)
|
|
else:
|
|
zf.write(abs_path, arcname=str(rel_path))
|
|
except (PermissionError, OSError, ValueError) as exc:
|
|
logger.debug("Skipping %s in zip backup: %s", rel_path, exc)
|
|
continue
|
|
if index % 500 == 0:
|
|
logger.info(
|
|
"automatic backup phase=archive status=progress completed=%d total=%d",
|
|
index,
|
|
len(files_to_add),
|
|
)
|
|
except (OSError, _SQLiteSnapshotError) as exc:
|
|
logger.warning("Full-zip backup: zip write failed: %s", exc)
|
|
# ``_atomic_output_path`` already removed the hidden partial. Do not
|
|
# unlink ``out_path`` here: it may be a previous valid backup that the
|
|
# atomic publisher deliberately preserved.
|
|
return None
|
|
|
|
logger.info(
|
|
"automatic backup phase=archive status=complete duration_ms=%.1f files=%d bytes=%d",
|
|
(time.monotonic() - archive_started) * 1000,
|
|
len(files_to_add),
|
|
out_path.stat().st_size,
|
|
)
|
|
|
|
return out_path
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|