The Sep 2026 decomposition (PR #102117) makes internal import paths a non-API: names now live in the focused modules that define them. This commit is the ONLY thing keeping the old paths alive, so external plugins have time to update. It is deliberately a single, unsquashed commit: git revert <this sha> removes every shim, stub and manifest at once on the announced date. Nothing in-tree may depend on these pointers: scripts/check_compat_pointers.py (wired into lint.yml) fails CI if it does. What it adds (see COMPAT_MANIFEST.md, compat_manifest.json): - 332 facade modules get one delimited `PLUGIN-COMPAT` block appended at the end of the file - 1,172 moved names resolved lazily via a module `__getattr__` (PEP 562) — never a top-level import, so no import cycles; facades that already had `__getattr__` get a chained one - 592 third-party/stdlib names the old modules used to expose, with their original import statements - 266 public definitions that had been deleted as unused, restored byte-for-byte from the pre-decomposition tree (+40 private helpers and 16 imports pulled in only because a restored definition needs them) - 3 deleted modules recreated as re-export stubs (gateway/startup_watchdog, hermes_cli/observability/ relay_runtime, tools/environments/modal_utils) - private names (`_x`) get no pointer: they were never API (3,792 skipped) Verified: all 335 touched modules import under a fresh HERMES_HOME and every manifest name resolves; the lint reports zero in-tree uses; ruff clean; targeted suites unchanged.
1654 lines
76 KiB
Python
1654 lines
76 KiB
Python
"""Backup and import commands for hermes CLI."""
|
|
|
|
import json
|
|
import logging
|
|
import os
|
|
import shutil
|
|
import sqlite3
|
|
import stat
|
|
import sys
|
|
import tempfile
|
|
import threading
|
|
import time
|
|
import zipfile
|
|
from contextlib import closing, contextmanager, suppress
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
from typing import Any, Dict, List, Optional, Tuple
|
|
|
|
from hermes_constants import (
|
|
_get_platform_default_hermes_home, get_default_hermes_root, get_hermes_home, display_hermes_home,
|
|
)
|
|
from utils import (
|
|
_preserve_file_mode, _preserve_file_owner, _restore_file_mode, _restore_file_owner, atomic_replace,
|
|
)
|
|
|
|
from hermes_cli.sizefmt import format_bytes as _format_size
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# --- Exclusion rules ---
|
|
|
|
# Where ``hermes backup --quick`` / ``/snapshot`` / the pre-update safety net write state
|
|
# snapshots (see ``create_quick_snapshot``); defined here because the exclusion set needs it.
|
|
_QUICK_SNAPSHOTS_DIR = "state-snapshots"
|
|
|
|
# Directory names to skip (matched against each path component). ``hermes-agent`` only matches at
|
|
# the root (``_should_exclude``) so skill dirs like ``skills/.../hermes-agent/`` survive. The
|
|
# dependency/cache entries matter: one plugin venv or pip/uv cache under HERMES_HOME walked
|
|
# file-by-file balloons a backup to hundreds of thousands of entries ("backup stuck for days").
|
|
# Mostly mirrors ``agent.skill_utils.EXCLUDED_SKILL_DIRS``; ``.cache`` is backup-only. ``.archive``
|
|
# is deliberately NOT excluded: the curator's ``skills/.archive/`` holds restorable user skills.
|
|
_EXCLUDED_DIRS = {
|
|
"hermes-agent", # the codebase repo — re-clone instead
|
|
"__pycache__", # bytecode caches — regenerated on import
|
|
".git", # nested git dirs (profiles shouldn't have these, but safety)
|
|
"node_modules", # js deps — reinstalled on demand
|
|
"backups", # prior auto-backups — don't nest backups exponentially
|
|
_QUICK_SNAPSHOTS_DIR, # each holds a full state.db copy — same reason as ``backups``
|
|
"checkpoints", # session-hash-keyed trajectory caches — regenerated, don't port
|
|
# Live CDP browser profiles: Chromium holds their SQLite DBs exclusively locked while running
|
|
# and sqlite3.backup() retries SQLITE_BUSY forever, hanging the backup. Regenerable anyway.
|
|
"browser-profiles",
|
|
# Real-profile browsing snapshot (browser.use_real_profile): copies of the user's Cookies /
|
|
# Login Data — a credential store that must NOT enter an archive. Regenerated on next launch.
|
|
"browser-profile",
|
|
# Python dependency trees (plugin / MCP-server venvs) — regenerated by reinstalling.
|
|
".venv", "venv", "site-packages",
|
|
# Tool / build caches — all regeneratable.
|
|
".cache", ".tox", ".nox", ".pytest_cache", ".mypy_cache", ".ruff_cache",
|
|
}
|
|
|
|
# Hermes-managed runtime downloads (GGUF models, llama.cpp runtimes, managed Node): re-downloaded
|
|
# on demand and routinely tens to hundreds of GB. Matched ONLY at the root of HERMES_HOME and at
|
|
# ``profiles/<name>/`` — a deeper dir of the same name (a skill's ``models/``) is user data.
|
|
_EXCLUDED_ROOT_DIRS = {"models", "runtimes", "node"}
|
|
|
|
|
|
def _in_excluded_root_dir(rel_path: Path) -> bool:
|
|
"""True when *rel_path* is, or sits inside, a managed runtime tree at a profile-home root."""
|
|
parts = rel_path.parts
|
|
return bool(parts) and (
|
|
parts[0] in _EXCLUDED_ROOT_DIRS
|
|
or (len(parts) >= 3 and parts[0] == "profiles" and parts[2] in _EXCLUDED_ROOT_DIRS))
|
|
|
|
|
|
# SQLite sidecars are excluded because ``*.db`` is snapshotted via ``sqlite3.backup()``:
|
|
# shipping the live WAL/SHM/journal alongside would pair a fresh snapshot with stale sidecar
|
|
# state and produce a torn restore on next open. They are regenerated on first connection.
|
|
_SQLITE_SIDECAR_SUFFIXES = (".db-wal", ".db-shm", ".db-journal")
|
|
_EXCLUDED_SUFFIXES = (".pyc", ".pyo", *_SQLITE_SIDECAR_SUFFIXES)
|
|
|
|
# File names to skip (runtime state that's meaningless on another machine)
|
|
_EXCLUDED_NAMES = {".backup.lock", "gateway.pid", "cron.pid"}
|
|
|
|
# The desktop updater's pre-flight drops ``state.db.pre-update-emergency-<ts>.bak`` at the root
|
|
# — a backup artifact like ``backups/``. Prefix-matched because the name carries a timestamp;
|
|
# a plain ``.bak`` suffix rule would drop user files.
|
|
_EXCLUDED_PREFIXES = ("state.db.pre-update-emergency-",)
|
|
|
|
# Files ``hermes import`` must never overwrite, matched by basename so root and named profiles are
|
|
# both covered. They hold runtime state namespaced to the SOURCE machine: ``gateway_state.json``
|
|
# drives the container-boot reconciler (a foreign value leaves the gateway stuck "starting" and
|
|
# disconnected from the Nous portal); PID/lock/registry files reference source PIDs. Mirrors
|
|
# ``container_boot._STALE_RUNTIME_FILES``; import filters too because older backups predate the
|
|
# backup-side exclusions.
|
|
_IMPORT_SKIP_NAMES = {"gateway_state.json", "gateway.pid", "cron.pid", "gateway.lock", "processes.json"}
|
|
|
|
# zipfile.open() drops Unix mode bits on extract; restore tightens these to 0600.
|
|
_SECRET_FILE_NAMES = {".env", "auth.json", "state.db"}
|
|
|
|
# Reserved archive subtree for memory-provider state OUTSIDE HERMES_HOME (e.g. ~/.honcho, via
|
|
# MemoryProvider.backup_paths()), stored and restored relative to the user's home; paths not
|
|
# under home are skipped.
|
|
_EXTERNAL_PREFIX = "_external/"
|
|
|
|
|
|
class BackupInProgressError(RuntimeError):
|
|
"""Raised when another process already owns the Hermes backup slot."""
|
|
|
|
|
|
class _SQLiteSnapshotError(RuntimeError):
|
|
pass
|
|
|
|
|
|
class _SQLiteBackupTimeout(RuntimeError):
|
|
"""Raised when a SQLite snapshot remains busy past its deadline."""
|
|
|
|
|
|
@contextmanager
|
|
def _backup_operation_lock(hermes_home: Path, timeout_seconds: float = 0.25):
|
|
"""Acquire one cross-process backup slot for full and quick snapshots."""
|
|
lock_path = hermes_home / ".backup.lock"
|
|
lock_path.parent.mkdir(parents=True, exist_ok=True)
|
|
handle = lock_path.open("a+b")
|
|
acquired = False
|
|
deadline = time.monotonic() + max(0.0, timeout_seconds)
|
|
try:
|
|
if os.name == "nt":
|
|
import msvcrt
|
|
if lock_path.stat().st_size == 0:
|
|
handle.write(b" ")
|
|
handle.flush()
|
|
def _lock_op(flag: int) -> None:
|
|
handle.seek(0)
|
|
msvcrt.locking(handle.fileno(), flag, 1)
|
|
lock_flag, unlock_flag = msvcrt.LK_NBLCK, msvcrt.LK_UNLCK
|
|
else:
|
|
import fcntl
|
|
def _lock_op(flag: int) -> None:
|
|
fcntl.flock(handle.fileno(), flag)
|
|
lock_flag, unlock_flag = fcntl.LOCK_EX | fcntl.LOCK_NB, fcntl.LOCK_UN
|
|
while not acquired:
|
|
try:
|
|
_lock_op(lock_flag)
|
|
acquired = True
|
|
except OSError:
|
|
if time.monotonic() >= deadline:
|
|
raise BackupInProgressError("another Hermes backup is already running")
|
|
time.sleep(0.05)
|
|
yield
|
|
finally:
|
|
if acquired:
|
|
with suppress(OSError):
|
|
_lock_op(unlock_flag)
|
|
handle.close()
|
|
|
|
|
|
@contextmanager
|
|
def _atomic_output_path(final_path: Path):
|
|
"""Yield a hidden sibling path and publish it only after a clean close."""
|
|
partial_path = final_path.with_name(f".{final_path.name}.{os.getpid()}-{threading.get_ident()}.partial")
|
|
partial_path.unlink(missing_ok=True)
|
|
try:
|
|
yield partial_path
|
|
os.replace(partial_path, final_path)
|
|
except BaseException:
|
|
partial_path.unlink(missing_ok=True)
|
|
raise
|
|
|
|
|
|
def _is_within(path: Path, root: Path) -> bool:
|
|
"""True when *path* resolves inside the already-resolved *root* (traversal / symlink guard)."""
|
|
return path.resolve().is_relative_to(root)
|
|
|
|
|
|
def _collect_memory_provider_external_paths() -> List[Path]:
|
|
"""Existing paths the active memory provider declares via ``backup_paths()``; ``[]`` on any
|
|
provider failure (backup must never fail because of a flaky plugin)."""
|
|
try:
|
|
from plugins.memory import _get_active_memory_provider, load_memory_provider
|
|
active = _get_active_memory_provider()
|
|
provider = load_memory_provider(active) if active else None
|
|
except Exception:
|
|
return []
|
|
if provider is None:
|
|
return []
|
|
try:
|
|
declared = provider.backup_paths() or []
|
|
except Exception as exc:
|
|
logger.warning("backup_paths() failed for memory provider %r: %s", active, exc)
|
|
return []
|
|
out: Dict[Path, Path] = {} # resolved -> first declared spelling
|
|
for raw in declared:
|
|
try:
|
|
p = Path(raw).expanduser()
|
|
except Exception:
|
|
continue
|
|
if not p.exists():
|
|
continue
|
|
try:
|
|
resolved = p.resolve()
|
|
except (OSError, ValueError):
|
|
continue
|
|
out.setdefault(resolved, p)
|
|
return list(out.values())
|
|
|
|
|
|
def _iter_external_files(base: Path) -> List[Path]:
|
|
"""Regular files under *base* (a file or a directory), skipping symlinks, caches, and pyc."""
|
|
if base.is_file() and not base.is_symlink():
|
|
return [base]
|
|
if not base.is_dir():
|
|
return []
|
|
files: List[Path] = []
|
|
for dirpath, dirnames, filenames in os.walk(base, followlinks=False):
|
|
dirnames[:] = [d for d in dirnames if d not in _EXCLUDED_DIRS]
|
|
files.extend(fp for fp in (Path(dirpath) / f for f in filenames)
|
|
if not (fp.is_symlink() or fp.name in _EXCLUDED_NAMES
|
|
or fp.name.endswith(_EXCLUDED_SUFFIXES)))
|
|
return files
|
|
|
|
|
|
def _should_exclude(rel_path: Path) -> bool:
|
|
"""Return True if *rel_path* (relative to hermes root) should be skipped."""
|
|
parts = rel_path.parts
|
|
if _in_excluded_root_dir(rel_path):
|
|
return True
|
|
# ``hermes-agent`` only matches at the root level; nested same-named dirs are preserved.
|
|
if any(p in _EXCLUDED_DIRS and (p != "hermes-agent" or p == parts[0]) for p in parts):
|
|
return True
|
|
name = rel_path.name
|
|
return name in _EXCLUDED_NAMES or name.startswith(_EXCLUDED_PREFIXES) or name.endswith(_EXCLUDED_SUFFIXES)
|
|
|
|
|
|
def _iter_backup_files(hermes_root: Path, out_path: Path, skipped_dirs: Optional[set] = None):
|
|
"""Yield ``(abs_path, rel_path)`` for every file a full backup should hold.
|
|
|
|
The one owner of the walk policy (directory pruning so os.walk never descends a multi-GB
|
|
excluded tree, the root-only ``hermes-agent`` carve-out, root runtime trees, per-file rules),
|
|
shared by ``hermes backup`` and the pre-update / pre-migration path so they can never drift.
|
|
"""
|
|
for dirpath, dirnames, filenames in os.walk(hermes_root, followlinks=False):
|
|
rel_dir = Path(dirpath).relative_to(hermes_root)
|
|
is_root = rel_dir == Path(".")
|
|
kept = [
|
|
d for d in dirnames
|
|
if (d not in _EXCLUDED_DIRS or (d == "hermes-agent" and not is_root))
|
|
and not _in_excluded_root_dir(rel_dir / d)]
|
|
if skipped_dirs is not None:
|
|
skipped_dirs.update(str(rel_dir / d) for d in set(dirnames) - set(kept))
|
|
dirnames[:] = kept
|
|
for fname in filenames:
|
|
rel = rel_dir / fname
|
|
fpath = hermes_root / rel
|
|
# zipfile.write() follows file symlinks, so skip links before any archive write can
|
|
# copy data from outside HERMES_HOME; never archive the output zip into itself.
|
|
if _should_exclude(rel) or fpath.is_symlink():
|
|
continue
|
|
with suppress(OSError, ValueError):
|
|
if fpath.resolve() == out_path.resolve():
|
|
continue
|
|
yield fpath, rel
|
|
|
|
|
|
# --- SQLite safe copy ---
|
|
|
|
def _close_quietly(conn: Optional[sqlite3.Connection]) -> None:
|
|
if conn is not None:
|
|
with suppress(Exception):
|
|
conn.close()
|
|
|
|
|
|
def _query_ro_sqlite(path: Path, fn):
|
|
"""Run ``fn(conn)`` on a read-only connection to *path*; return ``(value, None)`` or ``(None, exc)``."""
|
|
conn = None
|
|
try:
|
|
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True, timeout=1.0)
|
|
return fn(conn), None
|
|
except Exception as exc:
|
|
return None, exc
|
|
finally:
|
|
_close_quietly(conn)
|
|
|
|
|
|
def _safe_copy_db(src: Path, dst: Path, *, timeout_seconds: float = 10.0) -> bool:
|
|
"""Copy a SQLite database with the backup() API (WAL-safe consistent snapshot).
|
|
|
|
Fails closed when no consistent snapshot can be made: copying only the main file loses WAL data.
|
|
"""
|
|
conn = backup_conn = None
|
|
try:
|
|
# timeout=0.0 disables sqlite3's implicit busy wait so the progress callback owns the
|
|
# full locked-source deadline instead of adding the default timeout before each callback.
|
|
conn = sqlite3.connect(f"file:{src}?mode=ro", uri=True, timeout=0.0)
|
|
backup_conn = sqlite3.connect(str(dst))
|
|
busy_deadline = time.monotonic() + max(0.0, timeout_seconds)
|
|
|
|
def _check_backup_progress(status: int, _remaining: int, _total: int) -> None:
|
|
nonlocal busy_deadline
|
|
now = time.monotonic()
|
|
if status in (sqlite3.SQLITE_BUSY, sqlite3.SQLITE_LOCKED):
|
|
if now >= busy_deadline:
|
|
raise _SQLiteBackupTimeout(f"database remained locked for {timeout_seconds:g} seconds")
|
|
else:
|
|
busy_deadline = now + max(0.0, timeout_seconds)
|
|
|
|
conn.backup(backup_conn, pages=256, progress=_check_backup_progress, sleep=0.1)
|
|
return True
|
|
except Exception as exc:
|
|
logger.warning("SQLite safe copy failed for %s: %s", src, exc)
|
|
# Windows won't remove the partial destination while SQLite still has it open.
|
|
_close_quietly(backup_conn)
|
|
backup_conn = None
|
|
with suppress(OSError):
|
|
dst.unlink(missing_ok=True)
|
|
return False
|
|
finally:
|
|
_close_quietly(backup_conn)
|
|
_close_quietly(conn)
|
|
|
|
|
|
def is_zeroed_sqlite_file(path: Path, *, probe_bytes: int = 100, force: bool = False) -> bool:
|
|
"""True when *path* looks like the #68474 zeroed-state.db signature.
|
|
|
|
Only regular files qualify: probing a FIFO/device/socket could block indefinitely.
|
|
|
|
Signature: no ``SQLite format 3`` header and no data — either empty (size 0, the total-loss case,
|
|
#97568) or first *probe_bytes* all NUL. Used at SessionDB open and for snapshot diagnostics so a silent
|
|
all-zero file becomes a guided recovery instead of a generic failure.
|
|
"""
|
|
try:
|
|
if not path.is_file():
|
|
return False
|
|
except OSError:
|
|
return False
|
|
from hermes_cli.sqlite_safe_read import has_live_connection, read_header_bytes_preopen
|
|
if not force and has_live_connection(path):
|
|
return False
|
|
|
|
head = read_header_bytes_preopen(path, length=max(16, probe_bytes), force=force)
|
|
# Empty or all-NUL header => zeroed; a real header (or unreadable) => not.
|
|
return head is not None and not head.startswith(b"SQLite format 3") and not any(head)
|
|
|
|
|
|
# --- SQLite integrity verification ---
|
|
|
|
_SQLITE_HEADER = b"SQLite format 3\0"
|
|
|
|
# Above this size ``PRAGMA integrity_check`` (walks every b-tree page — minutes of pegged CPU on a
|
|
# 30 GB state.db, reading as a hung ``hermes update``) is replaced by the O(1) header+schema probe.
|
|
# Default ceiling above which ``PRAGMA integrity_check`` is skipped in favour of the (O(1)) header +
|
|
# structural probe. Sessions databases in the tens of GB are normal for heavy users, so the size-unbounded
|
|
# check is never an acceptable default on the update path. See #70553.
|
|
DEFAULT_INTEGRITY_CHECK_MAX_BYTES = 2 << 30 # 2 GiB
|
|
|
|
|
|
def verify_sqlite_integrity(
|
|
path: Path, *, check_header: bool = True, run_pragma: bool = True,
|
|
max_bytes: int = DEFAULT_INTEGRITY_CHECK_MAX_BYTES) -> dict:
|
|
"""Verify a SQLite database: existence + minimum size, header magic, then a read-only
|
|
``PRAGMA integrity_check`` (or a cheap structural probe above ``max_bytes``)."""
|
|
def _done(message: str, valid: bool = False, size: Optional[int] = None) -> dict:
|
|
return {"valid": valid, "message": message, "size": size}
|
|
try:
|
|
st = path.stat()
|
|
except FileNotFoundError:
|
|
return _done(f"not found: {path}")
|
|
except OSError as exc:
|
|
return _done(f"cannot stat: {exc}")
|
|
size = st.st_size
|
|
if size < 100: # SQLite minimum viable size (header + 1 page)
|
|
return _done(f"too small ({size} bytes) to be a valid SQLite database", size=size)
|
|
if check_header:
|
|
# Refused when a live connection exists (close() would cancel this process's POSIX locks
|
|
# — see sqlite_safe_read); verification targets offline snapshots/backup artifacts anyway.
|
|
from hermes_cli.sqlite_safe_read import read_header_bytes_preopen
|
|
head = read_header_bytes_preopen(path, length=len(_SQLITE_HEADER))
|
|
if head is None:
|
|
return _done("cannot read header", size=size)
|
|
if head != _SQLITE_HEADER:
|
|
return _done(f"missing SQLite header magic (got {head[:16].hex()!r})", size=size)
|
|
if max_bytes > 0 and size > max_bytes:
|
|
# O(1) probe: the header check caught the zeroed signature; reading sqlite_master + page
|
|
# geometry catches malformed-schema and truncated-header-page classes without a data walk.
|
|
_, exc = _query_ro_sqlite(path, lambda c: (
|
|
c.execute("PRAGMA schema_version").fetchone(),
|
|
c.execute("SELECT count(*) FROM sqlite_master").fetchone()))
|
|
if exc is not None:
|
|
kind = "failed" if isinstance(exc, sqlite3.DatabaseError) else "error"
|
|
return _done(f"schema probe {kind}: {exc}", size=size)
|
|
return _done(
|
|
f"size {size:,} bytes exceeds max_bytes {max_bytes:,}; "
|
|
"skipped PRAGMA integrity_check (header + schema probe passed)",
|
|
valid=True, size=size)
|
|
if run_pragma:
|
|
rows, exc = _query_ro_sqlite(
|
|
path, lambda c: [str(r[0]) for r in c.execute("PRAGMA integrity_check")])
|
|
if exc is not None:
|
|
kind = "cannot open database" if isinstance(exc, sqlite3.DatabaseError) else "integrity check error"
|
|
return _done(f"{kind}: {exc}", size=size)
|
|
if rows == ["ok"]:
|
|
return _done("integrity check passed", valid=True, size=size)
|
|
return _done(f"integrity check failed: {'; '.join(rows[:5])}", size=size)
|
|
return _done("header check passed", valid=True, size=size)
|
|
|
|
|
|
def _foreign_db_holder_pids(db_path: Path) -> Optional[List[int]]:
|
|
"""PIDs of OTHER processes holding *db_path* or its WAL/SHM open (Linux ``/proc`` scan).
|
|
|
|
An already-unlinked ``(deleted)`` sidecar — the #90950 split-brain fingerprint — still
|
|
counts as held. None off-Linux or when /proc fails.
|
|
"""
|
|
if not sys.platform.startswith("linux"):
|
|
return None
|
|
|
|
def _canonical(path: str) -> str:
|
|
return os.path.normcase(os.path.abspath(path.removesuffix(" (deleted)")))
|
|
|
|
def _holds_watched(fds: List[str], fd_dir: str) -> bool:
|
|
for fd in fds:
|
|
try:
|
|
target = os.readlink(f"{fd_dir}/{fd}")
|
|
except OSError:
|
|
continue
|
|
if _canonical(target) in watched:
|
|
return True
|
|
return False
|
|
|
|
canonical_db = _canonical(os.fspath(db_path))
|
|
watched = {canonical_db, canonical_db + "-wal", canonical_db + "-shm"}
|
|
pids: List[int] = []
|
|
try:
|
|
own_pid = os.getpid()
|
|
for pid_str in os.listdir("/proc"):
|
|
if not pid_str.isdigit() or int(pid_str) == own_pid:
|
|
continue
|
|
fd_dir = f"/proc/{pid_str}/fd"
|
|
try:
|
|
fds = os.listdir(fd_dir)
|
|
except OSError:
|
|
continue
|
|
if _holds_watched(fds, fd_dir):
|
|
pids.append(int(pid_str))
|
|
except OSError:
|
|
return None
|
|
return pids
|
|
|
|
|
|
def _safe_restore_db(src: Path, dst: Path) -> bool:
|
|
"""Restore snapshot *src* into live *dst* through the backup() API; unlink+move fallback.
|
|
|
|
Writing pages into the live file preserves its inode and WAL state, so other holders (gateway,
|
|
dashboard, another CLI) see the restored data instead of stale pages from a replaced inode.
|
|
The fallback runs ONLY when no other process or in-process connection holds the file
|
|
(replacing the inode under a live holder is the #90950 split-brain); otherwise it fails closed
|
|
(``False``) and the caller reports the file as skipped.
|
|
"""
|
|
try:
|
|
dst_conn = sqlite3.connect(str(dst))
|
|
# Checkpoint first so the backup starts clean rather than writing on top of a deep WAL.
|
|
with suppress(Exception):
|
|
dst_conn.execute("PRAGMA wal_checkpoint(TRUNCATE)")
|
|
with closing(sqlite3.connect(f"file:{src}?mode=ro", uri=True)) as src_conn:
|
|
src_conn.backup(dst_conn)
|
|
dst_conn.close()
|
|
with suppress(Exception):
|
|
dst.chmod(src.stat().st_mode)
|
|
return True
|
|
except Exception as exc:
|
|
logger.warning("SQLite safe restore failed for %s -> %s: %s", src, dst, exc)
|
|
return _unlink_move_restore_db(src, dst)
|
|
|
|
|
|
def _unlink_move_restore_db(src: Path, dst: Path) -> bool:
|
|
"""Fallback restore: unlink+move. Only safe when no process holds the DB open.
|
|
|
|
Replacing the inode under a live holder is the #90950 corruption class (the holder keeps
|
|
writing through a deleted-inode fd and loses its WAL index), so fail closed. The foreign-pid
|
|
scan excludes THIS process, so ``offline_file_access`` also fails CLOSED on any live
|
|
in-process connection to *dst* and holds the connection-lifecycle lock across the swap.
|
|
"""
|
|
from hermes_cli.sqlite_safe_read import LiveConnectionError, offline_file_access
|
|
try:
|
|
holders = _foreign_db_holder_pids(dst)
|
|
if holders:
|
|
logger.error("Refusing unlink+move restore of %s: process(es) %s still "
|
|
"hold the database or its WAL open. Stop them and retry.", dst, holders)
|
|
return False
|
|
with offline_file_access(dst, what="unlink+move restore of"):
|
|
tmp = dst.parent / f".{dst.name}.snap_restore"
|
|
shutil.copy2(src, tmp)
|
|
dst.unlink(missing_ok=True)
|
|
# The snapshot owns no WAL, so any -wal/-shm here belongs to the DB just unlinked (a
|
|
# killed gateway leaves them — exactly when a restore runs); SQLite would replay that
|
|
# foreign WAL over the restored file: "malformed" or resurrected post-snapshot rows.
|
|
for _sidecar_suffix in ("-wal", "-shm", "-journal"):
|
|
dst.with_name(dst.name + _sidecar_suffix).unlink(missing_ok=True)
|
|
shutil.move(str(tmp), str(dst))
|
|
return True
|
|
except LiveConnectionError as exc2:
|
|
logger.error("Refusing unlink+move restore of %s: %s Close the in-process "
|
|
"database handles (or restart Hermes) and retry.", dst, exc2)
|
|
return False
|
|
except Exception as exc2:
|
|
logger.error("Fallback restore also failed for %s -> %s: %s", src, dst, exc2)
|
|
return False
|
|
|
|
|
|
def _zip_sqlite_snapshot(zf: zipfile.ZipFile, abs_path: Path, rel_path: Path, out_path: Path) -> Optional[int]:
|
|
"""Add a WAL-safe snapshot of *abs_path* to *zf*; return its byte size, or None on failure.
|
|
|
|
Staged beside the output zip: /tmp may be a small tmpfs that cannot hold large databases.
|
|
"""
|
|
with tempfile.NamedTemporaryFile(suffix=".db", delete=False, dir=str(out_path.parent)) as tmp:
|
|
tmp_db = Path(tmp.name)
|
|
try:
|
|
if not _safe_copy_db(abs_path, tmp_db):
|
|
return None
|
|
zf.write(tmp_db, arcname=str(rel_path))
|
|
return tmp_db.stat().st_size
|
|
finally:
|
|
tmp_db.unlink(missing_ok=True)
|
|
|
|
|
|
def _write_zip_entries(
|
|
zf: zipfile.ZipFile, files_to_add: List[Tuple[Path, Path]], out_path: Path,
|
|
*, on_db_failure, on_error, on_progress, track_bytes: bool) -> int:
|
|
"""Add every ``(abs_path, rel_path)`` to *zf*, WAL-safe for ``*.db``; return bytes archived.
|
|
|
|
``on_db_failure(rel_path)`` runs when a SQLite snapshot fails (may raise to abort);
|
|
``on_error(rel_path, exc)`` records a read failure; ``on_progress(i)`` fires every 500 files;
|
|
``track_bytes`` stats plain files for the size total.
|
|
"""
|
|
total_bytes = 0
|
|
for i, (abs_path, rel_path) in enumerate(files_to_add, 1):
|
|
try:
|
|
if abs_path.suffix == ".db":
|
|
size = _zip_sqlite_snapshot(zf, abs_path, rel_path, out_path)
|
|
if size is None:
|
|
on_db_failure(rel_path)
|
|
continue
|
|
total_bytes += size
|
|
else:
|
|
zf.write(abs_path, arcname=str(rel_path))
|
|
if track_bytes:
|
|
total_bytes += abs_path.stat().st_size
|
|
except (PermissionError, OSError, ValueError) as exc:
|
|
on_error(rel_path, exc)
|
|
continue
|
|
if i % 500 == 0:
|
|
on_progress(i)
|
|
return total_bytes
|
|
|
|
|
|
def _print_capped(header: str, lines: List[str], indent: str) -> None:
|
|
"""Print *header*, then at most 10 of *lines* (each prefixed by *indent*) and a "... and N more" tail."""
|
|
print(header)
|
|
for line in lines[:10]:
|
|
print(f"{indent}{line}")
|
|
if len(lines) > 10:
|
|
print(f"{indent}... and {len(lines) - 10} more")
|
|
|
|
|
|
# --- Backup ---
|
|
|
|
def _resolve_backup_output_path(output: Optional[str]) -> Path:
|
|
"""Turn ``--output`` (file, directory, or None) into a ``.zip`` path whose parent exists;
|
|
an unwritable path exits with a one-line error, not a traceback."""
|
|
out_path = None
|
|
default_name = f"hermes-backup-{datetime.now().strftime('%Y-%m-%d-%H%M%S')}.zip"
|
|
try:
|
|
if output:
|
|
out_path = Path(output).expanduser().resolve()
|
|
if out_path.is_dir():
|
|
out_path = out_path / default_name
|
|
else:
|
|
out_path = Path.home() / default_name
|
|
if out_path.suffix.lower() != ".zip":
|
|
out_path = out_path.with_suffix(out_path.suffix + ".zip")
|
|
out_path.parent.mkdir(parents=True, exist_ok=True)
|
|
except OSError as exc:
|
|
print(f"Error: cannot write backup to {output or out_path}: {exc}")
|
|
raise SystemExit(1) from exc
|
|
return out_path
|
|
|
|
|
|
def _collect_external_entries() -> tuple[list[tuple[Path, str]], list[str]]:
|
|
"""``([(abs_path, arcname)], [skipped])`` for the memory provider's external state, arc-named
|
|
``_external/<home-relative>``; paths outside home are skipped (security + portability)."""
|
|
home_dir = Path.home().resolve()
|
|
external_to_add: list[tuple[Path, str]] = []
|
|
skipped_external: list[str] = []
|
|
for base in _collect_memory_provider_external_paths():
|
|
try:
|
|
base.resolve().relative_to(home_dir)
|
|
except (ValueError, OSError):
|
|
skipped_external.append(str(base))
|
|
continue
|
|
for fpath in _iter_external_files(base):
|
|
with suppress(ValueError, OSError):
|
|
rel_to_home = fpath.resolve().relative_to(home_dir)
|
|
external_to_add.append((fpath, _EXTERNAL_PREFIX + rel_to_home.as_posix()))
|
|
return external_to_add, skipped_external
|
|
|
|
|
|
def run_backup(args) -> None:
|
|
"""Create a zip backup of the Hermes home directory."""
|
|
hermes_root = get_default_hermes_root()
|
|
|
|
if not hermes_root.is_dir():
|
|
print(f"Error: Hermes home directory not found at {hermes_root}")
|
|
sys.exit(1)
|
|
|
|
try:
|
|
with _backup_operation_lock(hermes_root):
|
|
_run_backup_locked(args, hermes_root)
|
|
except BackupInProgressError as exc:
|
|
print(f"Error: {exc}")
|
|
raise SystemExit(2) from exc
|
|
|
|
|
|
def _run_backup_locked(args, hermes_root: Path) -> None:
|
|
"""Write a full backup while the cross-process backup slot is held."""
|
|
out_path = _resolve_backup_output_path(args.output)
|
|
scan_started = time.monotonic()
|
|
logger.info("backup phase=scan status=started")
|
|
print(f"Scanning {display_hermes_home()} ...")
|
|
skipped_dirs: set = set()
|
|
files_to_add: list[tuple[Path, Path]] = list(_iter_backup_files(hermes_root, out_path, skipped_dirs))
|
|
external_to_add, skipped_external = _collect_external_entries()
|
|
if not files_to_add and not external_to_add:
|
|
logger.info("backup phase=scan status=empty duration_ms=%.1f", (time.monotonic() - scan_started) * 1000)
|
|
print("No files to back up.")
|
|
return
|
|
|
|
file_count = len(files_to_add) + len(external_to_add)
|
|
logger.info("backup phase=scan status=complete duration_ms=%.1f files=%d",
|
|
(time.monotonic() - scan_started) * 1000, file_count)
|
|
logger.info("backup phase=archive status=started files=%d", file_count)
|
|
print(f"Backing up {file_count} files ...")
|
|
errors = []
|
|
t0 = time.monotonic()
|
|
|
|
def _progress(i: int) -> None:
|
|
print(f" {i}/{file_count} files ...")
|
|
logger.info("backup phase=archive status=progress completed=%d total=%d", i, file_count)
|
|
|
|
with _atomic_output_path(out_path) as archive_path, zipfile.ZipFile(
|
|
archive_path, "w", zipfile.ZIP_DEFLATED, compresslevel=6) as zf:
|
|
total_bytes = _write_zip_entries(
|
|
zf, files_to_add, out_path, on_progress=_progress, track_bytes=True,
|
|
on_db_failure=lambda rel: errors.append(f"{rel}: SQLite safe copy failed"),
|
|
on_error=lambda rel, exc: errors.append(f"{rel}: {exc}"))
|
|
# External memory-provider state never includes ``.db`` files in practice, so a
|
|
# straight zf.write is fine.
|
|
for abs_path, arcname in external_to_add:
|
|
try:
|
|
zf.write(abs_path, arcname=arcname)
|
|
total_bytes += abs_path.stat().st_size
|
|
except (PermissionError, OSError, ValueError) as exc:
|
|
errors.append(f"{arcname}: {exc}")
|
|
elapsed = time.monotonic() - t0
|
|
zip_size = out_path.stat().st_size
|
|
logger.info("backup phase=archive status=complete duration_ms=%.1f files=%d errors=%d bytes=%d",
|
|
elapsed * 1000, file_count, len(errors), zip_size)
|
|
print(f"\nBackup {'incomplete' if errors else 'complete'}: {out_path}\n"
|
|
f" Files: {file_count}\n"
|
|
f" Original: {_format_size(total_bytes)}\n"
|
|
f" Compressed: {_format_size(zip_size)}\n"
|
|
f" Time: {elapsed:.1f}s")
|
|
if external_to_add:
|
|
print(f"\n Included {len(external_to_add)} memory-provider file(s) stored outside {display_hermes_home()}.")
|
|
if skipped_external:
|
|
print(f"\n Skipped {len(skipped_external)} memory-provider path(s) outside your home directory "
|
|
"(not portable):\n" + "\n".join(f" {p}" for p in sorted(skipped_external)[:10]))
|
|
if skipped_dirs:
|
|
print("\n Excluded directories:\n" + "\n".join(f" {d}/" for d in sorted(skipped_dirs)))
|
|
if errors:
|
|
_print_capped(f"\n Warnings ({len(errors)} files skipped):", errors, " ")
|
|
else:
|
|
print(f"\nRestore with: hermes import {out_path.name}")
|
|
|
|
|
|
# --- Import ---
|
|
|
|
def _validate_backup_zip(zf: zipfile.ZipFile) -> tuple[bool, str]:
|
|
"""Check that a zip looks like a Hermes backup."""
|
|
names = zf.namelist()
|
|
if not names:
|
|
return False, "zip archive is empty"
|
|
# Telltale files a hermes home has — at the root or one level deep (zipped directory).
|
|
if not any(Path(n).name in {"config.yaml", ".env", "state.db"} for n in names):
|
|
return False, "zip does not appear to be a Hermes backup (no config.yaml, .env, or state databases found)"
|
|
return True, ""
|
|
|
|
|
|
def _detect_prefix(zf: zipfile.ZipFile) -> str:
|
|
"""Detect if the zip has a common directory prefix wrapping all entries."""
|
|
names = [n for n in zf.namelist() if not n.endswith("/")]
|
|
first_parts = {Path(n).parts[0] for n in names if len(Path(n).parts) > 1}
|
|
if len(first_parts) == 1 and first_parts <= {".hermes", "hermes"}:
|
|
return first_parts.pop() + "/"
|
|
return ""
|
|
|
|
|
|
def _default_new_file_mode() -> Optional[int]:
|
|
"""The mode ``open(path, "wb")`` gives a file it has to create.
|
|
|
|
``mkstemp`` always creates at 0600, so staging an import through a temp file would tighten
|
|
every *newly created* file to owner-only — the Docker/NAS volume-mount hazard
|
|
``utils._restore_file_mode`` documents.
|
|
"""
|
|
try:
|
|
current = os.umask(0o077)
|
|
os.umask(current)
|
|
except OSError:
|
|
return None
|
|
return 0o666 & ~current
|
|
|
|
|
|
def _extract_member_atomically(
|
|
zf: zipfile.ZipFile, member: str, target: Path, new_file_mode: Optional[int] = None) -> None:
|
|
"""Restore one zip member onto *target* with no truncation window.
|
|
|
|
``open(target, "wb")`` would truncate the user's file before any replacement bytes exist.
|
|
``atomic_replace`` (not bare ``os.replace``) resolves a symlinked target first, so a
|
|
dotfiles-linked ``config.yaml`` keeps the link (#16743), and falls back to copy/fsync/unlink
|
|
on ``EXDEV``/``EBUSY`` for cross-device and bind-mount installs.
|
|
"""
|
|
# Mode is None when the target does not exist: the umask-derived create-mode applies.
|
|
mode = _preserve_file_mode(target)
|
|
owner = _preserve_file_owner(target)
|
|
if mode is None:
|
|
mode = new_file_mode
|
|
else:
|
|
# Deliberately NOT a faithful copy: setuid/setgid are dropped. The bytes come from the
|
|
# archive, so carrying elevated bits would hand the zip's author the identity an existing
|
|
# setuid file runs as — and ``_external/`` publishes members anywhere under ``$HOME``.
|
|
mode &= ~(stat.S_ISUID | stat.S_ISGID)
|
|
# Truncate the stem: mkstemp adds ~16 chars and a member near NAME_MAX would otherwise fail.
|
|
fd, tmp_name = tempfile.mkstemp(dir=str(target.parent), prefix=f".{target.name[:80]}.", suffix=".partial")
|
|
try:
|
|
with os.fdopen(fd, "wb") as dst:
|
|
if mode is not None:
|
|
# Apply the mode BEFORE the replace so the target never transits through mkstemp's
|
|
# 0600 and the EXDEV/EBUSY ``copystat`` fallback copies the intended bits.
|
|
if hasattr(os, "fchmod"): # Unix-only; Windows takes the path-based chmod
|
|
os.fchmod(dst.fileno(), mode)
|
|
else:
|
|
os.chmod(tmp_name, mode)
|
|
# Stream: a multi-gigabyte state.db member must not be held in memory in one piece.
|
|
with zf.open(member) as src:
|
|
shutil.copyfileobj(src, dst)
|
|
dst.flush()
|
|
os.fsync(dst.fileno())
|
|
real_path = Path(atomic_replace(tmp_name, target))
|
|
# Owner first, mode second (as ``atomic_yaml_write``): chown drops setuid/setgid and
|
|
# ``mode`` no longer carries them, so neither step can re-elevate the restored file.
|
|
_restore_file_owner(real_path, owner)
|
|
_restore_file_mode(real_path, mode)
|
|
except BaseException:
|
|
with suppress(OSError):
|
|
os.unlink(tmp_name)
|
|
raise
|
|
|
|
|
|
def _count_session_rows(path: Path) -> Optional[Tuple[int, int]]:
|
|
"""``(sessions, messages)`` in session database *path*; read-only, best effort.
|
|
|
|
``None`` means "unknown" (missing, not a Hermes session store, unreadable) — never "zero":
|
|
acting on an unreadable database would mask the very loss this count exists to surface.
|
|
Same contract as :func:`_count_cron_jobs`.
|
|
"""
|
|
if not path.is_file():
|
|
return None
|
|
try:
|
|
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
|
|
except sqlite3.Error:
|
|
return None
|
|
try:
|
|
sessions = conn.execute("SELECT COUNT(*) FROM sessions").fetchone()[0]
|
|
messages = conn.execute("SELECT COUNT(*) FROM messages").fetchone()[0]
|
|
return int(sessions), int(messages)
|
|
except (sqlite3.Error, TypeError, ValueError):
|
|
return None
|
|
finally:
|
|
conn.close()
|
|
|
|
|
|
def _import_db_member(
|
|
zf: zipfile.ZipFile, member: str, target: Path, new_file_mode: Optional[int] = None) -> None:
|
|
"""Publish a SQLite ``.db`` member onto *target* without replacing its inode.
|
|
|
|
A rename-publish over a live database is the #65942 / #90950 corruption class: a gateway,
|
|
dashboard, or WebUI holding it open keeps serving the unlinked inode and writing sessions no
|
|
other process will see, and a sidecar WAL beside the new file describes the old database —
|
|
nothing fails, the sessions are simply gone (#100960). Route the member through the same
|
|
``_safe_restore_db`` page copy ``/snapshot restore`` uses, so the live inode is preserved and
|
|
every open connection converges. A target that does not exist yet has no holders, so it takes
|
|
the ordinary atomic publish. Raises ``OSError`` when the database could not be replaced
|
|
safely, so the caller reports a skipped file instead of a silent success.
|
|
"""
|
|
if not target.exists():
|
|
_extract_member_atomically(zf, member, target, new_file_mode)
|
|
return
|
|
# The database keeps its own mode/ownership: the bytes come from the archive, the file does not.
|
|
mode = _preserve_file_mode(target)
|
|
owner = _preserve_file_owner(target)
|
|
fd, tmp_name = tempfile.mkstemp(dir=str(target.parent), prefix=f".{target.name[:80]}.", suffix=".dbimport")
|
|
try:
|
|
with os.fdopen(fd, "wb") as dst:
|
|
with zf.open(member) as src: # stream: never hold a multi-GB state.db in memory
|
|
shutil.copyfileobj(src, dst)
|
|
dst.flush()
|
|
os.fsync(dst.fileno())
|
|
if not _safe_restore_db(Path(tmp_name), target):
|
|
raise OSError(
|
|
"live-safe restore refused or failed; the existing database was "
|
|
"left untouched. Stop the gateway/dashboard processes holding it "
|
|
"open and re-run the import."
|
|
)
|
|
_restore_file_owner(target, owner)
|
|
_restore_file_mode(target, mode)
|
|
finally:
|
|
with suppress(OSError):
|
|
os.unlink(tmp_name)
|
|
|
|
|
|
def _confirm_import_overwrite(hermes_root: Path) -> bool:
|
|
"""Prompt before importing over an existing installation; True when import may proceed."""
|
|
if not any((hermes_root / m).exists() for m in ("config.yaml", ".env")):
|
|
return True
|
|
print("\nWarning: Target directory already has Hermes configuration.\n"
|
|
"Importing will overwrite existing files with backup contents.\n")
|
|
try:
|
|
answer = input("Continue? [y/N] ").strip().lower()
|
|
except (EOFError, KeyboardInterrupt):
|
|
print("\nAborted.")
|
|
sys.exit(1)
|
|
if answer in {"y", "yes"}:
|
|
return True
|
|
print("Aborted.")
|
|
return False
|
|
|
|
|
|
def _import_members(
|
|
zf: zipfile.ZipFile, members: List[str], prefix: str, hermes_root: Path, file_count: int
|
|
) -> tuple[int, int, list[str], list[str], list[tuple[str, tuple[int, int], tuple[int, int]]]]:
|
|
"""Publish every member; return ``(restored, restored_external, errors, skipped_runtime, db_shrunk)``.
|
|
|
|
``db_shrunk`` holds ``(rel, live_counts, imported_counts)`` for every session database the
|
|
import replaced with one holding fewer rows — allowed, but never silent (#100960).
|
|
"""
|
|
errors: list[str] = []
|
|
skipped_runtime: list[str] = []
|
|
db_shrunk: list[tuple[str, tuple[int, int], tuple[int, int]]] = []
|
|
restored = restored_external = 0
|
|
home_dir = Path.home().resolve()
|
|
new_file_mode = _default_new_file_mode() # once: every member is published via mkstemp (0600)
|
|
for member in members:
|
|
# ``_external/`` members restore to their home-relative location (~/.honcho/config.json),
|
|
# NOT under HERMES_HOME; provider configs commonly hold credentials, so tighten to 0600.
|
|
external = member.startswith(_EXTERNAL_PREFIX)
|
|
if external:
|
|
rel = member[len(_EXTERNAL_PREFIX):]
|
|
target = home_dir / rel
|
|
root = home_dir
|
|
tighten = target.suffix in {".json", ".env", ".conf"} or target.name in _SECRET_FILE_NAMES
|
|
else:
|
|
rel = member[len(prefix):] if prefix and member.startswith(prefix) else member
|
|
if rel and Path(rel).name in _IMPORT_SKIP_NAMES: # see ``_IMPORT_SKIP_NAMES``
|
|
skipped_runtime.append(rel)
|
|
continue
|
|
# A ``.db`` member is page-restored into the live file; an archived WAL/SHM/journal
|
|
# describes a different database image and installed beside it (over a live sidecar)
|
|
# would replay a foreign WAL on next open. Current backups never ship these
|
|
# (_EXCLUDED_SUFFIXES); older or hand-built archives might.
|
|
if rel.endswith(_SQLITE_SIDECAR_SUFFIXES):
|
|
skipped_runtime.append(rel)
|
|
continue
|
|
target = hermes_root / rel
|
|
root = hermes_root.resolve()
|
|
tighten = target.name in _SECRET_FILE_NAMES
|
|
if not rel:
|
|
continue
|
|
|
|
label = member if external else rel
|
|
if not _is_within(target, root):
|
|
errors.append(f"{label}: path traversal blocked")
|
|
else:
|
|
try:
|
|
target.parent.mkdir(parents=True, exist_ok=True)
|
|
if target.suffix == ".db":
|
|
# Count before the write: afterwards the dropped rows are gone.
|
|
before = _count_session_rows(target)
|
|
_import_db_member(zf, member, target, new_file_mode)
|
|
after = _count_session_rows(target)
|
|
if before and after and after[1] < before[1]:
|
|
db_shrunk.append((rel, before, after))
|
|
else:
|
|
_extract_member_atomically(zf, member, target, new_file_mode)
|
|
if tighten:
|
|
try:
|
|
os.chmod(target, 0o600)
|
|
except OSError:
|
|
if not external: # external configs are tightened best-effort only
|
|
raise
|
|
restored += 1
|
|
restored_external += external
|
|
except (PermissionError, OSError) as exc:
|
|
errors.append(f"{label}: {exc}")
|
|
|
|
if restored % 500 == 0:
|
|
print(f" {restored}/{file_count} files ...")
|
|
|
|
return restored, restored_external, errors, skipped_runtime, db_shrunk
|
|
|
|
|
|
def run_import(args) -> None:
|
|
"""Restore a Hermes backup from a zip file."""
|
|
zip_path = Path(args.zipfile).expanduser().resolve()
|
|
if not zip_path.is_file():
|
|
print(f"Error: File not found: {zip_path}")
|
|
sys.exit(1)
|
|
if not zipfile.is_zipfile(zip_path):
|
|
print(f"Error: Not a valid zip file: {zip_path}")
|
|
sys.exit(1)
|
|
# The restore target is the home the command operates under (the printed "Target:");
|
|
# ``get_default_hermes_root()`` would silently retarget a profile restore at the live root.
|
|
hermes_root = get_hermes_home()
|
|
with zipfile.ZipFile(zip_path, "r") as zf:
|
|
ok, reason = _validate_backup_zip(zf)
|
|
if not ok:
|
|
print(f"Error: {reason}")
|
|
sys.exit(1)
|
|
prefix = _detect_prefix(zf)
|
|
members = [n for n in zf.namelist() if not n.endswith("/")]
|
|
file_count = len(members)
|
|
print(f"Backup contains {file_count} files\nTarget: {display_hermes_home()}")
|
|
if prefix:
|
|
print(f"Detected archive prefix: {prefix!r} (will be stripped)")
|
|
if not args.force and not _confirm_import_overwrite(hermes_root):
|
|
return
|
|
print(f"\nImporting {file_count} files ...")
|
|
hermes_root.mkdir(parents=True, exist_ok=True)
|
|
t0 = time.monotonic()
|
|
restored, restored_external, errors, skipped_runtime, db_shrunk = _import_members(
|
|
zf, members, prefix, hermes_root, file_count)
|
|
elapsed = time.monotonic() - t0
|
|
print(f"\nImport complete: {restored} files restored in {elapsed:.1f}s\n Target: {display_hermes_home()}")
|
|
if restored_external:
|
|
print(f"\n Restored {restored_external} memory-provider file(s) to "
|
|
f"their original location(s) outside {display_hermes_home()}.")
|
|
if errors:
|
|
_print_capped(f"\n Warnings ({len(errors)} files skipped):", errors, " ")
|
|
if db_shrunk:
|
|
# The backup predates work that is now overwritten — say so (#100960: twelve sessions
|
|
# disappeared with nothing logged anywhere).
|
|
print("\n ⚠ Session data replaced by older backup contents:")
|
|
for rel, before, after in db_shrunk:
|
|
print(f" {rel}: {before[0]} session(s) / {before[1]} message(s)"
|
|
f" -> {after[0]} / {after[1]}")
|
|
print(" Anything recorded after the backup was taken is not in it. "
|
|
"Recover from a newer backup or snapshot: hermes snapshot list")
|
|
if skipped_runtime:
|
|
_print_capped(f"\n Preserved {len(skipped_runtime)} runtime state "
|
|
f"file(s) (kept this machine's, not the backup's):",
|
|
sorted(skipped_runtime), " ")
|
|
restored_profiles = _restore_profile_wrappers(hermes_root)
|
|
print()
|
|
if not (hermes_root / "hermes-agent").is_dir():
|
|
print("Note: The hermes-agent codebase was not included in the backup.\n"
|
|
" If this is a fresh install, run: hermes update")
|
|
if restored_profiles:
|
|
print("\nTo re-enable gateway services for profiles:")
|
|
for pname in restored_profiles:
|
|
print(f" hermes -p {pname} gateway install")
|
|
_revive_gateway_after_import(hermes_root)
|
|
print("Done. Your Hermes configuration has been restored.")
|
|
|
|
|
|
def _restore_profile_wrappers(hermes_root: Path) -> List[str]:
|
|
"""Re-create shell wrapper scripts for restored named profiles; return the profile names seen."""
|
|
profiles_dir = hermes_root / "profiles"
|
|
restored_profiles: list[tuple[str, bool]] = []
|
|
if not profiles_dir.is_dir():
|
|
return []
|
|
try:
|
|
from hermes_cli.profiles import (
|
|
create_wrapper_script, check_alias_collision, _is_wrapper_dir_in_path, _get_wrapper_dir)
|
|
for entry in sorted(profiles_dir.iterdir()):
|
|
if not entry.is_dir() or not any((entry / m).exists() for m in ("config.yaml", ".env")):
|
|
continue # only profiles with config get wrappers
|
|
profile_name = entry.name
|
|
collision = check_alias_collision(profile_name)
|
|
if collision:
|
|
print(f" Skipped alias '{profile_name}': {collision}")
|
|
restored_profiles.append(
|
|
(profile_name, not collision and create_wrapper_script(profile_name) is not None))
|
|
if restored_profiles:
|
|
created = [n for n, ok in restored_profiles if ok]
|
|
skipped = [n for n, ok in restored_profiles if not ok]
|
|
if created:
|
|
print(f"\n Profile aliases restored: {', '.join(created)}")
|
|
if skipped:
|
|
print(f" Profile aliases skipped: {', '.join(skipped)}")
|
|
if not _is_wrapper_dir_in_path():
|
|
print(f"\n Note: {_get_wrapper_dir()} is not in your PATH.\n"
|
|
" Add to your shell config (~/.bashrc or ~/.zshrc):\n"
|
|
' export PATH="$HOME/.local/bin:$PATH"')
|
|
except ImportError: # hermes_cli.profiles unavailable (fresh install)
|
|
if any(profiles_dir.iterdir()):
|
|
print("\n Profiles detected but aliases could not be created.\n"
|
|
" Run: hermes profile list (after installing hermes)")
|
|
return [n for n, _ in restored_profiles]
|
|
|
|
|
|
def _revive_gateway_after_import(hermes_root: Path) -> None:
|
|
"""Install/start the gateway service after a restore, best-effort and prompt-free.
|
|
|
|
Bot tokens and cron jobs are inert without a gateway (a platform-less gateway is supported, so
|
|
this is safe for any backup); failures print a manual fallback, never fail the import. Only
|
|
revived when the restore landed in the default home or no other install exists: a sandbox or
|
|
profile restore must not install a second gateway on the default service name.
|
|
"""
|
|
native_default = _get_platform_default_hermes_home()
|
|
if hermes_root != native_default and any(
|
|
(native_default / marker).exists() for marker in ("config.yaml", ".env", "state.db")):
|
|
print("\nRestored into a non-default home; leaving the gateway service alone to avoid clashing "
|
|
f"with the install at {native_default}.\n"
|
|
"To start a gateway for this home, run: hermes gateway install")
|
|
return
|
|
try:
|
|
from hermes_cli.gateway import ensure_gateway_service, _is_service_running
|
|
if not _is_service_running():
|
|
print()
|
|
ensure_gateway_service(context="import")
|
|
except Exception:
|
|
print("\nStart the gateway to activate cron jobs and messaging:\n hermes gateway install")
|
|
|
|
|
|
# --- Quick state snapshots (used by /snapshot slash command and hermes backup --quick) ---
|
|
|
|
# Critical state files (relative to HERMES_HOME) for quick snapshots; everything else is
|
|
# regeneratable or managed separately (skills, repo, sessions/). Entries may be files OR
|
|
# directories (recursive); missing entries are skipped. Pairing data lives in platform JSON blobs
|
|
# outside state.db, so it is listed explicitly — ``hermes update`` snapshots this set (#15733).
|
|
_QUICK_STATE_FILES = (
|
|
"state.db", "config.yaml", ".env", "auth.json", "cron/jobs.json", "cron/executions.db",
|
|
"gateway_state.json", "channel_directory.json", "channel_aliases.json", "processes.json",
|
|
"gateway/discord_message_recovery.db", # Discord reconnect replay ledger
|
|
# Per-profile user stores, destroyed if the update flow replaces the file and the post-update
|
|
# schema-init re-creates an empty one (#52889). Skipped when outside HERMES_HOME.
|
|
"projects.db", # per-profile project store
|
|
"response_store.db", # gateway conversation history / tool payloads
|
|
"memory_store.db", # holographic memory facts/entities
|
|
"verification_evidence.db", # agent verification audit trail
|
|
"kanban.db", # default board (back-compat <root>/kanban.db)
|
|
"kanban/boards", # non-default boards (workspaces/ + attachments/ skipped as regenerable)
|
|
# Pairing stores (generic + per-platform JSONs outside state.db)
|
|
"pairing", # legacy location (gateway/pairing.py)
|
|
"platforms/pairing", # new location (gateway/pairing.py)
|
|
"feishu_comment_pairing.json", # Feishu comment subscription pairings
|
|
)
|
|
|
|
_QUICK_DEFAULT_KEEP = 20
|
|
|
|
|
|
def _quick_snapshot_root(hermes_home: Optional[Path] = None) -> Path:
|
|
home = hermes_home or get_hermes_home()
|
|
return home / _QUICK_SNAPSHOTS_DIR
|
|
|
|
|
|
def create_quick_snapshot(
|
|
label: Optional[str] = None, hermes_home: Optional[Path] = None, keep: Optional[int] = None,
|
|
max_file_size: Optional[int] = None) -> Optional[str]:
|
|
"""Create one atomic quick snapshot while holding the shared backup slot."""
|
|
home = hermes_home or get_hermes_home()
|
|
with _backup_operation_lock(home):
|
|
return _create_quick_snapshot_locked(label, home, keep, max_file_size)
|
|
|
|
|
|
def _quick_snapshot_candidates(home: Path):
|
|
"""Yield ``(src, rel_posix, in_dir)`` for every regular file a quick snapshot captures; heavy
|
|
regenerable per-board subtrees (workspaces, attachments) are skipped."""
|
|
for rel in _QUICK_STATE_FILES:
|
|
src = home / rel
|
|
if src.is_dir():
|
|
for sub in filter(Path.is_file, src.rglob("*")):
|
|
sub_rel = sub.relative_to(home).as_posix()
|
|
if "/workspaces/" in f"/{sub_rel}/" or "/attachments/" in f"/{sub_rel}/":
|
|
continue
|
|
yield sub, sub_rel, True
|
|
elif src.is_file():
|
|
yield src, rel, False
|
|
|
|
|
|
def _copy_quick_snapshot_files(
|
|
home: Path, staging_dir: Path, max_file_size: Optional[int]
|
|
) -> tuple[Dict[str, int], list[str], list[str]]:
|
|
"""Copy every quick-snapshot candidate into *staging_dir*.
|
|
|
|
Returns ``(manifest {rel: size}, failed_dbs, oversized_skipped)``. The last two are snapshot
|
|
incompleteness (#68805): the caller must suppress pruning so the older snapshot that may hold
|
|
the only recoverable DB survives.
|
|
"""
|
|
manifest: Dict[str, int] = {}
|
|
failed_dbs: list[str] = []
|
|
oversized_skipped: list[str] = []
|
|
for src, rel, in_dir in _quick_snapshot_candidates(home):
|
|
if max_file_size is not None:
|
|
try:
|
|
size = src.stat().st_size
|
|
except OSError:
|
|
size = None
|
|
if size is not None and size > max_file_size:
|
|
print(f" ⚠ Snapshot: skipping {rel} "
|
|
f"({_format_size(size)} exceeds {_format_size(max_file_size)} limit)")
|
|
logger.warning("Quick snapshot skipped %s: %d bytes exceeds %d byte limit", rel, size, max_file_size)
|
|
if src.suffix == ".db":
|
|
oversized_skipped.append(rel)
|
|
continue
|
|
dst = staging_dir / rel
|
|
dst.parent.mkdir(parents=True, exist_ok=True)
|
|
try:
|
|
# SQLite DBs go through the WAL-safe backup() path (the gateway may hold the WAL open).
|
|
if src.suffix == ".db":
|
|
if not _safe_copy_db(src, dst):
|
|
failed_dbs.append(rel)
|
|
print(f" ⚠ Snapshot: SQLite safe copy FAILED for {rel} — file may be locked or corrupted")
|
|
if is_zeroed_sqlite_file(src):
|
|
nuls = " of NULs?" if in_dir else ""
|
|
print(f" ⚠ Snapshot: {rel} looks ZEROED "
|
|
f"(no SQLite header; {src.stat().st_size} bytes{nuls})")
|
|
continue
|
|
else:
|
|
shutil.copy2(src, dst)
|
|
manifest[rel] = dst.stat().st_size
|
|
except (OSError, PermissionError) as exc:
|
|
logger.warning("Could not snapshot %s: %s", rel, exc)
|
|
return manifest, failed_dbs, oversized_skipped
|
|
|
|
|
|
def _create_quick_snapshot_locked(
|
|
label: Optional[str], home: Path, keep: Optional[int], max_file_size: Optional[int]
|
|
) -> Optional[str]:
|
|
"""Copy the quick-snapshot set to a timestamped dir under state-snapshots/ and prune old ones.
|
|
|
|
``max_file_size`` skips (with a warning) larger files: the pre-update snapshot uses it so a
|
|
multi-GB ``state.db`` never stalls ``hermes update`` while the small files are always captured.
|
|
"""
|
|
root = _quick_snapshot_root(home)
|
|
ts = datetime.now(timezone.utc).strftime("%Y%m%d-%H%M%S")
|
|
base_snap_id = f"{ts}-{label}" if label else ts
|
|
snap_id, suffix = base_snap_id, 2
|
|
while (root / snap_id).exists():
|
|
snap_id = f"{base_snap_id}-{suffix}"
|
|
suffix += 1
|
|
staging_dir = root / f".{snap_id}.{os.getpid()}.partial"
|
|
shutil.rmtree(staging_dir, ignore_errors=True)
|
|
staging_dir.mkdir(parents=True, exist_ok=False)
|
|
logger.info("quick snapshot phase=copy status=started id=%s", snap_id)
|
|
manifest, failed_dbs, oversized_skipped = _copy_quick_snapshot_files(home, staging_dir, max_file_size)
|
|
if failed_dbs:
|
|
# Surface on stdout: a log-and-continue made a missing state.db backup look like a
|
|
# successful pre-update snapshot (#68474).
|
|
print(f" ⚠ CRITICAL: could not snapshot DB file(s): {', '.join(failed_dbs)}\n"
|
|
f" ⚠ If sessions disappear after update, check {root} and run: hermes snapshot list")
|
|
logger.error("Quick snapshot failed to capture DB file(s): %s", ", ".join(failed_dbs))
|
|
if not manifest:
|
|
shutil.rmtree(staging_dir, ignore_errors=True)
|
|
if failed_dbs:
|
|
# Distinguish "nothing to snapshot" from "state.db present but unreadable"
|
|
print(f" ⚠ Snapshot aborted: no files captured (failed DBs: {', '.join(failed_dbs)})")
|
|
return None
|
|
meta = {
|
|
"id": snap_id, "timestamp": ts, "label": label, "file_count": len(manifest),
|
|
"total_size": sum(manifest.values()), "files": manifest,
|
|
"failed_dbs": failed_dbs, "oversized_skipped": oversized_skipped,
|
|
}
|
|
with open(staging_dir / "manifest.json", "w", encoding="utf-8") as f:
|
|
json.dump(meta, f, indent=2)
|
|
os.replace(staging_dir, root / snap_id)
|
|
# Auto-prune (pre-update callers pass a smaller keep so state.db copies don't accumulate).
|
|
# Skip when a DB failed to capture OR was skipped for size (#68805): the snapshot is
|
|
# incomplete and the older one may hold the only recoverable database.
|
|
if not (failed_dbs or oversized_skipped):
|
|
_prune_oldest(_snapshot_dirs(root), _QUICK_DEFAULT_KEEP if keep is None else keep, shutil.rmtree, "snapshot")
|
|
else:
|
|
if oversized_skipped:
|
|
print(" ⚠ Skipping snapshot prune: DB file(s) skipped for size: " + ", ".join(oversized_skipped))
|
|
logger.warning("Quick snapshot skipped oversized DB file(s): %s", ", ".join(oversized_skipped))
|
|
logger.warning(
|
|
"Skipping snapshot prune because %d DB(s) failed to capture and/or %d were oversized "
|
|
"— preserving older snapshots as recovery source",
|
|
len(failed_dbs), len(oversized_skipped))
|
|
logger.info("quick snapshot phase=copy status=complete id=%s files=%d bytes=%d",
|
|
snap_id, len(manifest), sum(manifest.values()))
|
|
return snap_id
|
|
|
|
|
|
def _newest_first(root: Path, keep_entry) -> List[Path]:
|
|
"""Entries of *root* passing ``keep_entry``, newest (by name) first; ``[]`` if *root* is missing."""
|
|
if not root.exists():
|
|
return []
|
|
return sorted(filter(keep_entry, root.iterdir()), key=lambda p: p.name, reverse=True)
|
|
|
|
|
|
def _snapshot_dirs(root: Path) -> List[Path]:
|
|
"""Published snapshot directories under *root*, newest first."""
|
|
return _newest_first(root, lambda d: d.is_dir() and not d.name.startswith(".")
|
|
and not d.name.endswith(".partial"))
|
|
|
|
|
|
def list_quick_snapshots(limit: int = 20, hermes_home: Optional[Path] = None) -> List[Dict[str, Any]]:
|
|
"""List existing quick state snapshots, most recent first."""
|
|
results = []
|
|
for d in _snapshot_dirs(_quick_snapshot_root(hermes_home)):
|
|
manifest_path = d / "manifest.json"
|
|
if manifest_path.exists():
|
|
try:
|
|
results.append(json.loads(manifest_path.read_text(encoding="utf-8")))
|
|
except (json.JSONDecodeError, OSError):
|
|
results.append({"id": d.name, "file_count": 0, "total_size": 0})
|
|
if len(results) >= limit:
|
|
break
|
|
return results
|
|
|
|
|
|
def restore_quick_snapshot(snapshot_id: str, hermes_home: Optional[Path] = None) -> bool:
|
|
"""Restore state from a quick snapshot."""
|
|
home = hermes_home or get_hermes_home()
|
|
root = _quick_snapshot_root(home)
|
|
# Reject ids with separators or traversal so ``root / snapshot_id`` stays inside root.
|
|
if not snapshot_id or "/" in snapshot_id or "\\" in snapshot_id or snapshot_id in (".", ".."):
|
|
logger.error("Invalid snapshot_id: %s", snapshot_id)
|
|
return False
|
|
snap_dir = root / snapshot_id
|
|
if not _is_within(snap_dir, root.resolve()): # handles symlinks etc.
|
|
logger.error("Snapshot path traversal blocked for id: %s", snapshot_id)
|
|
return False
|
|
manifest_path = snap_dir / "manifest.json"
|
|
if not snap_dir.is_dir() or not manifest_path.exists():
|
|
return False
|
|
with open(manifest_path, encoding="utf-8") as f:
|
|
meta = json.load(f)
|
|
snap_res, home_res = snap_dir.resolve(), home.resolve()
|
|
restored = 0
|
|
for rel in meta.get("files", {}):
|
|
src = snap_dir / rel
|
|
dst = home / rel
|
|
if not (_is_within(src, snap_res) and _is_within(dst, home_res)):
|
|
logger.error("Manifest path traversal blocked: %s", rel)
|
|
continue
|
|
if not src.exists():
|
|
continue
|
|
dst.parent.mkdir(parents=True, exist_ok=True)
|
|
try:
|
|
if dst.suffix == ".db":
|
|
# Through the backup API so live connections see the restored data instead of
|
|
# stale pages from a replaced inode (#65942).
|
|
if not _safe_restore_db(src, dst):
|
|
# Refused (live holder) or failed: destination untouched — a failure, not a restore.
|
|
logger.error("Failed to restore %s: live-safe restore refused", rel)
|
|
continue
|
|
else:
|
|
shutil.copy2(src, dst)
|
|
restored += 1
|
|
except (OSError, PermissionError) as exc:
|
|
logger.error("Failed to restore %s: %s", rel, exc)
|
|
logger.info("Restored %d files from snapshot %s", restored, snapshot_id)
|
|
return restored > 0
|
|
|
|
|
|
# Kept in sync with ``_QUICK_STATE_FILES`` and ``cron/jobs.py``'s ``JOBS_FILE``.
|
|
_CRON_JOBS_REL = "cron/jobs.json"
|
|
|
|
|
|
def _count_cron_jobs(path: Path) -> Optional[int]:
|
|
"""Number of cron jobs in ``path`` (canonical ``{"jobs": [...]}`` or legacy bare list).
|
|
|
|
``None`` if missing or unparseable — "unknown", not zero: acting on an unreadable file could
|
|
mask a real corruption the user needs to see.
|
|
"""
|
|
if not path.is_file():
|
|
return None
|
|
try:
|
|
# utf-8-sig as cron/jobs.load_jobs: a Windows-editor BOM would otherwise read as
|
|
# "unreadable" and silently disable the post-update auto-restore safety net.
|
|
with open(path, "r", encoding="utf-8-sig") as f:
|
|
data = json.load(f)
|
|
except (OSError, json.JSONDecodeError):
|
|
return None
|
|
if isinstance(data, dict):
|
|
data = data.get("jobs", [])
|
|
return len(data) if isinstance(data, list) else None
|
|
|
|
|
|
def restore_cron_jobs_if_emptied(snapshot_id: str, hermes_home: Optional[Path] = None) -> Optional[Dict[str, Any]]:
|
|
"""Safety net for silent cron-job loss across ``hermes update``.
|
|
|
|
Conservative: restores only when the snapshot had MORE jobs than the live file (a user who
|
|
deleted jobs is never second-guessed); an unreadable live file is left so corruption surfaces.
|
|
|
|
Config-version migrations have been observed to leave ``cron/jobs.json`` valid-but-empty after an
|
|
update, silently dropping every scheduled job (issue #34600). The desktop scheduler can also overwrite
|
|
the file with its own small set of internally-tracked crons, causing partial loss (issue 52144).
|
|
"""
|
|
if not snapshot_id:
|
|
return None
|
|
home = hermes_home or get_hermes_home()
|
|
live_path = home / _CRON_JOBS_REL
|
|
live_count = _count_cron_jobs(live_path)
|
|
if live_count is None:
|
|
return None
|
|
snap_path = _quick_snapshot_root(home) / snapshot_id / _CRON_JOBS_REL
|
|
snap_count = _count_cron_jobs(snap_path)
|
|
# Fewer live jobs than the snapshot catches both total loss (0 vs N) and partial loss
|
|
# (1 vs 19) — the desktop scheduler can overwrite jobs.json with its own small set.
|
|
if not snap_count or live_count >= snap_count: # None or 0 — nothing worth restoring
|
|
return None
|
|
try:
|
|
live_path.parent.mkdir(parents=True, exist_ok=True)
|
|
shutil.copy2(snap_path, live_path)
|
|
except (OSError, PermissionError) as exc:
|
|
logger.error("Cron jobs were emptied during update but auto-restore failed: %s", exc)
|
|
return None
|
|
logger.warning(
|
|
"Restored %d cron job(s) from pre-update snapshot %s "
|
|
"(live file had %d job(s), snapshot had %d — jobs were lost during migration)",
|
|
snap_count, snapshot_id, live_count, snap_count)
|
|
return {"restored": True, "job_count": snap_count, "snapshot_id": snapshot_id}
|
|
|
|
|
|
def _sibling_profile_homes(invoking_home: Path) -> list[tuple[str, Path]]:
|
|
"""(name, home) for every OTHER profile on this install (the invoking one is snapshotted
|
|
separately). The update's code swap touches every profile, so its snapshot must too (#66140).
|
|
Never raises."""
|
|
homes: list[tuple[str, Path]] = []
|
|
try:
|
|
from hermes_cli.profiles import _get_default_hermes_home, _get_profiles_root, _PROFILE_ID_RE
|
|
invoking = invoking_home.resolve()
|
|
default_home = _get_default_hermes_home()
|
|
if default_home.is_dir() and default_home.resolve() != invoking:
|
|
homes.append(("default", default_home))
|
|
root = _get_profiles_root()
|
|
if root.is_dir():
|
|
for entry in sorted(root.iterdir()):
|
|
if (entry.is_dir() and entry.name != "default" and _PROFILE_ID_RE.match(entry.name)
|
|
and entry.resolve() != invoking):
|
|
homes.append((entry.name, entry))
|
|
except Exception as exc:
|
|
logger.debug("Sibling profile enumeration failed: %s", exc)
|
|
return homes
|
|
|
|
|
|
def create_pre_update_snapshots_all_profiles(
|
|
invoking_home: Optional[Path] = None, keep: Optional[int] = None, max_file_size: Optional[int] = None
|
|
) -> Dict[str, str]:
|
|
"""Pre-update quick snapshots for every SIBLING profile (#66140), same set/size cap/keep policy
|
|
as the invoking profile's; each lands under its OWN ``<home>/state-snapshots/``."""
|
|
results: Dict[str, str] = {}
|
|
home = invoking_home or get_hermes_home()
|
|
for name, profile_home in _sibling_profile_homes(home):
|
|
try:
|
|
snap_id = create_quick_snapshot(
|
|
label="pre-update", hermes_home=profile_home, keep=keep, max_file_size=max_file_size)
|
|
if snap_id:
|
|
results[name] = snap_id
|
|
except Exception as exc:
|
|
logger.debug("Pre-update snapshot for profile %s failed: %s", name, exc)
|
|
return results
|
|
|
|
|
|
# Config paths the update flow must never change (#64160): model routing and the MoA section are
|
|
# consumed machine-wide, so an update/repair cycle that rewrites them silently redirects paid
|
|
# inference. Dotted paths into raw config.yaml; a single-element tuple protects a whole section.
|
|
_PROTECTED_CONFIG_PATHS: Tuple[Tuple[str, ...], ...] = (
|
|
("model", "provider"), ("model", "default"), ("model", "base_url"), ("model", "api_key"),
|
|
("moa",))
|
|
|
|
|
|
def _read_raw_yaml_dict(path: Path) -> Optional[Dict[str, Any]]:
|
|
"""Parse ``path`` as a YAML mapping. ``None`` = missing/unreadable/non-dict."""
|
|
if not path.is_file():
|
|
return None
|
|
try:
|
|
import yaml
|
|
data = yaml.safe_load(path.read_text(encoding="utf-8"))
|
|
except Exception:
|
|
return None
|
|
return data if isinstance(data, dict) else None
|
|
|
|
|
|
def _get_config_path_value(data: Dict[str, Any], dotted: Tuple[str, ...]) -> Any:
|
|
node: Any = data
|
|
for key in dotted:
|
|
if not isinstance(node, dict):
|
|
return None
|
|
node = node.get(key)
|
|
return node
|
|
|
|
|
|
def _set_config_path_value(data: Dict[str, Any], dotted: Tuple[str, ...], value: Any) -> None:
|
|
node = data
|
|
for key in dotted[:-1]:
|
|
if not isinstance(node.get(key), dict):
|
|
node[key] = {}
|
|
node = node[key]
|
|
node[dotted[-1]] = value
|
|
|
|
|
|
def restore_config_model_settings_if_rewritten(
|
|
snapshot_id: str, hermes_home: Optional[Path] = None) -> Optional[Dict[str, Any]]:
|
|
"""Safety net for silent config.yaml model/MoA loss across ``hermes update``.
|
|
|
|
Mirrors :func:`restore_cron_jobs_if_emptied`: restore only the protected keys — never the
|
|
whole file — whose user-set value in the same-run pre-update snapshot changed or vanished.
|
|
"""
|
|
if not snapshot_id:
|
|
return None
|
|
home = hermes_home or get_hermes_home()
|
|
live_path = home / "config.yaml"
|
|
snap = _read_raw_yaml_dict(_quick_snapshot_root(home) / snapshot_id / "config.yaml")
|
|
if not snap:
|
|
return None # no snapshot copy — nothing to compare against
|
|
live = _read_raw_yaml_dict(live_path)
|
|
if live is None:
|
|
# Missing/unparseable live config is a failure the user should see (matches the cron net).
|
|
return None
|
|
restored_keys: list[str] = []
|
|
for dotted in _PROTECTED_CONFIG_PATHS:
|
|
snap_val = _get_config_path_value(snap, dotted)
|
|
if snap_val in (None, "", {}, []):
|
|
continue # user never set it — nothing to protect
|
|
if _get_config_path_value(live, dotted) != snap_val:
|
|
_set_config_path_value(live, dotted, snap_val)
|
|
restored_keys.append(".".join(dotted))
|
|
if not restored_keys:
|
|
return None
|
|
try:
|
|
from utils import atomic_yaml_write
|
|
atomic_yaml_write(live_path, live)
|
|
except (OSError, PermissionError) as exc:
|
|
logger.error("config.yaml model settings were rewritten during update but auto-restore failed: %s", exc)
|
|
return None
|
|
logger.warning(
|
|
"Restored user config value(s) %s from pre-update snapshot %s — "
|
|
"the update flow rewrote them (#64160)", ", ".join(restored_keys), snapshot_id)
|
|
return {"restored": True, "keys": restored_keys, "snapshot_id": snapshot_id}
|
|
|
|
|
|
def _restore_all_sibling_profiles(
|
|
profile_snapshots: Dict[str, str], invoking_home: Optional[Path], restore_fn, failure_log: str
|
|
) -> list[Dict[str, Any]]:
|
|
"""Run ``restore_fn(snap_id, hermes_home=...)`` for every sibling against ITS OWN
|
|
same-generation snapshot; one result dict (plus ``profile`` key) per restored profile.
|
|
Never raises."""
|
|
restored: list[Dict[str, Any]] = []
|
|
if not profile_snapshots:
|
|
return restored
|
|
home = invoking_home or get_hermes_home()
|
|
by_name = dict(_sibling_profile_homes(home))
|
|
for name, snap_id in profile_snapshots.items():
|
|
profile_home = by_name.get(name)
|
|
if profile_home is None:
|
|
continue
|
|
try:
|
|
result = restore_fn(snap_id, hermes_home=profile_home)
|
|
except Exception as exc:
|
|
logger.debug(failure_log, name, exc)
|
|
continue
|
|
if result:
|
|
result["profile"] = name
|
|
restored.append(result)
|
|
return restored
|
|
|
|
|
|
def restore_config_model_settings_all_profiles(
|
|
profile_snapshots: Dict[str, str], invoking_home: Optional[Path] = None) -> list[Dict[str, Any]]:
|
|
"""Run the config model-settings safety net for every sibling profile (see ``_restore_all_sibling_profiles``)."""
|
|
return _restore_all_sibling_profiles(
|
|
profile_snapshots, invoking_home, restore_config_model_settings_if_rewritten,
|
|
"Config model-settings restore check for profile %s failed: %s")
|
|
|
|
|
|
def restore_cron_jobs_all_profiles(
|
|
profile_snapshots: Dict[str, str], invoking_home: Optional[Path] = None) -> list[Dict[str, Any]]:
|
|
"""Run the cron-jobs safety net for every sibling profile (#66140); ``profile_snapshots`` comes
|
|
from :func:`create_pre_update_snapshots_all_profiles`, so restores are same-generation."""
|
|
return _restore_all_sibling_profiles(
|
|
profile_snapshots, invoking_home, restore_cron_jobs_if_emptied,
|
|
"Cron restore check for profile %s failed: %s")
|
|
|
|
|
|
def _prune_oldest(newest_first: List[Path], keep: int, remove, what: str) -> int:
|
|
"""``remove(path)`` every entry past the first *keep*; return how many succeeded."""
|
|
deleted = 0
|
|
for p in newest_first[keep:]:
|
|
try:
|
|
remove(p)
|
|
deleted += 1
|
|
except OSError as exc:
|
|
logger.warning("Failed to prune %s %s: %s", what, p.name, exc)
|
|
return deleted
|
|
|
|
|
|
def prune_quick_snapshots(keep: int = _QUICK_DEFAULT_KEEP, hermes_home: Optional[Path] = None) -> int:
|
|
"""Remove oldest quick snapshots beyond the keep limit. Returns count deleted."""
|
|
return _prune_oldest(_snapshot_dirs(_quick_snapshot_root(hermes_home)), keep, shutil.rmtree, "snapshot")
|
|
|
|
|
|
def run_quick_backup(args) -> None:
|
|
"""CLI entry point for hermes backup --quick."""
|
|
snap_id = create_quick_snapshot(label=getattr(args, "label", None))
|
|
if snap_id:
|
|
print(f"State snapshot created: {snap_id}\n"
|
|
f" {len(list_quick_snapshots())} snapshot(s) stored in {display_hermes_home()}/state-snapshots/\n"
|
|
f" Restore with: /snapshot restore {snap_id}")
|
|
else:
|
|
print("No state files found to snapshot.")
|
|
|
|
|
|
# --- Shared full-zip backup helper ---
|
|
|
|
def _write_full_zip_backup(out_path: Path, hermes_root: Path) -> Optional[Path]:
|
|
"""Full zip snapshot of ``hermes_root`` to ``out_path`` under the backup slot (same rules as
|
|
:func:`run_backup`); None when nothing to back up, another backup running, or write error."""
|
|
try:
|
|
with _backup_operation_lock(hermes_root):
|
|
return _write_full_zip_backup_locked(out_path, hermes_root)
|
|
except BackupInProgressError as exc:
|
|
logger.warning("Full-zip backup skipped: %s", exc)
|
|
return None
|
|
|
|
|
|
def _write_full_zip_backup_locked(out_path: Path, hermes_root: Path) -> Optional[Path]:
|
|
scan_started = time.monotonic()
|
|
logger.info("automatic backup phase=scan status=started")
|
|
try:
|
|
files_to_add = list(_iter_backup_files(hermes_root, out_path))
|
|
except OSError as exc:
|
|
logger.warning("Full-zip backup: walk failed: %s", exc)
|
|
return None
|
|
if not files_to_add:
|
|
return None
|
|
logger.info("automatic backup phase=scan status=complete duration_ms=%.1f files=%d",
|
|
(time.monotonic() - scan_started) * 1000, len(files_to_add))
|
|
|
|
def _db_failure(rel_path: Path) -> None:
|
|
logger.warning("Full-zip backup aborted: SQLite snapshot failed for %s", rel_path)
|
|
raise _SQLiteSnapshotError(str(rel_path))
|
|
|
|
archive_started = time.monotonic()
|
|
try:
|
|
with _atomic_output_path(out_path) as archive_path, zipfile.ZipFile(
|
|
archive_path, "w", zipfile.ZIP_DEFLATED, compresslevel=6) as zf:
|
|
_write_zip_entries(
|
|
zf, files_to_add, out_path, on_db_failure=_db_failure, track_bytes=False,
|
|
on_error=lambda rel, exc: logger.debug("Skipping %s in zip backup: %s", rel, exc),
|
|
on_progress=lambda i: logger.info(
|
|
"automatic backup phase=archive status=progress completed=%d total=%d", i, len(files_to_add)))
|
|
except (OSError, _SQLiteSnapshotError) as exc:
|
|
# The hidden partial is already gone; ``out_path`` may be a previous valid backup: keep it.
|
|
logger.warning("Full-zip backup: zip write failed: %s", exc)
|
|
return None
|
|
logger.info("automatic backup phase=archive status=complete duration_ms=%.1f files=%d bytes=%d",
|
|
(time.monotonic() - archive_started) * 1000, len(files_to_add),
|
|
out_path.stat().st_size)
|
|
return out_path
|
|
|
|
|
|
# --- Pre-update / pre-migration auto-backups ---
|
|
|
|
_PRE_UPDATE_BACKUPS_DIR = "backups"
|
|
_PRE_UPDATE_PREFIX = "pre-update-"
|
|
_PRE_UPDATE_DEFAULT_KEEP = 5
|
|
_PRE_MIGRATION_PREFIX = "pre-migration-"
|
|
_PRE_MIGRATION_DEFAULT_KEEP = 5
|
|
|
|
|
|
def _prune_prefixed_zips(backup_dir: Path, prefix: str, keep: int, what: str) -> int:
|
|
"""Remove oldest ``<prefix>*.zip`` in *backup_dir* beyond *keep*; return count deleted.
|
|
|
|
Only prefix-matched files are touched, so hand-made zips or other backup kinds survive.
|
|
"""
|
|
backups = _newest_first(backup_dir, lambda p: p.is_file() and p.name.startswith(prefix)
|
|
and p.suffix.lower() == ".zip")
|
|
return _prune_oldest(backups, keep, Path.unlink, what)
|
|
|
|
|
|
def _create_prefixed_full_backup(
|
|
hermes_home: Optional[Path], prefix: str, keep: int, what: str, prune_what: str) -> Optional[Path]:
|
|
"""Write ``<HERMES_HOME>/backups/<prefix><timestamp>.zip`` and prune older same-prefix zips.
|
|
Returns the path, or ``None`` if nothing to back up or the write failed. Never raises."""
|
|
hermes_root = hermes_home or get_default_hermes_root()
|
|
if not hermes_root.is_dir():
|
|
return None
|
|
backup_dir = hermes_root / _PRE_UPDATE_BACKUPS_DIR
|
|
try:
|
|
backup_dir.mkdir(parents=True, exist_ok=True)
|
|
except OSError as exc:
|
|
logger.warning("Could not create %s backup dir %s: %s", what, backup_dir, exc)
|
|
return None
|
|
out_path = backup_dir / f"{prefix}{datetime.now().strftime('%Y-%m-%d-%H%M%S')}.zip"
|
|
if _write_full_zip_backup(out_path, hermes_root) is None:
|
|
return None
|
|
_prune_prefixed_zips(backup_dir, prefix, keep, prune_what)
|
|
return out_path
|
|
|
|
|
|
def create_pre_update_backup(
|
|
hermes_home: Optional[Path] = None, keep: int = _PRE_UPDATE_DEFAULT_KEEP) -> Optional[Path]:
|
|
"""Full zip backup to ``backups/pre-update-<timestamp>.zip``, auto-pruned; ``None`` if nothing
|
|
was found or the backup failed. Never raises — ``hermes update`` continues anyway."""
|
|
return _create_prefixed_full_backup(hermes_home, _PRE_UPDATE_PREFIX, max(keep, 1), "pre-update", "backup")
|
|
|
|
|
|
def create_pre_migration_backup(
|
|
hermes_home: Optional[Path] = None, keep: int = _PRE_MIGRATION_DEFAULT_KEEP) -> Optional[Path]:
|
|
"""Full zip backup to ``backups/pre-migration-<timestamp>.zip`` before ``hermes claw migrate``
|
|
(same dir as update backups so listings/``hermes import`` find it); ``None`` if nothing was
|
|
found or the write failed. Never raises."""
|
|
return _create_prefixed_full_backup(
|
|
hermes_home, _PRE_MIGRATION_PREFIX, max(keep, 0), "pre-migration", "pre-migration backup")
|
|
|
|
|
|
# ---- BEGIN PLUGIN-COMPAT (revert-scheduled; see COMPAT_MANIFEST.md) ----
|
|
# Names external plugins imported from this module before the Sep 2026 decomposition.
|
|
# Internal code MUST NOT use these (scripts/check_compat_pointers.py fails CI if it does).
|
|
# The whole block is removed by reverting the commit that added it.
|
|
|
|
def copy_db_and_verify(src: Path, dst: Path) -> bool:
|
|
"""Like :func:`_safe_copy_db` but verifies the destination after copy.
|
|
|
|
Returns True only when the copy succeeded AND the destination is valid
|
|
SQLite (header + integrity check). Verification honours the default
|
|
size ceiling — a multi-GB destination gets the header + schema probe
|
|
rather than a full ``PRAGMA integrity_check`` that would page through
|
|
the whole file.
|
|
"""
|
|
if not _safe_copy_db(src, dst):
|
|
return False
|
|
integrity = verify_sqlite_integrity(dst, run_pragma=True)
|
|
if not integrity.get("valid"):
|
|
try:
|
|
dst.unlink(missing_ok=True)
|
|
except OSError:
|
|
pass
|
|
logger.warning("Backup of %s failed integrity verification: %s", src, integrity.get("message"))
|
|
return False
|
|
return True
|
|
# ---- END PLUGIN-COMPAT ----
|