Files
hermes-agent/hermes_cli/backup.py

1821 lines
85 KiB
Python

"""Backup and import commands for hermes CLI."""
import json
import logging
import os
import shutil
import sqlite3
import stat
import sys
import tempfile
import threading
import time
import zipfile
import zlib
from contextlib import closing, contextmanager, suppress
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
from hermes_constants import (
LOCAL_RUNTIME_ROOT_DIRS, _get_platform_default_hermes_home, get_default_hermes_root, get_hermes_home,
display_hermes_home,
)
from hermes_state_dbfile import RETIRED_GENERATION_DIR_SUFFIX
from utils import (
_preserve_file_mode, _preserve_file_owner, _restore_file_mode, _restore_file_owner, atomic_replace,
default_new_file_mode,
)
from hermes_cli.sizefmt import format_bytes as _format_size
logger = logging.getLogger(__name__)
# --- Exclusion rules ---
# Where ``hermes backup --quick`` / ``/snapshot`` / the pre-update safety net write state
# snapshots (see ``create_quick_snapshot``); defined here because the exclusion set needs it.
_QUICK_SNAPSHOTS_DIR = "state-snapshots"
def _snapshot_recovery_hint() -> str:
"""How to restore a state snapshot. There is no `hermes snapshot` subcommand — only the /snapshot
slash command inside a `hermes` session (hermes_cli/commands.py)."""
return ("To restore a newer snapshot, start `hermes` in a terminal and run `/snapshot list`, then "
"`/snapshot restore <id>` (CLI only).")
# Directory names to skip (matched against each path component). ``hermes-agent`` only matches at
# the root (``_should_exclude``) so skill dirs like ``skills/.../hermes-agent/`` survive. The
# dependency/cache entries matter: one plugin venv or pip/uv cache under HERMES_HOME walked
# file-by-file balloons a backup to hundreds of thousands of entries ("backup stuck for days").
# Mostly mirrors ``agent.skill_utils.EXCLUDED_SKILL_DIRS``; ``.cache`` is backup-only. ``.archive``
# is deliberately NOT excluded: the curator's ``skills/.archive/`` holds restorable user skills.
_EXCLUDED_DIRS = {
"hermes-agent", # the codebase repo — re-clone instead
"__pycache__", # bytecode caches — regenerated on import
".git", # nested git dirs (profiles shouldn't have these, but safety)
"node_modules", # js deps — reinstalled on demand
"backups", # prior auto-backups — don't nest backups exponentially
_QUICK_SNAPSHOTS_DIR, # each holds a full state.db copy — same reason as ``backups``
"checkpoints", # session-hash-keyed trajectory caches — regenerated, don't port
# Live CDP browser profiles: Chromium holds their SQLite DBs exclusively locked while running
# and sqlite3.backup() retries SQLITE_BUSY forever, hanging the backup. Regenerable anyway.
"browser-profiles",
# Real-profile browsing snapshot (browser.use_real_profile): copies of the user's Cookies /
# Login Data — a credential store that must NOT enter an archive. Regenerated on next launch.
"browser-profile",
# Python dependency trees (plugin / MCP-server venvs) — regenerated by reinstalling.
".venv", "venv", "site-packages",
# Tool / build caches — all regeneratable.
".cache", ".tox", ".nox", ".pytest_cache", ".mypy_cache", ".ruff_cache",
}
# Hermes-managed runtime downloads (see ``LOCAL_RUNTIME_ROOT_DIRS``). Matched ONLY at the root of
# HERMES_HOME and at ``profiles/<name>/`` — a deeper dir of the same name (a skill's ``models/``)
# is user data.
_EXCLUDED_ROOT_DIRS = LOCAL_RUNTIME_ROOT_DIRS
# Browser Use CLI profile dir (browser.backend: browser-use): Chromium user-data with Login Data
# / Cookies. Root-scoped like models/ — a skill's own browser_profiles/ is user data. Backup-only:
# do not fold into LOCAL_RUNTIME_ROOT_DIRS (clone-all identity contract).
_EXCLUDED_BACKUP_ROOT_DIRS = frozenset({"browser_profiles"})
# ``cache/`` at those same roots mixes regenerable state (model/plugin catalogs, stamps, browser
# profiles with locked SQLite, tool-output spill) with durable artifacts nothing can rebuild: media
# the gateway delivered to or received from the user (``gateway.platforms.base``'s media-delivery
# subdirs) and the grounded-citations evidence ledger. Only these subdirs are archived.
_KEPT_CACHE_SUBDIRS = {"images", "audio", "videos", "documents", "screenshots", "citations"}
def _in_excluded_root_dir(rel_path: Path) -> bool:
"""True when *rel_path* is inside a regenerable tree at a profile-home root."""
parts = rel_path.parts
if len(parts) >= 3 and parts[0] == "profiles":
parts = parts[2:]
if not parts:
return False
if parts[0] in _EXCLUDED_ROOT_DIRS or parts[0] in _EXCLUDED_BACKUP_ROOT_DIRS:
return True
return parts[0] == "cache" and len(parts) >= 2 and parts[1] not in _KEPT_CACHE_SUBDIRS
# SQLite sidecars are excluded because ``*.db`` is snapshotted via ``sqlite3.backup()``:
# shipping the live WAL/SHM/journal alongside would pair a fresh snapshot with stale sidecar
# state and produce a torn restore on next open. They are regenerated on first connection.
_SQLITE_SIDECAR_SUFFIXES = (".db-wal", ".db-shm", ".db-journal")
_EXCLUDED_SUFFIXES = (".pyc", ".pyo", *_SQLITE_SIDECAR_SUFFIXES)
# File names to skip (runtime state that's meaningless on another machine)
_EXCLUDED_NAMES = {".backup.lock", "gateway.pid", "cron.pid"}
# The desktop updater's pre-flight drops ``state.db.pre-update-emergency-<ts>.bak`` at the root
# — a backup artifact like ``backups/``. Prefix-matched because the name carries a timestamp;
# a plain ``.bak`` suffix rule would drop user files.
# Retired-WAL capture dirs (``<name>.retired-wal-<ts>-<pid>/``) are excluded whole: a
# ``sqlite3.backup()`` snapshot of the live db paired with the captured ``-wal`` is exactly the
# torn-restore hazard the sidecar exclusion below exists to prevent, and the capture is an
# operator-recovery artifact that must move as a unit (manifest + image + WAL), never partially.
_EXCLUDED_PREFIXES = (
"state.db.pre-update-emergency-",
f"state.db{RETIRED_GENERATION_DIR_SUFFIX}",
)
# Files ``hermes import`` must never overwrite, matched by basename so root and named profiles are
# both covered. They hold runtime state namespaced to the SOURCE machine: ``gateway_state.json``
# drives the container-boot reconciler (a foreign value leaves the gateway stuck "starting" and
# disconnected from the Nous portal); PID/lock/registry files reference source PIDs. Mirrors
# ``container_boot._STALE_RUNTIME_FILES``; import filters too because older backups predate the
# backup-side exclusions.
_IMPORT_SKIP_NAMES = {"gateway_state.json", "gateway.pid", "cron.pid", "gateway.lock", "processes.json"}
try: # zipfile already imports lzma (free); it is absent only from Pythons built without liblzma
import lzma
_LZMA_ERRORS: tuple[type[BaseException], ...] = (lzma.LZMAError,)
except ImportError: # pragma: no cover
_LZMA_ERRORS = ()
# What reading a member's data raises when the archive itself is bad (a bzip2 bad stream and a
# media read error are OSError, caught alongside): bad deflate stream, bad CRC, truncated stream.
_ZIP_MEMBER_READ_ERRORS: tuple[type[BaseException], ...] = (
zipfile.BadZipFile, zlib.error, EOFError, *_LZMA_ERRORS)
# zipfile.open() drops Unix mode bits on extract; restore tightens these to 0600.
# vault.key / vault.json.enc: the local credential vault (agent/vault_store.py)
# IS included in backups (user-entered secrets, not regenerable — unlike the
# excluded browser-profile/ snapshot) but must come back owner-only.
_SECRET_FILE_NAMES = {".env", "auth.json", "state.db", "vault.key", "vault.json.enc"}
# Reserved archive subtree for memory-provider state OUTSIDE HERMES_HOME (e.g. ~/.honcho, via
# MemoryProvider.backup_paths()), stored and restored relative to the user's home; paths not
# under home are skipped.
_EXTERNAL_PREFIX = "_external/"
class BackupInProgressError(RuntimeError):
"""Raised when another process already owns the Hermes backup slot."""
class _SQLiteSnapshotError(RuntimeError):
pass
class _SQLiteBackupTimeout(RuntimeError):
"""Raised when a SQLite snapshot remains busy past its deadline."""
@contextmanager
def _backup_operation_lock(hermes_home: Path, timeout_seconds: float = 0.25):
"""Acquire one cross-process backup slot for full and quick snapshots."""
lock_path = hermes_home / ".backup.lock"
lock_path.parent.mkdir(parents=True, exist_ok=True)
handle = lock_path.open("a+b")
acquired = False
deadline = time.monotonic() + max(0.0, timeout_seconds)
try:
if os.name == "nt":
import msvcrt
if lock_path.stat().st_size == 0:
handle.write(b" ")
handle.flush()
def _lock_op(flag: int) -> None:
handle.seek(0)
msvcrt.locking(handle.fileno(), flag, 1)
lock_flag, unlock_flag = msvcrt.LK_NBLCK, msvcrt.LK_UNLCK
else:
import fcntl
def _lock_op(flag: int) -> None:
fcntl.flock(handle.fileno(), flag)
lock_flag, unlock_flag = fcntl.LOCK_EX | fcntl.LOCK_NB, fcntl.LOCK_UN
while not acquired:
try:
_lock_op(lock_flag)
acquired = True
except OSError:
if time.monotonic() >= deadline:
raise BackupInProgressError("another Hermes backup is already running")
time.sleep(0.05)
yield
finally:
if acquired:
with suppress(OSError):
_lock_op(unlock_flag)
handle.close()
@contextmanager
def _atomic_output_path(final_path: Path):
"""Yield a hidden sibling path and publish it only after a clean close."""
partial_path = final_path.with_name(f".{final_path.name}.{os.getpid()}-{threading.get_ident()}.partial")
partial_path.unlink(missing_ok=True)
try:
yield partial_path
os.replace(partial_path, final_path)
except BaseException:
partial_path.unlink(missing_ok=True)
raise
def _is_within(path: Path, root: Path) -> bool:
"""True when *path* resolves inside the already-resolved *root* (traversal / symlink guard)."""
return path.resolve().is_relative_to(root)
def _collect_memory_provider_external_paths() -> List[Path]:
"""Existing paths the active memory provider declares via ``backup_paths()``; ``[]`` on any
provider failure (backup must never fail because of a flaky plugin)."""
try:
from plugins.memory import _get_active_memory_provider, load_memory_provider
active = _get_active_memory_provider()
provider = load_memory_provider(active) if active else None
except Exception:
return []
if provider is None:
return []
try:
declared = provider.backup_paths() or []
except Exception as exc:
logger.warning("backup_paths() failed for memory provider %r: %s", active, exc)
return []
out: Dict[Path, Path] = {} # resolved -> first declared spelling
for raw in declared:
try:
p = Path(raw).expanduser()
except Exception:
continue
if not p.exists():
continue
try:
resolved = p.resolve()
except (OSError, ValueError):
continue
out.setdefault(resolved, p)
return list(out.values())
def _iter_external_files(base: Path) -> List[Path]:
"""Regular files under *base* (a file or a directory), skipping symlinks, caches, and pyc."""
if base.is_file() and not base.is_symlink():
return [base]
if not base.is_dir():
return []
files: List[Path] = []
for dirpath, dirnames, filenames in os.walk(base, followlinks=False):
dirnames[:] = [d for d in dirnames if d not in _EXCLUDED_DIRS]
files.extend(fp for fp in (Path(dirpath) / f for f in filenames)
if not (_is_non_regular_path(fp) or fp.name in _EXCLUDED_NAMES
or fp.name.endswith(_EXCLUDED_SUFFIXES)))
return files
def _is_non_regular_path(path: Path) -> bool:
"""True for symlinks, sockets, devices, and other non-regular filesystem entries.
A failed ``lstat`` is not treated as an exclusion: the archive writer must see the path and
report the read failure instead of silently claiming a complete backup.
"""
try:
return not stat.S_ISREG(path.lstat().st_mode)
except OSError:
return False
def _should_exclude(rel_path: Path) -> bool:
"""Return True if *rel_path* (relative to hermes root) should be skipped."""
parts = rel_path.parts
if _in_excluded_root_dir(rel_path):
return True
# ``hermes-agent`` only matches at the root level; nested same-named dirs are preserved.
if any(p in _EXCLUDED_DIRS and (p != "hermes-agent" or p == parts[0]) for p in parts):
return True
name = rel_path.name
return name in _EXCLUDED_NAMES or name.startswith(_EXCLUDED_PREFIXES) or name.endswith(_EXCLUDED_SUFFIXES)
def _iter_backup_files(hermes_root: Path, out_path: Path, skipped_dirs: Optional[set] = None):
"""Yield ``(abs_path, rel_path)`` for every file a full backup should hold.
The one owner of the walk policy (directory pruning so os.walk never descends a multi-GB
excluded tree, the root-only ``hermes-agent`` carve-out, root runtime trees, per-file rules),
shared by ``hermes backup`` and the pre-update / pre-migration path so they can never drift.
"""
for dirpath, dirnames, filenames in os.walk(hermes_root, followlinks=False):
rel_dir = Path(dirpath).relative_to(hermes_root)
is_root = rel_dir == Path(".")
kept = [
d for d in dirnames
if (d not in _EXCLUDED_DIRS or (d == "hermes-agent" and not is_root))
and not _in_excluded_root_dir(rel_dir / d)]
if skipped_dirs is not None:
skipped_dirs.update(str(rel_dir / d) for d in set(dirnames) - set(kept))
dirnames[:] = kept
for fname in filenames:
rel = rel_dir / fname
fpath = hermes_root / rel
# zipfile.write() follows file symlinks, so skip links before any archive write can
# copy data from outside HERMES_HOME; never archive the output zip into itself.
if _should_exclude(rel) or _is_non_regular_path(fpath):
continue
with suppress(OSError, ValueError):
if fpath.resolve() == out_path.resolve():
continue
yield fpath, rel
# --- SQLite safe copy ---
def _close_quietly(conn: Optional[sqlite3.Connection]) -> None:
if conn is not None:
with suppress(Exception):
conn.close()
def _query_ro_sqlite(path: Path, fn):
"""Run ``fn(conn)`` on a read-only connection to *path*; return ``(value, None)`` or ``(None, exc)``."""
conn = None
try:
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True, timeout=1.0)
return fn(conn), None
except Exception as exc:
return None, exc
finally:
_close_quietly(conn)
def _safe_copy_db(src: Path, dst: Path, *, timeout_seconds: float = 10.0) -> bool:
"""Copy a SQLite database with the backup() API (WAL-safe consistent snapshot).
Fails closed when no consistent snapshot can be made: copying only the main file loses WAL data.
"""
conn = backup_conn = None
try:
# sqlite3.connect() creates a missing destination with the process
# umask, which is commonly 0022 (0644). Snapshot databases contain
# session and tool state, so create the inode owner-only before SQLite
# writes its first byte. O_NOFOLLOW also refuses a planted symlink on
# platforms that support it. Tighten an existing internal staging
# file as well (NamedTemporaryFile callers already create it 0600).
if os.name != "nt":
open_flags = os.O_WRONLY | os.O_CREAT
if hasattr(os, "O_NOFOLLOW"):
open_flags |= os.O_NOFOLLOW
secure_fd = os.open(dst, open_flags, 0o600)
try:
os.fchmod(secure_fd, 0o600)
finally:
os.close(secure_fd)
# timeout=0.0 disables sqlite3's implicit busy wait so the progress callback owns the
# full locked-source deadline instead of adding the default timeout before each callback.
conn = sqlite3.connect(f"file:{src}?mode=ro", uri=True, timeout=0.0)
backup_conn = sqlite3.connect(str(dst))
busy_deadline = time.monotonic() + max(0.0, timeout_seconds)
def _check_backup_progress(status: int, _remaining: int, _total: int) -> None:
nonlocal busy_deadline
now = time.monotonic()
if status in (sqlite3.SQLITE_BUSY, sqlite3.SQLITE_LOCKED):
if now >= busy_deadline:
raise _SQLiteBackupTimeout(f"database remained locked for {timeout_seconds:g} seconds")
else:
busy_deadline = now + max(0.0, timeout_seconds)
conn.backup(backup_conn, pages=256, progress=_check_backup_progress, sleep=0.1)
return True
except Exception as exc:
logger.warning("SQLite safe copy failed for %s: %s", src, exc)
# Windows won't remove the partial destination while SQLite still has it open.
_close_quietly(backup_conn)
backup_conn = None
with suppress(OSError):
dst.unlink(missing_ok=True)
return False
finally:
_close_quietly(backup_conn)
_close_quietly(conn)
def is_zeroed_sqlite_file(path: Path, *, probe_bytes: int = 100, force: bool = False) -> bool:
"""True when *path* looks like the #68474 zeroed-state.db signature.
Only regular files qualify: probing a FIFO/device/socket could block indefinitely.
Signature: no ``SQLite format 3`` header and no data — either empty (size 0, the total-loss case,
#97568) or first *probe_bytes* all NUL. Used at SessionDB open and for snapshot diagnostics so a silent
all-zero file becomes a guided recovery instead of a generic failure.
"""
try:
if not path.is_file():
return False
except OSError:
return False
from hermes_cli.sqlite_safe_read import has_live_connection, read_header_bytes_preopen
if not force and has_live_connection(path):
return False
head = read_header_bytes_preopen(path, length=max(16, probe_bytes), force=force)
# Empty or all-NUL header => zeroed; a real header (or unreadable) => not.
return head is not None and not head.startswith(b"SQLite format 3") and not any(head)
# --- SQLite integrity verification ---
_SQLITE_HEADER = b"SQLite format 3\0"
# Above this size ``PRAGMA integrity_check`` (walks every b-tree page — minutes of pegged CPU on a
# 30 GB state.db, reading as a hung ``hermes update``) is replaced by the O(1) header+schema probe.
# Default ceiling above which ``PRAGMA integrity_check`` is skipped in favour of the (O(1)) header +
# structural probe. Sessions databases in the tens of GB are normal for heavy users, so the size-unbounded
# check is never an acceptable default on the update path. See #70553.
DEFAULT_INTEGRITY_CHECK_MAX_BYTES = 2 << 30 # 2 GiB
def verify_sqlite_integrity(
path: Path, *, check_header: bool = True, run_pragma: bool = True,
max_bytes: int = DEFAULT_INTEGRITY_CHECK_MAX_BYTES) -> dict:
"""Verify a SQLite database: existence + minimum size, header magic, then a read-only
``PRAGMA integrity_check`` (or a cheap structural probe above ``max_bytes``)."""
def _done(message: str, valid: bool = False, size: Optional[int] = None) -> dict:
return {"valid": valid, "message": message, "size": size}
try:
st = path.stat()
except FileNotFoundError:
return _done(f"not found: {path}")
except OSError as exc:
return _done(f"cannot stat: {exc}")
size = st.st_size
if size < 100: # SQLite minimum viable size (header + 1 page)
return _done(f"too small ({size} bytes) to be a valid SQLite database", size=size)
if check_header:
# Refused when a live connection exists (close() would cancel this process's POSIX locks
# — see sqlite_safe_read); verification targets offline snapshots/backup artifacts anyway.
from hermes_cli.sqlite_safe_read import read_header_bytes_preopen
head = read_header_bytes_preopen(path, length=len(_SQLITE_HEADER))
if head is None:
return _done("cannot read header", size=size)
if head != _SQLITE_HEADER:
return _done(f"missing SQLite header magic (got {head[:16].hex()!r})", size=size)
if max_bytes > 0 and size > max_bytes:
# O(1) probe: the header check caught the zeroed signature; reading sqlite_master + page
# geometry catches malformed-schema and truncated-header-page classes without a data walk.
_, exc = _query_ro_sqlite(path, lambda c: (
c.execute("PRAGMA schema_version").fetchone(),
c.execute("SELECT count(*) FROM sqlite_master").fetchone()))
if exc is not None:
kind = "failed" if isinstance(exc, sqlite3.DatabaseError) else "error"
return _done(f"schema probe {kind}: {exc}", size=size)
return _done(
f"size {size:,} bytes exceeds max_bytes {max_bytes:,}; "
"skipped PRAGMA integrity_check (header + schema probe passed)",
valid=True, size=size)
if run_pragma:
rows, exc = _query_ro_sqlite(
path, lambda c: [str(r[0]) for r in c.execute("PRAGMA integrity_check")])
if exc is not None:
kind = "cannot open database" if isinstance(exc, sqlite3.DatabaseError) else "integrity check error"
return _done(f"{kind}: {exc}", size=size)
if rows == ["ok"]:
return _done("integrity check passed", valid=True, size=size)
return _done(f"integrity check failed: {'; '.join(rows[:5])}", size=size)
return _done("header check passed", valid=True, size=size)
def _foreign_db_holder_pids(db_path: Path) -> Optional[List[int]]:
"""PIDs of OTHER processes holding *db_path* or its WAL/SHM open (Linux ``/proc`` scan).
An already-unlinked ``(deleted)`` sidecar — the #90950 split-brain fingerprint — still
counts as held. None off-Linux or when /proc fails.
"""
if not sys.platform.startswith("linux"):
return None
def _canonical(path: str) -> str:
return os.path.normcase(os.path.abspath(path.removesuffix(" (deleted)")))
def _holds_watched(fds: List[str], fd_dir: str) -> bool:
for fd in fds:
try:
target = os.readlink(f"{fd_dir}/{fd}")
except OSError:
continue
if _canonical(target) in watched:
return True
return False
canonical_db = _canonical(os.fspath(db_path))
watched = {canonical_db, canonical_db + "-wal", canonical_db + "-shm"}
pids: List[int] = []
try:
own_pid = os.getpid()
for pid_str in os.listdir("/proc"):
if not pid_str.isdigit() or int(pid_str) == own_pid:
continue
fd_dir = f"/proc/{pid_str}/fd"
try:
fds = os.listdir(fd_dir)
except OSError:
continue
if _holds_watched(fds, fd_dir):
pids.append(int(pid_str))
except OSError:
return None
return pids
def _safe_restore_db(src: Path, dst: Path) -> bool:
"""Restore snapshot *src* into live *dst* through the backup() API; unlink+move fallback.
Writing pages into the live file preserves its inode and WAL state, so other holders (gateway,
dashboard, another CLI) see the restored data instead of stale pages from a replaced inode.
The fallback runs ONLY when no other process or in-process connection holds the file
(replacing the inode under a live holder is the #90950 split-brain); otherwise it fails closed
(``False``) and the caller reports the file as skipped.
"""
try:
dst_conn = sqlite3.connect(str(dst))
# Checkpoint first so the backup starts clean rather than writing on top of a deep WAL.
with suppress(Exception):
dst_conn.execute("PRAGMA wal_checkpoint(TRUNCATE)")
with closing(sqlite3.connect(f"file:{src}?mode=ro", uri=True)) as src_conn:
src_conn.backup(dst_conn)
dst_conn.close()
with suppress(Exception):
dst.chmod(src.stat().st_mode)
return True
except Exception as exc:
logger.warning("SQLite safe restore failed for %s -> %s: %s", src, dst, exc)
return _unlink_move_restore_db(src, dst)
def _unlink_move_restore_db(src: Path, dst: Path) -> bool:
"""Fallback restore: unlink+move. Only safe when no process holds the DB open.
Replacing the inode under a live holder is the #90950 corruption class (the holder keeps
writing through a deleted-inode fd and loses its WAL index), so fail closed. The foreign-pid
scan excludes THIS process, so ``offline_file_access`` also fails CLOSED on any live
in-process connection to *dst* and holds the connection-lifecycle lock across the swap.
"""
from hermes_cli.sqlite_safe_read import LiveConnectionError, offline_file_access
try:
holders = _foreign_db_holder_pids(dst)
if holders:
logger.error("Refusing unlink+move restore of %s: process(es) %s still "
"hold the database or its WAL open. Stop them and retry.", dst, holders)
return False
with offline_file_access(dst, what="unlink+move restore of"):
tmp = dst.parent / f".{dst.name}.snap_restore"
shutil.copy2(src, tmp)
dst.unlink(missing_ok=True)
# The snapshot owns no WAL, so any -wal/-shm here belongs to the DB just unlinked (a
# killed gateway leaves them — exactly when a restore runs); SQLite would replay that
# foreign WAL over the restored file: "malformed" or resurrected post-snapshot rows.
for _sidecar_suffix in ("-wal", "-shm", "-journal"):
dst.with_name(dst.name + _sidecar_suffix).unlink(missing_ok=True)
shutil.move(str(tmp), str(dst))
return True
except LiveConnectionError as exc2:
logger.error("Refusing unlink+move restore of %s: %s Close the in-process "
"database handles (or restart Hermes) and retry.", dst, exc2)
return False
except Exception as exc2:
logger.error("Fallback restore also failed for %s -> %s: %s", src, dst, exc2)
return False
def _zip_sqlite_snapshot(zf: zipfile.ZipFile, abs_path: Path, rel_path: Path, out_path: Path) -> Optional[int]:
"""Add a WAL-safe snapshot of *abs_path* to *zf*; return its byte size, or None on failure.
Staged beside the output zip: /tmp may be a small tmpfs that cannot hold large databases.
"""
with tempfile.NamedTemporaryFile(suffix=".db", delete=False, dir=str(out_path.parent)) as tmp:
tmp_db = Path(tmp.name)
try:
if not _safe_copy_db(abs_path, tmp_db):
return None
zf.write(tmp_db, arcname=str(rel_path))
return tmp_db.stat().st_size
finally:
tmp_db.unlink(missing_ok=True)
def _write_zip_entries(
zf: zipfile.ZipFile, files_to_add: List[Tuple[Path, Path]], out_path: Path,
*, on_db_failure, on_error, on_progress, track_bytes: bool) -> int:
"""Add every ``(abs_path, rel_path)`` to *zf*, WAL-safe for ``*.db``; return bytes archived.
``on_db_failure(rel_path)`` runs when a SQLite snapshot fails (may raise to abort);
``on_error(rel_path, exc)`` records a read failure; ``on_progress(i)`` fires every 500 files;
``track_bytes`` stats plain files for the size total.
"""
total_bytes = 0
for i, (abs_path, rel_path) in enumerate(files_to_add, 1):
try:
if abs_path.suffix == ".db":
size = _zip_sqlite_snapshot(zf, abs_path, rel_path, out_path)
if size is None:
on_db_failure(rel_path)
continue
total_bytes += size
else:
zf.write(abs_path, arcname=str(rel_path))
if track_bytes:
total_bytes += abs_path.stat().st_size
except (PermissionError, OSError, ValueError) as exc:
on_error(rel_path, exc)
continue
if i % 500 == 0:
on_progress(i)
return total_bytes
def _print_capped(header: str, lines: List[str], indent: str) -> None:
"""Print *header*, then at most 10 of *lines* (each prefixed by *indent*) and a "... and N more" tail."""
print(header)
for line in lines[:10]:
print(f"{indent}{line}")
if len(lines) > 10:
print(f"{indent}... and {len(lines) - 10} more")
# --- Backup ---
_RUN_BACKUP_PREFIX = "hermes-backup-"
def _resolve_backup_output_path(output: Optional[str]) -> Path:
"""Turn ``--output`` (file, directory, or None) into a ``.zip`` path whose parent exists;
an unwritable path exits with a one-line error, not a traceback."""
out_path = None
default_name = f"{_RUN_BACKUP_PREFIX}{datetime.now().strftime('%Y-%m-%d-%H%M%S')}.zip"
try:
if output:
out_path = Path(output).expanduser().resolve()
if out_path.is_dir():
out_path = out_path / default_name
else:
out_path = Path.home() / default_name
if out_path.suffix.lower() != ".zip":
out_path = out_path.with_suffix(out_path.suffix + ".zip")
out_path.parent.mkdir(parents=True, exist_ok=True)
except OSError as exc:
print(f"Error: cannot write backup to {output or out_path}: {exc}")
raise SystemExit(1) from exc
return out_path
def _collect_external_entries() -> tuple[list[tuple[Path, str]], list[str]]:
"""``([(abs_path, arcname)], [skipped])`` for the memory provider's external state, arc-named
``_external/<home-relative>``; paths outside home are skipped (security + portability)."""
home_dir = Path.home().resolve()
external_to_add: list[tuple[Path, str]] = []
skipped_external: list[str] = []
for base in _collect_memory_provider_external_paths():
try:
base.resolve().relative_to(home_dir)
except (ValueError, OSError):
skipped_external.append(str(base))
continue
for fpath in _iter_external_files(base):
with suppress(ValueError, OSError):
rel_to_home = fpath.resolve().relative_to(home_dir)
external_to_add.append((fpath, _EXTERNAL_PREFIX + rel_to_home.as_posix()))
return external_to_add, skipped_external
def run_backup(args) -> bool:
"""Create a zip backup of the Hermes home directory.
True when every selected file landed in the archive (or there was nothing to back up); False
when the zip was written but is incomplete — it is kept so the rest can still be restored, and
the caller turns False into exit status 1 so a cron/systemd timer never publishes a "successful"
archive that is missing state.db. Hard failures keep raising ``SystemExit``.
"""
hermes_root = get_default_hermes_root()
if not hermes_root.is_dir():
print(f"Error: Hermes home directory not found at {hermes_root}")
sys.exit(1)
try:
with _backup_operation_lock(hermes_root):
return _run_backup_locked(args, hermes_root)
except BackupInProgressError as exc:
print(f"Error: {exc}")
raise SystemExit(2) from exc
def _run_backup_locked(args, hermes_root: Path) -> bool:
"""Write a full backup while the cross-process backup slot is held."""
out_path = _resolve_backup_output_path(args.output)
scan_started = time.monotonic()
logger.info("backup phase=scan status=started")
print(f"Scanning {display_hermes_home()} ...")
skipped_dirs: set = set()
files_to_add: list[tuple[Path, Path]] = list(_iter_backup_files(hermes_root, out_path, skipped_dirs))
external_to_add, skipped_external = _collect_external_entries()
if not files_to_add and not external_to_add:
logger.info("backup phase=scan status=empty duration_ms=%.1f", (time.monotonic() - scan_started) * 1000)
print("No files to back up.")
return True
file_count = len(files_to_add) + len(external_to_add)
logger.info("backup phase=scan status=complete duration_ms=%.1f files=%d",
(time.monotonic() - scan_started) * 1000, file_count)
logger.info("backup phase=archive status=started files=%d", file_count)
print(f"Backing up {file_count} files ...")
errors = []
t0 = time.monotonic()
def _progress(i: int) -> None:
print(f" {i}/{file_count} files ...")
logger.info("backup phase=archive status=progress completed=%d total=%d", i, file_count)
with _atomic_output_path(out_path) as archive_path, zipfile.ZipFile(
archive_path, "w", zipfile.ZIP_DEFLATED, compresslevel=6) as zf:
total_bytes = _write_zip_entries(
zf, files_to_add, out_path, on_progress=_progress, track_bytes=True,
on_db_failure=lambda rel: errors.append(f"{rel}: SQLite safe copy failed"),
on_error=lambda rel, exc: errors.append(f"{rel}: {exc}"))
# External memory-provider state never includes ``.db`` files in practice, so a
# straight zf.write is fine.
for abs_path, arcname in external_to_add:
try:
zf.write(abs_path, arcname=arcname)
total_bytes += abs_path.stat().st_size
except (PermissionError, OSError, ValueError) as exc:
errors.append(f"{arcname}: {exc}")
elapsed = time.monotonic() - t0
zip_size = out_path.stat().st_size
logger.info("backup phase=archive status=complete duration_ms=%.1f files=%d errors=%d bytes=%d",
elapsed * 1000, file_count, len(errors), zip_size)
print(f"\nBackup {'incomplete' if errors else 'complete'}: {out_path}\n"
f" Files: {file_count}\n"
f" Original: {_format_size(total_bytes)}\n"
f" Compressed: {_format_size(zip_size)}\n"
f" Time: {elapsed:.1f}s")
if external_to_add:
print(f"\n Included {len(external_to_add)} memory-provider file(s) stored outside {display_hermes_home()}.")
if skipped_external:
print(f"\n Skipped {len(skipped_external)} memory-provider path(s) outside your home directory "
"(not portable):\n" + "\n".join(f" {p}" for p in sorted(skipped_external)[:10]))
if skipped_dirs:
print("\n Excluded directories:\n" + "\n".join(f" {d}/" for d in sorted(skipped_dirs)))
if errors:
_print_capped(f"\n Archive kept, but {len(errors)} file(s) could not be added:", errors, " ")
else:
print(f"\nRestore with: hermes import {out_path.name}")
# Prune only after a complete archive: a timer hitting the same unreadable file every run must
# not rotate the last good backups out in favour of incomplete ones.
keep = getattr(args, "keep", 0) # 0 / absent: never prune (non-CLI callers)
if keep and not errors and out_path.name.startswith(_RUN_BACKUP_PREFIX):
pruned = _prune_prefixed_zips(out_path.parent, _RUN_BACKUP_PREFIX, keep, "backup")
if pruned:
print(f" Pruned {pruned} older {_RUN_BACKUP_PREFIX}*.zip (keeping {keep}).")
return not errors
# --- Import ---
def _validate_backup_zip(zf: zipfile.ZipFile) -> tuple[bool, str]:
"""Check that a zip looks like a Hermes backup."""
names = zf.namelist()
if not names:
return False, "zip archive is empty"
# Telltale files a hermes home has — at the root or one level deep (zipped directory).
if not any(Path(n).name in {"config.yaml", ".env", "state.db"} for n in names):
return False, "zip does not appear to be a Hermes backup (no config.yaml, .env, or state databases found)"
return True, ""
def _find_corrupt_members(zf: zipfile.ZipFile, members: List[str]) -> List[str]:
"""Return ``"<member>: <error>"`` for every member whose data does not decompress or
fails its CRC, streaming each one in 1 MiB chunks so a multi-GB ``state.db`` is never
held in memory.
``is_zipfile()``/``namelist()`` only read the central directory, so an archive with a
rotten member passes them and the damage surfaces as ``zlib.error``/``BadZipFile`` in
the middle of the restore, after earlier members already replaced the user's files
(#121258). Not ``zf.testzip()``: it lets ``zlib.error`` escape and names at most the
first bad member.
"""
bad: list[str] = []
for member in members:
try:
with zf.open(member) as src:
while src.read(1 << 20): # CRC is checked when the stream hits EOF
pass
except (OSError, *_ZIP_MEMBER_READ_ERRORS) as exc:
bad.append(f"{member}: {exc}")
return bad
def _import_skipped(rel: str) -> bool:
"""True for a HERMES_HOME-relative member the import deliberately does not restore: runtime
state (``_IMPORT_SKIP_NAMES``), an archived SQLite WAL/SHM/journal -- a ``.db`` member is
page-restored into the live file, and a sidecar from a different database image installed
beside it would replay a foreign WAL on next open (current backups never ship these, older or
hand-built archives might)."""
return Path(rel).name in _IMPORT_SKIP_NAMES or rel.endswith(_SQLITE_SIDECAR_SUFFIXES)
def _import_member_rel(member: str, prefix: str) -> tuple[str, bool]:
"""Classify an archive member exactly as the restore does: return ``(rel, skipped)``.
``_external/`` members are home-relative and never skipped; every other member is
HERMES_HOME-relative after stripping the archive ``prefix``. Shared by the integrity
pre-flight and ``_import_members`` so the two cannot disagree on what gets restored."""
if member.startswith(_EXTERNAL_PREFIX):
return member[len(_EXTERNAL_PREFIX):], False
rel = member[len(prefix):] if prefix and member.startswith(prefix) else member
return rel, _import_skipped(rel)
def _detect_prefix(zf: zipfile.ZipFile) -> str:
"""Detect if the zip has a common directory prefix wrapping all entries."""
names = [n for n in zf.namelist() if not n.endswith("/")]
first_parts = {Path(n).parts[0] for n in names if len(Path(n).parts) > 1}
if len(first_parts) == 1 and first_parts <= {".hermes", "hermes"}:
return first_parts.pop() + "/"
return ""
def _extract_member_atomically(
zf: zipfile.ZipFile, member: str, target: Path, new_file_mode: Optional[int] = None) -> None:
"""Restore one zip member onto *target* with no truncation window.
``open(target, "wb")`` would truncate the user's file before any replacement bytes exist.
``atomic_replace`` (not bare ``os.replace``) resolves a symlinked target first, so a
dotfiles-linked ``config.yaml`` keeps the link (#16743), and falls back to copy/fsync/unlink
on ``EXDEV``/``EBUSY`` for cross-device and bind-mount installs.
"""
# Mode is None when the target does not exist: the umask-derived create-mode applies.
mode = _preserve_file_mode(target)
owner = _preserve_file_owner(target)
if mode is None:
mode = new_file_mode
else:
# Deliberately NOT a faithful copy: setuid/setgid are dropped. The bytes come from the
# archive, so carrying elevated bits would hand the zip's author the identity an existing
# setuid file runs as — and ``_external/`` publishes members anywhere under ``$HOME``.
mode &= ~(stat.S_ISUID | stat.S_ISGID)
# Truncate the stem: mkstemp adds ~16 chars and a member near NAME_MAX would otherwise fail.
fd, tmp_name = tempfile.mkstemp(dir=str(target.parent), prefix=f".{target.name[:80]}.", suffix=".partial")
try:
with os.fdopen(fd, "wb") as dst:
if mode is not None:
# Apply the mode BEFORE the replace so the target never transits through mkstemp's
# 0600 and the EXDEV/EBUSY ``copystat`` fallback copies the intended bits.
if hasattr(os, "fchmod"): # Unix-only; Windows takes the path-based chmod
os.fchmod(dst.fileno(), mode)
else:
os.chmod(tmp_name, mode)
# Stream: a multi-gigabyte state.db member must not be held in memory in one piece.
with zf.open(member) as src:
shutil.copyfileobj(src, dst)
dst.flush()
os.fsync(dst.fileno())
real_path = Path(atomic_replace(tmp_name, target))
# Owner first, mode second (as ``atomic_yaml_write``): chown drops setuid/setgid and
# ``mode`` no longer carries them, so neither step can re-elevate the restored file.
_restore_file_owner(real_path, owner)
_restore_file_mode(real_path, mode)
except BaseException:
with suppress(OSError):
os.unlink(tmp_name)
raise
def _count_session_rows(path: Path) -> Optional[Tuple[int, int]]:
"""``(sessions, messages)`` in session database *path*; read-only, best effort.
``None`` means "unknown" (missing, not a Hermes session store, unreadable) — never "zero":
acting on an unreadable database would mask the very loss this count exists to surface.
Same contract as :func:`_count_cron_jobs`.
"""
if not path.is_file():
return None
try:
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
except sqlite3.Error:
return None
try:
sessions = conn.execute("SELECT COUNT(*) FROM sessions").fetchone()[0]
messages = conn.execute("SELECT COUNT(*) FROM messages").fetchone()[0]
return int(sessions), int(messages)
except (sqlite3.Error, TypeError, ValueError):
return None
finally:
conn.close()
def _import_db_member(
zf: zipfile.ZipFile, member: str, target: Path, new_file_mode: Optional[int] = None) -> None:
"""Publish a SQLite ``.db`` member onto *target* without replacing its inode.
A rename-publish over a live database is the #65942 / #90950 corruption class: a gateway,
dashboard, or WebUI holding it open keeps serving the unlinked inode and writing sessions no
other process will see, and a sidecar WAL beside the new file describes the old database —
nothing fails, the sessions are simply gone (#100960). Route the member through the same
``_safe_restore_db`` page copy ``/snapshot restore`` uses, so the live inode is preserved and
every open connection converges. Raises ``OSError`` when the database could not be
replaced safely, so the caller reports a skipped file instead of a silent success.
"""
if not target.exists():
# "Missing" is not "unheld": a gateway or dashboard that had the database open when it
# was unlinked still writes the deleted inode (the ``(deleted)`` fingerprint of #90950).
# Publishing a fresh inode here re-creates the same split brain the branch below exists
# to prevent, so refuse and name the holders instead (#110179).
holders = _foreign_db_holder_pids(target)
if holders:
raise OSError(
f"{target.name} was deleted but is still open in PID(s) "
f"{', '.join(str(pid) for pid in sorted(holders))}; publishing a new file would "
"leave them writing an invisible database. Stop those processes and re-run the import."
)
_extract_member_atomically(zf, member, target, new_file_mode)
return
# The database keeps its own mode/ownership: the bytes come from the archive, the file does not.
mode = _preserve_file_mode(target)
owner = _preserve_file_owner(target)
fd, tmp_name = tempfile.mkstemp(dir=str(target.parent), prefix=f".{target.name[:80]}.", suffix=".dbimport")
try:
with os.fdopen(fd, "wb") as dst:
with zf.open(member) as src: # stream: never hold a multi-GB state.db in memory
shutil.copyfileobj(src, dst)
dst.flush()
os.fsync(dst.fileno())
if not _safe_restore_db(Path(tmp_name), target):
raise OSError(
"live-safe restore refused or failed; the existing database was "
"left untouched. Stop the gateway/dashboard processes holding it "
"open and re-run the import."
)
_restore_file_owner(target, owner)
_restore_file_mode(target, mode)
finally:
with suppress(OSError):
os.unlink(tmp_name)
def _confirm_import_overwrite(hermes_root: Path) -> bool:
"""Prompt before importing over an existing installation; True when import may proceed."""
if not any((hermes_root / m).exists() for m in ("config.yaml", ".env")):
return True
print("\nWarning: Target directory already has Hermes configuration.\n"
"Importing will overwrite existing files with backup contents.\n")
try:
answer = input("Continue? [y/N] ").strip().lower()
except (EOFError, KeyboardInterrupt):
print("\nAborted.")
sys.exit(1)
if answer in {"y", "yes"}:
return True
print("Aborted.")
return False
def _import_members(
zf: zipfile.ZipFile, members: List[str], prefix: str, hermes_root: Path, file_count: int
) -> tuple[int, int, list[str], list[str], list[tuple[str, tuple[int, int], tuple[int, int]]]]:
"""Publish every member; return ``(restored, restored_external, errors, skipped_runtime, db_shrunk)``.
``db_shrunk`` holds ``(rel, live_counts, imported_counts)`` for every session database the
import replaced with one holding fewer rows — allowed, but never silent (#100960).
"""
errors: list[str] = []
skipped_runtime: list[str] = []
db_shrunk: list[tuple[str, tuple[int, int], tuple[int, int]]] = []
restored = restored_external = 0
home_dir = Path.home().resolve()
new_file_mode = default_new_file_mode() # once: every member is published via mkstemp (0600)
for member in members:
# ``_external/`` members restore to their home-relative location (~/.honcho/config.json),
# NOT under HERMES_HOME; provider configs commonly hold credentials, so tighten to 0600.
external = member.startswith(_EXTERNAL_PREFIX)
rel, skipped = _import_member_rel(member, prefix)
if skipped:
skipped_runtime.append(rel)
continue
if external:
target = home_dir / rel
root = home_dir
tighten = target.suffix in {".json", ".env", ".conf"} or target.name in _SECRET_FILE_NAMES
else:
target = hermes_root / rel
root = hermes_root.resolve()
tighten = target.name in _SECRET_FILE_NAMES
if not rel:
continue
label = member if external else rel
if not _is_within(target, root):
errors.append(f"{label}: path traversal blocked")
else:
try:
target.parent.mkdir(parents=True, exist_ok=True)
if target.suffix == ".db":
# Count before the write: afterwards the dropped rows are gone.
before = _count_session_rows(target)
_import_db_member(zf, member, target, new_file_mode)
after = _count_session_rows(target)
if before and after and after[1] < before[1]:
db_shrunk.append((rel, before, after))
else:
_extract_member_atomically(zf, member, target, new_file_mode)
if tighten:
try:
os.chmod(target, 0o600)
except OSError:
if not external: # external configs are tightened best-effort only
raise
restored += 1
restored_external += external
except (OSError, *_ZIP_MEMBER_READ_ERRORS) as exc:
# _ZIP_MEMBER_READ_ERRORS: the pre-flight in run_import already refused
# archives that fail to decompress; this keeps a member that rots between the two
# passes (archive on failing media) from aborting the rest of the restore.
errors.append(f"{label}: {exc}")
if restored % 500 == 0:
print(f" {restored}/{file_count} files ...")
return restored, restored_external, errors, skipped_runtime, db_shrunk
def run_import(args) -> Optional[int]:
"""Restore a Hermes backup from a zip file.
Return 1 when the archive is damaged (refused before anything is written) or the restore is
incomplete (some members were not written); None on success or when the overwrite prompt is
declined. A missing, non-zip or invalid archive exits 1 via ``sys.exit``.
"""
zip_path = Path(args.zipfile).expanduser().resolve()
if not zip_path.is_file():
print(f"Error: File not found: {zip_path}")
sys.exit(1)
if not zipfile.is_zipfile(zip_path):
print(f"Error: Not a valid zip file: {zip_path}")
sys.exit(1)
# The restore target is the home the command operates under (the printed "Target:");
# ``get_default_hermes_root()`` would silently retarget a profile restore at the live root.
hermes_root = get_hermes_home()
with zipfile.ZipFile(zip_path, "r") as zf:
ok, reason = _validate_backup_zip(zf)
if not ok:
print(f"Error: {reason}")
sys.exit(1)
prefix = _detect_prefix(zf)
members = [n for n in zf.namelist() if not n.endswith("/")]
file_count = len(members)
print(f"Backup contains {file_count} files\nTarget: {display_hermes_home()}")
if prefix:
print(f"Detected archive prefix: {prefix!r} (will be stripped)")
if not args.force and not _confirm_import_overwrite(hermes_root):
return
# Every member is decompressed once here and once again below: a damaged archive
# must be refused while the home is still untouched, not half-way through the restore.
print("\nChecking archive integrity ...")
# Members the restore skips anyway (gateway.pid, WAL sidecars) cannot block it.
corrupt = _find_corrupt_members(zf, [m for m in members if not _import_member_rel(m, prefix)[1]])
if corrupt:
_print_capped(f"Error: backup archive is damaged ({len(corrupt)} member(s) fail to "
f"decompress or fail their CRC); nothing was restored:", corrupt, " ")
return 1
print(f"\nImporting {file_count} files ...")
hermes_root.mkdir(parents=True, exist_ok=True)
t0 = time.monotonic()
restored, restored_external, errors, skipped_runtime, db_shrunk = _import_members(
zf, members, prefix, hermes_root, file_count)
elapsed = time.monotonic() - t0
print(f"\nImport {'incomplete' if errors else 'complete'}: {restored} files restored in {elapsed:.1f}s\n"
f" Target: {display_hermes_home()}")
if restored_external:
print(f"\n Restored {restored_external} memory-provider file(s) to "
f"their original location(s) outside {display_hermes_home()}.")
if errors:
_print_capped(f"\n Warnings ({len(errors)} files skipped):", errors, " ")
if db_shrunk:
# The backup predates work that is now overwritten — say so (#100960: twelve sessions
# disappeared with nothing logged anywhere).
print("\n ⚠ Session data replaced by older backup contents:")
for rel, before, after in db_shrunk:
print(f" {rel}: {before[0]} session(s) / {before[1]} message(s)"
f" -> {after[0]} / {after[1]}")
print(f" Anything recorded after the backup was taken is not in it. {_snapshot_recovery_hint()}")
if skipped_runtime:
_print_capped(f"\n Preserved {len(skipped_runtime)} runtime state "
f"file(s) (kept this machine's, not the backup's):",
sorted(skipped_runtime), " ")
restored_profiles = _restore_profile_wrappers(hermes_root)
print()
if not (hermes_root / "hermes-agent").is_dir():
print("Note: The hermes-agent codebase was not included in the backup.\n"
" If this is a fresh install, run: hermes update")
if restored_profiles:
print("\nTo re-enable gateway services for profiles:")
for pname in restored_profiles:
print(f" hermes -p {pname} gateway install")
_revive_gateway_after_import(hermes_root)
if errors:
# A refused state.db leaves the old sessions in place; exit 0 would let scripts and
# the dashboard's "done" badge treat that as a full restore.
print(f"Import incomplete: {len(errors)} file(s) were not restored (see Warnings above). "
"Fix the cause and re-run the import.")
return 1
print("Done. Your Hermes configuration has been restored.")
def _restore_profile_wrappers(hermes_root: Path) -> List[str]:
"""Re-create shell wrapper scripts for restored named profiles; return the profile names seen."""
profiles_dir = hermes_root / "profiles"
restored_profiles: list[tuple[str, bool]] = []
if not profiles_dir.is_dir():
return []
try:
from hermes_cli.profiles import (
create_wrapper_script, check_alias_collision, _is_wrapper_dir_in_path, _get_wrapper_dir)
for entry in sorted(profiles_dir.iterdir()):
if not entry.is_dir() or not any((entry / m).exists() for m in ("config.yaml", ".env")):
continue # only profiles with config get wrappers
profile_name = entry.name
collision = check_alias_collision(profile_name)
if collision:
print(f" Skipped alias '{profile_name}': {collision}")
restored_profiles.append(
(profile_name, not collision and create_wrapper_script(profile_name) is not None))
if restored_profiles:
created = [n for n, ok in restored_profiles if ok]
skipped = [n for n, ok in restored_profiles if not ok]
if created:
print(f"\n Profile aliases restored: {', '.join(created)}")
if skipped:
print(f" Profile aliases skipped: {', '.join(skipped)}")
if not _is_wrapper_dir_in_path():
print(f"\n Note: {_get_wrapper_dir()} is not in your PATH.\n"
" Add to your shell config (~/.bashrc or ~/.zshrc):\n"
' export PATH="$HOME/.local/bin:$PATH"')
except ImportError: # hermes_cli.profiles unavailable (fresh install)
if any(profiles_dir.iterdir()):
print("\n Profiles detected but aliases could not be created.\n"
" Run: hermes profile list (after installing hermes)")
return [n for n, _ in restored_profiles]
def _revive_gateway_after_import(hermes_root: Path) -> None:
"""Install/start the gateway service after a restore, best-effort and prompt-free.
Bot tokens and cron jobs are inert without a gateway (a platform-less gateway is supported, so
this is safe for any backup); failures print a manual fallback, never fail the import. Only
revived when the restore landed in the default home or no other install exists: a sandbox or
profile restore must not install a second gateway on the default service name.
"""
native_default = _get_platform_default_hermes_home()
if hermes_root != native_default and any(
(native_default / marker).exists() for marker in ("config.yaml", ".env", "state.db")):
print("\nRestored into a non-default home; leaving the gateway service alone to avoid clashing "
f"with the install at {native_default}.\n"
"To start a gateway for this home, run: hermes gateway install")
return
try:
from hermes_cli.gateway import ensure_gateway_service, _is_service_running
if not _is_service_running():
print()
ensure_gateway_service(context="import")
except Exception:
print("\nStart the gateway to activate cron jobs and messaging:\n hermes gateway install")
# --- Quick state snapshots (used by /snapshot slash command and hermes backup --quick) ---
# Critical state files (relative to HERMES_HOME) for quick snapshots; everything else is
# regeneratable or managed separately (skills, repo, sessions/). Entries may be files OR
# directories (recursive); missing entries are skipped. Pairing data lives in platform JSON blobs
# outside state.db, so it is listed explicitly — ``hermes update`` snapshots this set (#15733).
_QUICK_STATE_FILES = (
"state.db", "config.yaml", ".env", "auth.json", "cron/jobs.json", "cron/executions.db",
"gateway_state.json", "channel_directory.json", "channel_aliases.json", "processes.json",
"gateway/discord_message_recovery.db", # Discord reconnect replay ledger
# Per-profile user stores, destroyed if the update flow replaces the file and the post-update
# schema-init re-creates an empty one (#52889). Skipped when outside HERMES_HOME.
"projects.db", # per-profile project store
"response_store.db", # gateway conversation history / tool payloads
"memory_store.db", # holographic memory facts/entities
"verification_evidence.db", # agent verification audit trail
"kanban.db", # default board (back-compat <root>/kanban.db)
"kanban/boards", # non-default boards (workspaces/ + attachments/ skipped as regenerable)
# Pairing stores (generic + per-platform JSONs outside state.db)
"pairing", # legacy location (gateway/pairing.py)
"platforms/pairing", # new location (gateway/pairing.py)
"feishu_comment_pairing.json", # Feishu comment subscription pairings
)
_QUICK_DEFAULT_KEEP = 20
def _quick_snapshot_root(hermes_home: Optional[Path] = None) -> Path:
home = hermes_home or get_hermes_home()
return home / _QUICK_SNAPSHOTS_DIR
def create_quick_snapshot(
label: Optional[str] = None, hermes_home: Optional[Path] = None, keep: Optional[int] = None,
max_file_size: Optional[int] = None) -> Optional[str]:
"""Create one atomic quick snapshot while holding the shared backup slot."""
home = hermes_home or get_hermes_home()
with _backup_operation_lock(home):
return _create_quick_snapshot_locked(label, home, keep, max_file_size)
def _quick_snapshot_candidates(home: Path):
"""Yield ``(src, rel_posix, in_dir)`` for every regular file a quick snapshot captures; heavy
regenerable per-board subtrees (workspaces, attachments) are skipped."""
for rel in _QUICK_STATE_FILES:
src = home / rel
if src.is_dir():
for sub in filter(Path.is_file, src.rglob("*")):
sub_rel = sub.relative_to(home).as_posix()
if "/workspaces/" in f"/{sub_rel}/" or "/attachments/" in f"/{sub_rel}/":
continue
yield sub, sub_rel, True
elif src.is_file():
yield src, rel, False
def _copy_quick_snapshot_files(
home: Path, staging_dir: Path, max_file_size: Optional[int]
) -> tuple[Dict[str, int], list[str], list[str]]:
"""Copy every quick-snapshot candidate into *staging_dir*.
Returns ``(manifest {rel: size}, failed_dbs, oversized_skipped)``. The last two are snapshot
incompleteness (#68805): the caller must suppress pruning so the older snapshot that may hold
the only recoverable DB survives.
"""
manifest: Dict[str, int] = {}
failed_dbs: list[str] = []
oversized_skipped: list[str] = []
for src, rel, in_dir in _quick_snapshot_candidates(home):
if max_file_size is not None:
try:
size = src.stat().st_size
except OSError:
size = None
if size is not None and size > max_file_size:
print(f" ⚠ Snapshot: skipping {rel} "
f"({_format_size(size)} exceeds {_format_size(max_file_size)} limit)")
logger.warning("Quick snapshot skipped %s: %d bytes exceeds %d byte limit", rel, size, max_file_size)
if src.suffix == ".db":
oversized_skipped.append(rel)
continue
dst = staging_dir / rel
dst.parent.mkdir(parents=True, exist_ok=True)
try:
# SQLite DBs go through the WAL-safe backup() path (the gateway may hold the WAL open).
if src.suffix == ".db":
if not _safe_copy_db(src, dst):
failed_dbs.append(rel)
print(f" ⚠ Snapshot: SQLite safe copy FAILED for {rel} — file may be locked or corrupted")
if is_zeroed_sqlite_file(src):
nuls = " of NULs?" if in_dir else ""
print(f" ⚠ Snapshot: {rel} looks ZEROED "
f"(no SQLite header; {src.stat().st_size} bytes{nuls})")
continue
else:
shutil.copy2(src, dst)
manifest[rel] = dst.stat().st_size
except (OSError, PermissionError) as exc:
logger.warning("Could not snapshot %s: %s", rel, exc)
return manifest, failed_dbs, oversized_skipped
def _secure_quick_snapshot_tree(root: Path, snapshot_dir: Path) -> None:
"""Make a staged quick snapshot owner-only before it is published.
The staging directory is private from creation, so copied source modes can
be normalized safely before the final atomic rename exposes the snapshot.
Permission failures are intentionally fatal: publishing a readable
recovery bundle is worse than reporting a failed snapshot.
"""
if os.name == "nt":
return
os.chmod(root, 0o700)
os.chmod(snapshot_dir, 0o700)
for path in snapshot_dir.rglob("*"):
if path.is_dir():
os.chmod(path, 0o700)
elif path.is_file():
os.chmod(path, 0o600)
def _create_quick_snapshot_locked(
label: Optional[str], home: Path, keep: Optional[int], max_file_size: Optional[int]
) -> Optional[str]:
"""Copy the quick-snapshot set to a timestamped dir under state-snapshots/ and prune old ones.
``max_file_size`` skips (with a warning) larger files: the pre-update snapshot uses it so a
multi-GB ``state.db`` never stalls ``hermes update`` while the small files are always captured.
"""
root = _quick_snapshot_root(home)
ts = datetime.now(timezone.utc).strftime("%Y%m%d-%H%M%S")
base_snap_id = f"{ts}-{label}" if label else ts
snap_id, suffix = base_snap_id, 2
while (root / snap_id).exists():
snap_id = f"{base_snap_id}-{suffix}"
suffix += 1
staging_dir = root / f".{snap_id}.{os.getpid()}.partial"
shutil.rmtree(staging_dir, ignore_errors=True)
root.mkdir(parents=True, exist_ok=True, mode=0o700)
if os.name != "nt":
os.chmod(root, 0o700)
staging_dir.mkdir(mode=0o700, exist_ok=False)
logger.info("quick snapshot phase=copy status=started id=%s", snap_id)
manifest, failed_dbs, oversized_skipped = _copy_quick_snapshot_files(home, staging_dir, max_file_size)
if failed_dbs:
# Surface on stdout: a log-and-continue made a missing state.db backup look like a
# successful pre-update snapshot (#68474).
print(f" ⚠ CRITICAL: could not snapshot DB file(s): {', '.join(failed_dbs)}\n"
f" ⚠ If sessions disappear after the update, check {root}. {_snapshot_recovery_hint()}")
logger.error("Quick snapshot failed to capture DB file(s): %s", ", ".join(failed_dbs))
if not manifest:
shutil.rmtree(staging_dir, ignore_errors=True)
if failed_dbs:
# Distinguish "nothing to snapshot" from "state.db present but unreadable"
print(f" ⚠ Snapshot aborted: no files captured (failed DBs: {', '.join(failed_dbs)})")
return None
meta = {
"id": snap_id, "timestamp": ts, "label": label, "file_count": len(manifest),
"total_size": sum(manifest.values()), "files": manifest,
"failed_dbs": failed_dbs, "oversized_skipped": oversized_skipped,
}
with open(staging_dir / "manifest.json", "w", encoding="utf-8") as f:
json.dump(meta, f, indent=2)
_secure_quick_snapshot_tree(root, staging_dir)
os.replace(staging_dir, root / snap_id)
# Auto-prune (pre-update callers pass a smaller keep so state.db copies don't accumulate).
# Skip when a DB failed to capture OR was skipped for size (#68805): the snapshot is
# incomplete and the older one may hold the only recoverable database.
if not (failed_dbs or oversized_skipped):
_prune_oldest(_snapshot_dirs(root), _QUICK_DEFAULT_KEEP if keep is None else keep, shutil.rmtree, "snapshot")
else:
if oversized_skipped:
print(" ⚠ Skipping snapshot prune: DB file(s) skipped for size: " + ", ".join(oversized_skipped))
logger.warning("Quick snapshot skipped oversized DB file(s): %s", ", ".join(oversized_skipped))
logger.warning(
"Skipping snapshot prune because %d DB(s) failed to capture and/or %d were oversized "
"— preserving older snapshots as recovery source",
len(failed_dbs), len(oversized_skipped))
logger.info("quick snapshot phase=copy status=complete id=%s files=%d bytes=%d",
snap_id, len(manifest), sum(manifest.values()))
return snap_id
def _newest_first(root: Path, keep_entry) -> List[Path]:
"""Entries of *root* passing ``keep_entry``, newest (by name) first; ``[]`` if *root* is missing."""
if not root.exists():
return []
return sorted(filter(keep_entry, root.iterdir()), key=lambda p: p.name, reverse=True)
def _snapshot_dirs(root: Path) -> List[Path]:
"""Published snapshot directories under *root*, newest first."""
return _newest_first(root, lambda d: d.is_dir() and not d.name.startswith(".")
and not d.name.endswith(".partial"))
def list_quick_snapshots(limit: int = 20, hermes_home: Optional[Path] = None) -> List[Dict[str, Any]]:
"""List existing quick state snapshots, most recent first."""
results = []
for d in _snapshot_dirs(_quick_snapshot_root(hermes_home)):
manifest_path = d / "manifest.json"
if manifest_path.exists():
try:
results.append(json.loads(manifest_path.read_text(encoding="utf-8")))
except (json.JSONDecodeError, OSError):
results.append({"id": d.name, "file_count": 0, "total_size": 0})
if len(results) >= limit:
break
return results
def restore_quick_snapshot(snapshot_id: str, hermes_home: Optional[Path] = None) -> bool:
"""Restore state from a quick snapshot."""
home = hermes_home or get_hermes_home()
root = _quick_snapshot_root(home)
# Reject ids with separators or traversal so ``root / snapshot_id`` stays inside root.
if not snapshot_id or "/" in snapshot_id or "\\" in snapshot_id or snapshot_id in (".", ".."):
logger.error("Invalid snapshot_id: %s", snapshot_id)
return False
snap_dir = root / snapshot_id
if not _is_within(snap_dir, root.resolve()): # handles symlinks etc.
logger.error("Snapshot path traversal blocked for id: %s", snapshot_id)
return False
manifest_path = snap_dir / "manifest.json"
if not snap_dir.is_dir() or not manifest_path.exists():
return False
with open(manifest_path, encoding="utf-8") as f:
meta = json.load(f)
snap_res, home_res = snap_dir.resolve(), home.resolve()
restored = 0
for rel in meta.get("files", {}):
src = snap_dir / rel
dst = home / rel
if not (_is_within(src, snap_res) and _is_within(dst, home_res)):
logger.error("Manifest path traversal blocked: %s", rel)
continue
if not src.exists():
continue
dst.parent.mkdir(parents=True, exist_ok=True)
try:
if dst.suffix == ".db":
# Through the backup API so live connections see the restored data instead of
# stale pages from a replaced inode (#65942).
if not _safe_restore_db(src, dst):
# Refused (live holder) or failed: destination untouched — a failure, not a restore.
logger.error("Failed to restore %s: live-safe restore refused", rel)
continue
else:
shutil.copy2(src, dst)
restored += 1
except (OSError, PermissionError) as exc:
logger.error("Failed to restore %s: %s", rel, exc)
logger.info("Restored %d files from snapshot %s", restored, snapshot_id)
return restored > 0
# Kept in sync with ``_QUICK_STATE_FILES`` and ``cron/jobs.py``'s ``JOBS_FILE``.
_CRON_JOBS_REL = "cron/jobs.json"
def _count_cron_jobs(path: Path) -> Optional[int]:
"""Number of cron jobs in ``path`` (canonical ``{"jobs": [...]}`` or legacy bare list).
``None`` if missing or unparseable — "unknown", not zero: acting on an unreadable file could
mask a real corruption the user needs to see.
"""
if not path.is_file():
return None
try:
# utf-8-sig as cron/jobs.load_jobs: a Windows-editor BOM would otherwise read as
# "unreadable" and silently disable the post-update auto-restore safety net.
with open(path, "r", encoding="utf-8-sig") as f:
data = json.load(f)
except (OSError, json.JSONDecodeError):
return None
if isinstance(data, dict):
data = data.get("jobs", [])
return len(data) if isinstance(data, list) else None
def restore_cron_jobs_if_emptied(snapshot_id: str, hermes_home: Optional[Path] = None) -> Optional[Dict[str, Any]]:
"""Safety net for silent cron-job loss across ``hermes update``.
Conservative: restores only when the snapshot had MORE jobs than the live file (a user who
deleted jobs is never second-guessed); an unreadable live file is left so corruption surfaces.
Config-version migrations have been observed to leave ``cron/jobs.json`` valid-but-empty after an
update, silently dropping every scheduled job (issue #34600). The desktop scheduler can also overwrite
the file with its own small set of internally-tracked crons, causing partial loss (issue 52144).
"""
if not snapshot_id:
return None
home = hermes_home or get_hermes_home()
live_path = home / _CRON_JOBS_REL
live_count = _count_cron_jobs(live_path)
if live_count is None:
return None
snap_path = _quick_snapshot_root(home) / snapshot_id / _CRON_JOBS_REL
snap_count = _count_cron_jobs(snap_path)
# Fewer live jobs than the snapshot catches both total loss (0 vs N) and partial loss
# (1 vs 19) — the desktop scheduler can overwrite jobs.json with its own small set.
if not snap_count or live_count >= snap_count: # None or 0 — nothing worth restoring
return None
try:
live_path.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(snap_path, live_path)
except (OSError, PermissionError) as exc:
logger.error("Cron jobs were emptied during update but auto-restore failed: %s", exc)
return None
logger.warning(
"Restored %d cron job(s) from pre-update snapshot %s "
"(live file had %d job(s), snapshot had %d — jobs were lost during migration)",
snap_count, snapshot_id, live_count, snap_count)
return {"restored": True, "job_count": snap_count, "snapshot_id": snapshot_id}
def _sibling_profile_homes(invoking_home: Path) -> list[tuple[str, Path]]:
"""(name, home) for every OTHER profile on this install (the invoking one is snapshotted
separately). The update's code swap touches every profile, so its snapshot must too (#66140).
Never raises."""
homes: list[tuple[str, Path]] = []
try:
from hermes_cli.profiles import _get_default_hermes_home, _get_profiles_root, _PROFILE_ID_RE
invoking = invoking_home.resolve()
default_home = _get_default_hermes_home()
if default_home.is_dir() and default_home.resolve() != invoking:
homes.append(("default", default_home))
root = _get_profiles_root()
if root.is_dir():
for entry in sorted(root.iterdir()):
if (entry.is_dir() and entry.name != "default" and _PROFILE_ID_RE.match(entry.name)
and entry.resolve() != invoking):
homes.append((entry.name, entry))
except Exception as exc:
logger.debug("Sibling profile enumeration failed: %s", exc)
return homes
def create_pre_update_snapshots_all_profiles(
invoking_home: Optional[Path] = None, keep: Optional[int] = None, max_file_size: Optional[int] = None
) -> Dict[str, str]:
"""Pre-update quick snapshots for every SIBLING profile (#66140), same set/size cap/keep policy
as the invoking profile's; each lands under its OWN ``<home>/state-snapshots/``."""
results: Dict[str, str] = {}
home = invoking_home or get_hermes_home()
for name, profile_home in _sibling_profile_homes(home):
try:
snap_id = create_quick_snapshot(
label="pre-update", hermes_home=profile_home, keep=keep, max_file_size=max_file_size)
if snap_id:
results[name] = snap_id
except Exception as exc:
logger.debug("Pre-update snapshot for profile %s failed: %s", name, exc)
return results
# Config paths the update flow must never change (#64160): model routing and the MoA section are
# consumed machine-wide, so an update/repair cycle that rewrites them silently redirects paid
# inference. Dotted paths into raw config.yaml; a single-element tuple protects a whole section.
_PROTECTED_CONFIG_PATHS: Tuple[Tuple[str, ...], ...] = (
("model", "provider"), ("model", "default"), ("model", "base_url"), ("model", "api_key"),
("moa",))
def _read_raw_yaml_dict(path: Path) -> Optional[Dict[str, Any]]:
"""Parse ``path`` as a YAML mapping. ``None`` = missing/unreadable/non-dict."""
if not path.is_file():
return None
try:
import yaml
data = yaml.safe_load(path.read_text(encoding="utf-8"))
except Exception:
return None
return data if isinstance(data, dict) else None
def _get_config_path_value(data: Dict[str, Any], dotted: Tuple[str, ...]) -> Any:
node: Any = data
for key in dotted:
if not isinstance(node, dict):
return None
node = node.get(key)
return node
def _set_config_path_value(data: Dict[str, Any], dotted: Tuple[str, ...], value: Any) -> None:
node = data
for key in dotted[:-1]:
if not isinstance(node.get(key), dict):
node[key] = {}
node = node[key]
node[dotted[-1]] = value
def restore_config_model_settings_if_rewritten(
snapshot_id: str, hermes_home: Optional[Path] = None) -> Optional[Dict[str, Any]]:
"""Safety net for silent config.yaml model/MoA loss across ``hermes update``.
Mirrors :func:`restore_cron_jobs_if_emptied`: restore only the protected keys — never the
whole file — whose user-set value in the same-run pre-update snapshot changed or vanished.
"""
if not snapshot_id:
return None
home = hermes_home or get_hermes_home()
live_path = home / "config.yaml"
snap = _read_raw_yaml_dict(_quick_snapshot_root(home) / snapshot_id / "config.yaml")
if not snap:
return None # no snapshot copy — nothing to compare against
live = _read_raw_yaml_dict(live_path)
if live is None:
# Missing/unparseable live config is a failure the user should see (matches the cron net).
return None
restored_keys: list[str] = []
for dotted in _PROTECTED_CONFIG_PATHS:
snap_val = _get_config_path_value(snap, dotted)
if snap_val in (None, "", {}, []):
continue # user never set it — nothing to protect
if _get_config_path_value(live, dotted) != snap_val:
_set_config_path_value(live, dotted, snap_val)
restored_keys.append(".".join(dotted))
if not restored_keys:
return None
try:
from hermes_cli.config import atomic_config_write
atomic_config_write(live_path, live)
except (OSError, PermissionError) as exc:
logger.error("config.yaml model settings were rewritten during update but auto-restore failed: %s", exc)
return None
logger.warning(
"Restored user config value(s) %s from pre-update snapshot %s — "
"the update flow rewrote them (#64160)", ", ".join(restored_keys), snapshot_id)
return {"restored": True, "keys": restored_keys, "snapshot_id": snapshot_id}
def _restore_all_sibling_profiles(
profile_snapshots: Dict[str, str], invoking_home: Optional[Path], restore_fn, failure_log: str
) -> list[Dict[str, Any]]:
"""Run ``restore_fn(snap_id, hermes_home=...)`` for every sibling against ITS OWN
same-generation snapshot; one result dict (plus ``profile`` key) per restored profile.
Never raises."""
restored: list[Dict[str, Any]] = []
if not profile_snapshots:
return restored
home = invoking_home or get_hermes_home()
by_name = dict(_sibling_profile_homes(home))
for name, snap_id in profile_snapshots.items():
profile_home = by_name.get(name)
if profile_home is None:
continue
try:
result = restore_fn(snap_id, hermes_home=profile_home)
except Exception as exc:
logger.debug(failure_log, name, exc)
continue
if result:
result["profile"] = name
restored.append(result)
return restored
def restore_config_model_settings_all_profiles(
profile_snapshots: Dict[str, str], invoking_home: Optional[Path] = None) -> list[Dict[str, Any]]:
"""Run the config model-settings safety net for every sibling profile (see ``_restore_all_sibling_profiles``)."""
return _restore_all_sibling_profiles(
profile_snapshots, invoking_home, restore_config_model_settings_if_rewritten,
"Config model-settings restore check for profile %s failed: %s")
def restore_cron_jobs_all_profiles(
profile_snapshots: Dict[str, str], invoking_home: Optional[Path] = None) -> list[Dict[str, Any]]:
"""Run the cron-jobs safety net for every sibling profile (#66140); ``profile_snapshots`` comes
from :func:`create_pre_update_snapshots_all_profiles`, so restores are same-generation."""
return _restore_all_sibling_profiles(
profile_snapshots, invoking_home, restore_cron_jobs_if_emptied,
"Cron restore check for profile %s failed: %s")
def _prune_oldest(newest_first: List[Path], keep: int, remove, what: str) -> int:
"""``remove(path)`` every entry past the first *keep*; return how many succeeded."""
deleted = 0
for p in newest_first[keep:]:
try:
remove(p)
deleted += 1
except OSError as exc:
logger.warning("Failed to prune %s %s: %s", what, p.name, exc)
return deleted
def prune_quick_snapshots(keep: int = _QUICK_DEFAULT_KEEP, hermes_home: Optional[Path] = None) -> int:
"""Remove oldest quick snapshots beyond the keep limit. Returns count deleted."""
return _prune_oldest(_snapshot_dirs(_quick_snapshot_root(hermes_home)), keep, shutil.rmtree, "snapshot")
def run_quick_backup(args) -> None:
"""CLI entry point for hermes backup --quick."""
snap_id = create_quick_snapshot(label=getattr(args, "label", None))
if snap_id:
print(f"State snapshot created: {snap_id}\n"
f" {len(list_quick_snapshots())} snapshot(s) stored in {display_hermes_home()}/state-snapshots/\n"
f" Restore with: /snapshot restore {snap_id}")
else:
print("No state files found to snapshot.")
# --- Shared full-zip backup helper ---
def _write_full_zip_backup(out_path: Path, hermes_root: Path) -> Optional[Path]:
"""Full zip snapshot of ``hermes_root`` to ``out_path`` under the backup slot (same rules as
:func:`run_backup`); None when nothing to back up, another backup running, or write error."""
try:
with _backup_operation_lock(hermes_root):
return _write_full_zip_backup_locked(out_path, hermes_root)
except BackupInProgressError as exc:
logger.warning("Full-zip backup skipped: %s", exc)
return None
def _write_full_zip_backup_locked(out_path: Path, hermes_root: Path) -> Optional[Path]:
scan_started = time.monotonic()
logger.info("automatic backup phase=scan status=started")
try:
files_to_add = list(_iter_backup_files(hermes_root, out_path))
except OSError as exc:
logger.warning("Full-zip backup: walk failed: %s", exc)
return None
if not files_to_add:
return None
logger.info("automatic backup phase=scan status=complete duration_ms=%.1f files=%d",
(time.monotonic() - scan_started) * 1000, len(files_to_add))
def _db_failure(rel_path: Path) -> None:
logger.warning("Full-zip backup aborted: SQLite snapshot failed for %s", rel_path)
raise _SQLiteSnapshotError(str(rel_path))
archive_started = time.monotonic()
try:
with _atomic_output_path(out_path) as archive_path, zipfile.ZipFile(
archive_path, "w", zipfile.ZIP_DEFLATED, compresslevel=6) as zf:
_write_zip_entries(
zf, files_to_add, out_path, on_db_failure=_db_failure, track_bytes=False,
on_error=lambda rel, exc: logger.debug("Skipping %s in zip backup: %s", rel, exc),
on_progress=lambda i: logger.info(
"automatic backup phase=archive status=progress completed=%d total=%d", i, len(files_to_add)))
except (OSError, _SQLiteSnapshotError) as exc:
# The hidden partial is already gone; ``out_path`` may be a previous valid backup: keep it.
logger.warning("Full-zip backup: zip write failed: %s", exc)
return None
logger.info("automatic backup phase=archive status=complete duration_ms=%.1f files=%d bytes=%d",
(time.monotonic() - archive_started) * 1000, len(files_to_add),
out_path.stat().st_size)
return out_path
# --- Pre-update / pre-migration auto-backups ---
_PRE_UPDATE_BACKUPS_DIR = "backups"
_PRE_UPDATE_PREFIX = "pre-update-"
_PRE_UPDATE_DEFAULT_KEEP = 5
_PRE_MIGRATION_PREFIX = "pre-migration-"
_PRE_MIGRATION_DEFAULT_KEEP = 5
def _prune_prefixed_zips(backup_dir: Path, prefix: str, keep: int, what: str) -> int:
"""Remove oldest ``<prefix>*.zip`` in *backup_dir* beyond *keep*; return count deleted.
Only prefix-matched files are touched, so hand-made zips or other backup kinds survive.
"""
backups = _newest_first(backup_dir, lambda p: p.is_file() and p.name.startswith(prefix)
and p.suffix.lower() == ".zip")
return _prune_oldest(backups, keep, Path.unlink, what)
def _create_prefixed_full_backup(
hermes_home: Optional[Path], prefix: str, keep: int, what: str, prune_what: str) -> Optional[Path]:
"""Write ``<HERMES_HOME>/backups/<prefix><timestamp>.zip`` and prune older same-prefix zips.
Returns the path, or ``None`` if nothing to back up or the write failed. Never raises."""
hermes_root = hermes_home or get_default_hermes_root()
if not hermes_root.is_dir():
return None
backup_dir = hermes_root / _PRE_UPDATE_BACKUPS_DIR
try:
backup_dir.mkdir(parents=True, exist_ok=True)
except OSError as exc:
logger.warning("Could not create %s backup dir %s: %s", what, backup_dir, exc)
return None
out_path = backup_dir / f"{prefix}{datetime.now().strftime('%Y-%m-%d-%H%M%S')}.zip"
if _write_full_zip_backup(out_path, hermes_root) is None:
return None
_prune_prefixed_zips(backup_dir, prefix, keep, prune_what)
return out_path
def create_pre_update_backup(
hermes_home: Optional[Path] = None, keep: int = _PRE_UPDATE_DEFAULT_KEEP) -> Optional[Path]:
"""Full zip backup to ``backups/pre-update-<timestamp>.zip``, auto-pruned; ``None`` if nothing
was found or the backup failed. Never raises — ``hermes update`` continues anyway."""
return _create_prefixed_full_backup(hermes_home, _PRE_UPDATE_PREFIX, max(keep, 1), "pre-update", "backup")
def create_pre_migration_backup(
hermes_home: Optional[Path] = None, keep: int = _PRE_MIGRATION_DEFAULT_KEEP) -> Optional[Path]:
"""Full zip backup to ``backups/pre-migration-<timestamp>.zip`` before ``hermes claw migrate``
(same dir as update backups so listings/``hermes import`` find it); ``None`` if nothing was
found or the write failed. Never raises."""
return _create_prefixed_full_backup(
hermes_home, _PRE_MIGRATION_PREFIX, max(keep, 0), "pre-migration", "pre-migration backup")
# ---- BEGIN PLUGIN-COMPAT (revert-scheduled; see COMPAT_MANIFEST.md) ----
# Names external plugins imported from this module before the Sep 2026 decomposition.
# Internal code MUST NOT use these (scripts/check_compat_pointers.py fails CI if it does).
# The whole block is removed by reverting the commit that added it.
def copy_db_and_verify(src: Path, dst: Path) -> bool:
"""Like :func:`_safe_copy_db` but verifies the destination after copy.
Returns True only when the copy succeeded AND the destination is valid
SQLite (header + integrity check). Verification honours the default
size ceiling — a multi-GB destination gets the header + schema probe
rather than a full ``PRAGMA integrity_check`` that would page through
the whole file.
"""
if not _safe_copy_db(src, dst):
return False
integrity = verify_sqlite_integrity(dst, run_pragma=True)
if not integrity.get("valid"):
try:
dst.unlink(missing_ok=True)
except OSError:
pass
logger.warning("Backup of %s failed integrity verification: %s", src, integrity.get("message"))
return False
return True
# ---- END PLUGIN-COMPAT ----