Files
hermes-agent/hermes_cli/backup.py

2164 lines
87 KiB
Python

"""Backup and import commands for hermes CLI."""
import json
import logging
import os
import shutil
import sqlite3
import stat
import sys
import tempfile
import threading
import time
import zipfile
from contextlib import contextmanager, suppress
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
from hermes_constants import (
_get_platform_default_hermes_home,
get_default_hermes_root,
get_hermes_home,
display_hermes_home,
)
from utils import (
_preserve_file_mode,
_preserve_file_owner,
_restore_file_mode,
_restore_file_owner,
atomic_replace,
)
# Shared formatter; the private alias is kept because claw.py and the backup
# tests import ``_format_size`` from this module.
from hermes_cli.sizefmt import format_bytes as _format_size
logger = logging.getLogger(__name__)
# ---------------------------------------------------------------------------
# Exclusion rules
# ---------------------------------------------------------------------------
# Where ``hermes backup --quick`` / ``/snapshot`` / the pre-update safety net
# write their state snapshots (see ``create_quick_snapshot`` below). Defined up
# here because the exclusion set needs it.
_QUICK_SNAPSHOTS_DIR = "state-snapshots"
# Directory names to skip entirely (matched against each path component)
# ``hermes-agent`` is special-cased to root level only in ``_should_exclude``
# so that skill directories like ``skills/autonomous-ai-agents/hermes-agent/``
# are not accidentally excluded.
#
# The dependency/cache entries below matter for more than tidiness: without
# them a single plugin venv, MCP-server install, or pip/uv cache living under
# HERMES_HOME gets walked file-by-file, ballooning a backup to hundreds of
# thousands of entries that crawl for hours — the exact "backup stuck for
# days / 426543 files" symptom users hit. The dependency/test-env names mostly
# mirror ``agent.skill_utils.EXCLUDED_SKILL_DIRS`` (the project's canonical
# "regeneratable dir" set); ``.cache`` is an additional backup-only entry, as
# it names a broad regeneratable cache convention (pip/uv/etc.) that the skill
# scanner doesn't need to prune but a backup walk does. We deliberately do NOT
# exclude ``.archive`` here because the curator's ``skills/.archive/`` holds
# restorable user skills that must survive a backup.
_EXCLUDED_DIRS = {
"hermes-agent", # the codebase repo — re-clone instead
"__pycache__", # bytecode caches — regenerated on import
".git", # nested git dirs (profiles shouldn't have these, but safety)
"node_modules", # js deps — reinstalled on demand
"backups", # prior auto-backups — don't nest backups exponentially
_QUICK_SNAPSHOTS_DIR, # quick/pre-update state snapshots — same reason as
# ``backups``: each holds a full copy of state.db, so
# zipping them re-ships the DB once per snapshot
"checkpoints", # session-local trajectory caches — regenerated per-session,
# session-hash-keyed so they don't port to another machine anyway
# Live browser profiles (e.g. the CDP Brave profile under browser-profiles/).
# Chromium holds its SQLite DBs with exclusive locks while running, and
# sqlite3.Connection.backup() retries SQLITE_BUSY forever instead of honoring
# the busy timeout — a full backup hangs mid-archive on the first locked DB.
# Profiles are regenerable (cache + re-login) and unsafe to snapshot live.
"browser-profiles",
# Real-profile browsing snapshot (browser.use_real_profile). Holds copies of
# the user's Cookies / Login Data / Web Data — a credential-bearing store
# that must NOT enter a backup archive. It is regenerated from the user's
# live profile on the next consented launch. Singular, distinct from the
# ``browser-profiles`` CDP dir above; both are excluded.
"browser-profile",
# Python dependency trees (plugin / MCP-server venvs under HERMES_HOME) —
# regenerated by reinstalling; never irreplaceable state.
".venv",
"venv",
"site-packages",
# Tool / build caches — all regeneratable.
".cache",
".tox",
".nox",
".pytest_cache",
".mypy_cache",
".ruff_cache",
}
# Hermes-managed runtime downloads that only exist at the top of a profile
# home: local GGUF models, llama.cpp runtime binaries, and the managed Node
# installation. All of them are re-downloaded on demand (model catalog,
# runtime bootstrap, node installer) and routinely reach tens to hundreds of
# GB, so zipping them turns a backup into an hours-long compress of
# incompressible weights (the "backup stuck at N files" symptom). Matched
# ONLY at the root of HERMES_HOME and at ``profiles/<name>/`` — a deeper
# directory that happens to share one of these names (a skill's ``models/``,
# a user checkout) is user data and stays in the backup.
_EXCLUDED_ROOT_DIRS = {"models", "runtimes", "node"}
def _in_excluded_root_dir(rel_path: Path) -> bool:
"""True when *rel_path* (relative to HERMES_HOME) is, or sits inside, a
Hermes-managed runtime tree at the top of a profile home."""
parts = rel_path.parts
# Named profiles are profile homes too: profiles/<name>/models etc.
return bool(parts) and (
parts[0] in _EXCLUDED_ROOT_DIRS
or (len(parts) >= 3 and parts[0] == "profiles" and parts[2] in _EXCLUDED_ROOT_DIRS)
)
# File-name suffixes to skip
_EXCLUDED_SUFFIXES = (
".pyc",
".pyo",
# SQLite sidecar files — the backup takes a consistent snapshot of ``*.db``
# via ``sqlite3.backup()``, so shipping the live WAL / shared-memory /
# rollback-journal alongside would pair a fresh snapshot with stale sidecar
# state and produce a torn restore on the next open. They're transient and
# regenerated on first connection anyway.
".db-wal",
".db-shm",
".db-journal",
)
# File names to skip (runtime state that's meaningless on another machine)
_EXCLUDED_NAMES = {".backup.lock", "gateway.pid", "cron.pid"}
# File-name prefixes to skip. The desktop updater's pre-flight drops
# ``state.db.pre-update-emergency-<timestamp>.bak`` at the HERMES_HOME root
# (apps/desktop/electron/main.ts preflightStateDb) — a backup artifact in
# the same class as ``backups/`` and ``state-snapshots/``, so a full backup
# must not re-ship it. Matched by prefix because the name carries a
# timestamp; a plain ``.bak`` suffix rule would drop user files.
_EXCLUDED_PREFIXES = (
"state.db.pre-update-emergency-",
)
# File names that ``hermes import`` must never overwrite, matched by basename so
# they're caught for the root profile (``gateway_state.json``) and for named
# profiles alike (``profiles/<name>/gateway_state.json``).
#
# These hold *volatile gateway/process runtime state that is namespaced to the
# machine or container the backup was taken on* — PIDs in a dead process
# namespace, a runtime lock, the process registry, and the gateway's last
# recorded run/desired state. Restoring them onto a different host (or a hosted
# container) is at best meaningless and at worst actively harmful:
#
# - ``gateway_state.json`` drives the container-boot reconciler
# (``container_boot._read_desired_state``), which only auto-starts a
# gateway whose recorded state is ``running``. A backup taken from a
# machine where the gateway was stopped (or carrying a stale/foreign
# value) overwrites the container's own state and leaves the gateway
# stuck "starting"/"cooking", disconnecting it from the Nous portal
# (NS-508 / the second half of NS-501).
# - ``gateway.pid`` / ``cron.pid`` / ``gateway.lock`` / ``processes.json``
# reference PIDs and locks in the *source* machine's process namespace; a
# numerically-equal PID in the new environment is a different process.
# These mirror exactly what ``container_boot._STALE_RUNTIME_FILES`` already
# sweeps on every container boot.
#
# Older backups predate the backup-side exclusions, so we filter on import too
# rather than trusting the archive's contents.
_IMPORT_SKIP_NAMES = {"gateway_state.json", "gateway.pid", "cron.pid", "gateway.lock", "processes.json"}
# zipfile.open() drops Unix mode bits on extract; restore tightens these to 0600.
_SECRET_FILE_NAMES = {".env", "auth.json", "state.db"}
# Reserved archive subtree for provider state that lives OUTSIDE HERMES_HOME
# (e.g. ~/.honcho, ~/.hindsight). The active memory provider declares these via
# MemoryProvider.backup_paths(); they're stored under this prefix encoded
# relative to the user's home directory, and restored to their original
# home-relative location on import. Anything not under home is skipped.
_EXTERNAL_PREFIX = "_external/"
class BackupInProgressError(RuntimeError):
"""Raised when another process already owns the Hermes backup slot."""
class _SQLiteSnapshotError(RuntimeError):
pass
class _SQLiteBackupTimeout(RuntimeError):
"""Raised when a SQLite snapshot remains busy past its deadline."""
@contextmanager
def _backup_operation_lock(hermes_home: Path, timeout_seconds: float = 0.25):
"""Acquire one cross-process backup slot for full and quick snapshots."""
lock_path = hermes_home / ".backup.lock"
lock_path.parent.mkdir(parents=True, exist_ok=True)
handle = lock_path.open("a+b")
acquired = False
deadline = time.monotonic() + max(0.0, timeout_seconds)
try:
if os.name == "nt":
import msvcrt
if lock_path.stat().st_size == 0:
handle.write(b" ")
handle.flush()
def _lock_op(flag: int) -> None:
handle.seek(0)
msvcrt.locking(handle.fileno(), flag, 1)
lock_flag, unlock_flag = msvcrt.LK_NBLCK, msvcrt.LK_UNLCK
else:
import fcntl
def _lock_op(flag: int) -> None:
fcntl.flock(handle.fileno(), flag)
lock_flag, unlock_flag = fcntl.LOCK_EX | fcntl.LOCK_NB, fcntl.LOCK_UN
while True:
try:
_lock_op(lock_flag)
acquired = True
break
except OSError:
if time.monotonic() >= deadline:
raise BackupInProgressError("another Hermes backup is already running")
time.sleep(0.05)
yield
finally:
if acquired:
with suppress(OSError):
_lock_op(unlock_flag)
handle.close()
@contextmanager
def _atomic_output_path(final_path: Path):
"""Yield a hidden sibling path and publish it only after a clean close."""
partial_path = final_path.with_name(f".{final_path.name}.{os.getpid()}-{threading.get_ident()}.partial")
partial_path.unlink(missing_ok=True)
try:
yield partial_path
os.replace(partial_path, final_path)
except BaseException:
partial_path.unlink(missing_ok=True)
raise
def _collect_memory_provider_external_paths() -> List[Path]:
"""Return existing absolute paths the active memory provider stores
Reads ``memory.provider``, loads just that provider, and asks it for ``backup_paths()``.
Returns ``[]`` when no external provider is active or it can't be loaded: backup must never
fail because of a flaky plugin.
"""
try:
from plugins.memory import _get_active_memory_provider, load_memory_provider
active = _get_active_memory_provider()
provider = load_memory_provider(active) if active else None
except Exception:
return []
if not active or provider is None:
return []
try:
declared = provider.backup_paths() or []
except Exception as exc:
logger.warning("backup_paths() failed for memory provider %r: %s", active, exc)
return []
out: List[Path] = []
seen: set = set()
for raw in declared:
try:
p = Path(raw).expanduser()
resolved = p.resolve() if p.exists() else None
except Exception:
continue
if resolved is not None and resolved not in seen:
seen.add(resolved)
out.append(p)
return out
def _iter_external_files(base: Path) -> List[Path]:
"""Yield regular files under *base* (a file or a directory), skipping
symlinks, caches, and pyc files. *base* itself may be a file."""
if base.is_file() and not base.is_symlink():
return [base]
files: List[Path] = []
if not base.is_dir():
return files
for dirpath, dirnames, filenames in os.walk(base, followlinks=False):
dp = Path(dirpath)
dirnames[:] = [d for d in dirnames if d not in _EXCLUDED_DIRS]
for fname in filenames:
fpath = dp / fname
if fpath.is_symlink() or fname in _EXCLUDED_NAMES or fname.endswith(_EXCLUDED_SUFFIXES):
continue
files.append(fpath)
return files
def _should_exclude(rel_path: Path) -> bool:
"""Return True if *rel_path* (relative to hermes root) should be skipped."""
parts = rel_path.parts
if _in_excluded_root_dir(rel_path):
return True
# ``hermes-agent`` only matches at the root level (first component).
# Nested directories with the same name — e.g.
# ``skills/autonomous-ai-agents/hermes-agent/`` — must be preserved.
if any(p in _EXCLUDED_DIRS and (p != "hermes-agent" or p == parts[0]) for p in parts):
return True
name = rel_path.name
return (
name in _EXCLUDED_NAMES
or name.startswith(_EXCLUDED_PREFIXES)
or name.endswith(_EXCLUDED_SUFFIXES)
)
def _should_skip_backup_file(abs_path: Path, rel_path: Path, out_path: Path) -> bool:
"""Return True when a candidate file should not be written to a backup zip."""
if _should_exclude(rel_path):
return True
# zipfile.write() follows file symlinks, so skip links before any archive
# write can copy data from outside HERMES_HOME.
if abs_path.is_symlink():
return True
try:
return abs_path.resolve() == out_path.resolve()
except (OSError, ValueError):
return False
def _iter_backup_files(hermes_root: Path, out_path: Path, skipped_dirs: Optional[set] = None):
"""Yield ``(abs_path, rel_path)`` for every file a full backup should hold.
The one owner of the backup walk policy: directory pruning (so os.walk never descends a multi-GB
excluded tree), the root-only ``hermes-agent`` carve-out, profile-home-root runtime trees, and
the per-file exclusion rules — shared by the manual ``hermes backup`` path and the automatic
pre-update/pre-migration path so the two can never drift.
"""
for dirpath, dirnames, filenames in os.walk(hermes_root, followlinks=False):
rel_dir = Path(dirpath).relative_to(hermes_root)
# ``hermes-agent`` is only pruned at the root level; nested dirs
# with the same name (e.g. in skills/) must be preserved. Managed
# runtime trees (models/, runtimes/, node/) are pruned only at a
# profile-home root — see _EXCLUDED_ROOT_DIRS.
is_root = rel_dir == Path(".")
orig_dirnames = dirnames[:]
dirnames[:] = [
d for d in dirnames
if (d not in _EXCLUDED_DIRS or (d == "hermes-agent" and not is_root))
and not _in_excluded_root_dir(rel_dir / d)
]
if skipped_dirs is not None:
for removed in set(orig_dirnames) - set(dirnames):
skipped_dirs.add(str(rel_dir / removed))
for fname in filenames:
rel = rel_dir / fname
fpath = hermes_root / rel
if _should_skip_backup_file(fpath, rel, out_path):
continue
yield fpath, rel
# ---------------------------------------------------------------------------
# SQLite safe copy
# ---------------------------------------------------------------------------
def _close_quietly(conn: Optional[sqlite3.Connection]) -> None:
if conn is not None:
with suppress(Exception):
conn.close()
def _query_ro_sqlite(path: Path, fn):
"""Run ``fn(conn)`` on a read-only connection to *path*; return ``(value, None)`` or ``(None, exc)``."""
conn = None
try:
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True, timeout=1.0)
return fn(conn), None
except Exception as exc:
return None, exc
finally:
_close_quietly(conn)
def _safe_copy_db(src: Path, dst: Path, *, timeout_seconds: float = 10.0) -> bool:
"""Copy a SQLite database safely using the backup() API.
Handles WAL mode — produces a consistent snapshot even while the DB is being written to. Fail
closed if a consistent snapshot cannot be created: copying only the live main file can omit
committed WAL data.
"""
conn = None
backup_conn = None
try:
# Disable sqlite3's implicit busy wait so backup() progress callbacks
# control the full locked-source deadline instead of adding the
# connection's default timeout before each callback.
conn = sqlite3.connect(f"file:{src}?mode=ro", uri=True, timeout=0.0)
backup_conn = sqlite3.connect(str(dst))
busy_deadline = time.monotonic() + max(0.0, timeout_seconds)
def _check_backup_progress(status: int, _remaining: int, _total: int) -> None:
nonlocal busy_deadline
now = time.monotonic()
if status in (sqlite3.SQLITE_BUSY, sqlite3.SQLITE_LOCKED):
if now >= busy_deadline:
raise _SQLiteBackupTimeout(f"database remained locked for {timeout_seconds:g} seconds")
else:
busy_deadline = now + max(0.0, timeout_seconds)
conn.backup(backup_conn, pages=256, progress=_check_backup_progress, sleep=0.1)
return True
except Exception as exc:
logger.warning("SQLite safe copy failed for %s: %s", src, exc)
# Windows will not remove the partial destination while SQLite still
# has it open. Close it before fail-closed cleanup; the finally block
# still owns the source and any close failure.
_close_quietly(backup_conn)
backup_conn = None
with suppress(OSError):
dst.unlink(missing_ok=True)
return False
finally:
_close_quietly(backup_conn)
_close_quietly(conn)
def is_zeroed_sqlite_file(path: Path, *, probe_bytes: int = 100, force: bool = False) -> bool:
"""True when *path* looks like the #68474 zeroed-state.db signature.
Only regular files qualify: a special file at the path (FIFO, device, socket) is never "zeroed"
— and probing one could block indefinitely (opening a FIFO for read waits for a writer), so
refuse before any I/O.
"""
try:
if not path.is_file():
return False
except OSError:
return False
from hermes_cli.sqlite_safe_read import has_live_connection, read_header_bytes_preopen
if not force and has_live_connection(path):
return False
head = read_header_bytes_preopen(path, length=max(16, probe_bytes), force=force)
# Empty or all-NUL header => zeroed; a real header (or unreadable) => not.
return head is not None and not head.startswith(b"SQLite format 3") and not any(head)
# ---------------------------------------------------------------------------
# SQLite integrity verification
# ---------------------------------------------------------------------------
_SQLITE_HEADER = b"SQLite format 3\0"
# Default ceiling above which ``PRAGMA integrity_check`` is skipped in favour
# of the (O(1)) header + structural probe. ``integrity_check`` walks every
# b-tree page in the file, so its cost scales with database size: on a 30 GB
# state.db it runs for many minutes of pegged CPU with no output, which reads
# to the user as a hung `hermes update` (#70553 follow-up). Sessions databases
# in the tens of GB are normal for heavy users, so the size-unbounded check is
# never an acceptable default on the update path.
DEFAULT_INTEGRITY_CHECK_MAX_BYTES = 2 << 30 # 2 GiB
def verify_sqlite_integrity(
path: Path,
*,
check_header: bool = True,
run_pragma: bool = True,
max_bytes: int = DEFAULT_INTEGRITY_CHECK_MAX_BYTES,
) -> dict:
"""Verify that a SQLite database at *path* is intact.
Checks, in order: 1. File exists and has an expected minimum size. 2. SQLite header magic bytes
are present. 3. For files at or under ``max_bytes``, a read-only ``PRAGMA integrity_check``. For
larger files, a cheap structural probe (schema read) instead — see ``max_bytes``.
"""
result: dict = {"valid": False, "message": "", "size": None}
def _done(message: str, valid: bool = False) -> dict:
result["valid"] = valid
result["message"] = message
return result
try:
st = path.stat()
except FileNotFoundError:
return _done(f"not found: {path}")
except OSError as exc:
return _done(f"cannot stat: {exc}")
result["size"] = st.st_size
if st.st_size < 100: # SQLite minimum viable size (header + 1 page)
return _done(f"too small ({st.st_size} bytes) to be a valid SQLite database")
oversized = max_bytes > 0 and st.st_size > max_bytes
if check_header:
# Byte-level read: refused when a live connection exists, because
# close() would cancel this process's POSIX locks on the file (see
# hermes_cli.sqlite_safe_read). Verification targets snapshots and
# backup artifacts, which are offline by construction.
from hermes_cli.sqlite_safe_read import read_header_bytes_preopen
head = read_header_bytes_preopen(path, length=len(_SQLITE_HEADER))
if head is None:
return _done("cannot read header")
if head != _SQLITE_HEADER:
return _done(f"missing SQLite header magic (got {head[:16].hex()!r})")
if oversized:
# Too large to page through PRAGMA integrity_check (which is O(file
# size) and would peg a CPU for minutes on a multi-GB state.db).
# Fall back to a cheap O(1) structural probe: the header check above
# catches the #68474 zeroed signature, and opening the DB read-only
# plus reading sqlite_master + the page geometry catches the
# malformed-schema and truncated-header-page classes. Both are
# constant-time — they parse the schema, they do not walk the data.
_, exc = _query_ro_sqlite(
path,
lambda c: (
c.execute("PRAGMA schema_version").fetchone(),
c.execute("SELECT count(*) FROM sqlite_master").fetchone(),
),
)
if exc is not None:
kind = "failed" if isinstance(exc, sqlite3.DatabaseError) else "error"
return _done(f"schema probe {kind}: {exc}")
return _done(
f"size {st.st_size:,} bytes exceeds max_bytes {max_bytes:,}; "
"skipped PRAGMA integrity_check (header + schema probe passed)",
valid=True,
)
if run_pragma:
rows, exc = _query_ro_sqlite(
path,
lambda c: [str(r[0]) for r in c.execute("PRAGMA integrity_check").fetchall()],
)
if exc is not None:
kind = "cannot open database" if isinstance(exc, sqlite3.DatabaseError) else "integrity check error"
return _done(f"{kind}: {exc}")
if rows == ["ok"]:
return _done("integrity check passed", valid=True)
return _done(f"integrity check failed: {'; '.join(rows[:5])}")
return _done("header check passed", valid=True)
def _foreign_db_holder_pids(db_path: Path) -> Optional[List[int]]:
"""PIDs of OTHER processes holding *db_path* or its WAL/SHM open.
Linux-only ``/proc/<pid>/fd`` scan (no psutil dependency), preserving the kernel's ``(deleted)``
suffix so an already-unlinked sidecar generation — the #90950 split-brain fingerprint — still
counts as held.
"""
if not sys.platform.startswith("linux"):
return None
def _canonical(path: str) -> str:
return os.path.normcase(os.path.abspath(path.removesuffix(" (deleted)")))
def _holds_watched(fd_dir: str) -> bool:
for fd in os.listdir(fd_dir):
try:
target = os.readlink(f"{fd_dir}/{fd}")
except OSError:
continue
if _canonical(target) in watched:
return True
return False
canonical_db = _canonical(os.fspath(db_path))
watched = {canonical_db, canonical_db + "-wal", canonical_db + "-shm"}
pids: List[int] = []
try:
own_pid = os.getpid()
for pid_str in os.listdir("/proc"):
if not pid_str.isdigit() or int(pid_str) == own_pid:
continue
try:
if _holds_watched(f"/proc/{pid_str}/fd"):
pids.append(int(pid_str))
except OSError:
continue
except OSError:
return None
return pids
def _safe_restore_db(src: Path, dst: Path) -> bool:
"""Restore a SQLite database from snapshot *src* into live *dst*.
Uses SQLite's backup() API to write snapshot pages into the live database file, preserving the
file's inode and WAL state so that any other process still holding the DB open (gateway,
dashboard, another CLI session) sees the restored data on the next read — instead of continuing
to serve stale cached pages from a replaced inode.
Falls back to the unlink+move approach on failure so restore never blocks on a transient error.
"""
try:
dst_conn = sqlite3.connect(str(dst))
# Force a WAL checkpoint so the backup starts from a clean
# state rather than writing on top of a deep WAL.
with suppress(Exception):
dst_conn.execute("PRAGMA wal_checkpoint(TRUNCATE)")
src_conn = sqlite3.connect(f"file:{src}?mode=ro", uri=True)
try:
src_conn.backup(dst_conn)
finally:
src_conn.close()
dst_conn.close()
# Restore original file permissions from the snapshot
with suppress(Exception):
dst.chmod(src.stat().st_mode)
return True
except Exception as exc:
logger.warning("SQLite safe restore failed for %s -> %s: %s", src, dst, exc)
return _unlink_move_restore_db(src, dst)
def _unlink_move_restore_db(src: Path, dst: Path) -> bool:
"""Fallback restore: unlink+move (the old approach). Works when no process holds the DB open.
Replacing the inode under a live holder is the #90950 corruption class: the holder keeps
writing through a deleted-inode fd (split brain), and removing its sidecars detaches the WAL
index it is checkpointing through. The backup-API path is the live-safe route; if it failed,
fail closed rather than corrupt. The foreign-pid scan deliberately excludes THIS process, but
an in-process SessionDB (the agent's own handle during /snapshot restore, a second SessionDB
instance, a read pool) is exactly as much of a live holder, so ``offline_file_access`` fails
CLOSED when any tracked connection to *dst* is live and holds the connection-lifecycle lock
across the whole swap so no new connection can appear mid-replace.
"""
from hermes_cli.sqlite_safe_read import LiveConnectionError, offline_file_access
try:
holders = _foreign_db_holder_pids(dst)
if holders:
logger.error(
"Refusing unlink+move restore of %s: process(es) %s still "
"hold the database or its WAL open. Stop them and retry.",
dst, holders,
)
return False
with offline_file_access(dst, what="unlink+move restore of"):
tmp = dst.parent / f".{dst.name}.snap_restore"
shutil.copy2(src, tmp)
dst.unlink(missing_ok=True)
# Drop the destination's sidecars before installing the snapshot. The
# snapshot is a checkpointed ``sqlite3.backup()`` image that owns no
# WAL, so any ``-wal``/``-shm`` still here describes the database we
# just unlinked (an ungracefully killed gateway leaves them behind —
# exactly when a restore gets run). SQLite would replay that foreign
# WAL over the restored file on next open and come up "malformed" (or
# silently resurrect post-snapshot rows). Same reasoning as
# ``_EXCLUDED_SUFFIXES``, applied to the restore destination.
for _sidecar_suffix in ("-wal", "-shm", "-journal"):
dst.with_name(dst.name + _sidecar_suffix).unlink(missing_ok=True)
shutil.move(str(tmp), str(dst))
return True
except LiveConnectionError as exc2:
logger.error(
"Refusing unlink+move restore of %s: %s Close the in-process "
"database handles (or restart Hermes) and retry.",
dst, exc2,
)
return False
except Exception as exc2:
logger.error("Fallback restore also failed for %s -> %s: %s", src, dst, exc2)
return False
def _zip_sqlite_snapshot(
zf: zipfile.ZipFile, abs_path: Path, rel_path: Path, out_path: Path
) -> Optional[int]:
"""Add a WAL-safe snapshot of *abs_path* to *zf*; return its byte size, or None on failure.
The snapshot is staged alongside the output zip so the temp file lives on the same
filesystem: the system default (/tmp) may be a small tmpfs that cannot hold large databases,
causing silent backup incompleteness.
"""
with tempfile.NamedTemporaryFile(
suffix=".db", delete=False, dir=str(out_path.parent)
) as tmp:
tmp_db = Path(tmp.name)
try:
if not _safe_copy_db(abs_path, tmp_db):
return None
zf.write(tmp_db, arcname=str(rel_path))
return tmp_db.stat().st_size
finally:
tmp_db.unlink(missing_ok=True)
def _write_zip_entries(
zf: zipfile.ZipFile,
files_to_add: List[Tuple[Path, Path]],
out_path: Path,
*,
on_db_failure,
on_error,
on_progress,
track_bytes: bool,
) -> int:
"""Add every ``(abs_path, rel_path)`` to *zf*, WAL-safe for ``*.db``; return bytes archived.
``on_db_failure(rel_path)`` runs when a SQLite snapshot fails (it may raise to abort);
``on_error(rel_path, exc)`` records a per-file read failure; ``on_progress(index)`` fires
every 500 files. ``track_bytes`` stats each archived plain file for the size total.
"""
total_bytes = 0
for i, (abs_path, rel_path) in enumerate(files_to_add, 1):
try:
if abs_path.suffix == ".db":
size = _zip_sqlite_snapshot(zf, abs_path, rel_path, out_path)
if size is None:
on_db_failure(rel_path)
continue
total_bytes += size
else:
zf.write(abs_path, arcname=str(rel_path))
if track_bytes:
total_bytes += abs_path.stat().st_size
except (PermissionError, OSError, ValueError) as exc:
on_error(rel_path, exc)
continue
if i % 500 == 0:
on_progress(i)
return total_bytes
def _print_capped(header: str, lines: List[str], indent: str) -> None:
"""Print *header*, then at most 10 of *lines* (each prefixed by *indent*) and a "... and N more" tail."""
print(header)
for line in lines[:10]:
print(f"{indent}{line}")
if len(lines) > 10:
print(f"{indent}... and {len(lines) - 10} more")
def _print_skipped_warnings(errors: List[str]) -> None:
_print_capped(f"\n Warnings ({len(errors)} files skipped):", errors, " ")
# ---------------------------------------------------------------------------
# Backup
# ---------------------------------------------------------------------------
def _resolve_backup_output_path(output: Optional[str]) -> Path:
"""Turn ``--output`` (file, directory, or None) into a ``.zip`` path whose parent exists.
A bad/unwritable output path (permission denied, unreadable parent, etc.) gives a clean
one-line error, not a raw traceback: is_dir() and mkdir() both hit the filesystem.
"""
out_path = None
default_name = f"hermes-backup-{datetime.now().strftime('%Y-%m-%d-%H%M%S')}.zip"
try:
if output:
out_path = Path(output).expanduser().resolve()
# If user gave a directory, put the zip inside it
if out_path.is_dir():
out_path = out_path / default_name
else:
out_path = Path.home() / default_name
if out_path.suffix.lower() != ".zip":
out_path = out_path.with_suffix(out_path.suffix + ".zip")
out_path.parent.mkdir(parents=True, exist_ok=True)
except OSError as exc:
print(f"Error: cannot write backup to {output or out_path}: {exc}")
raise SystemExit(1) from exc
return out_path
def _collect_external_entries() -> tuple[list[tuple[Path, str]], list[str]]:
"""``([(abs_path, arcname)], [skipped])`` for the active memory provider's external state.
Provider state (e.g. ~/.honcho, ~/.hindsight) lives outside HERMES_HOME, so the backup walk
never sees it; it is staged under the reserved ``_external/`` arc prefix, encoded relative to
the user's home dir. Only paths under home are captured (security + portability); anything
else is returned as skipped so the caller can note it.
"""
home_dir = Path.home().resolve()
external_to_add: list[tuple[Path, str]] = []
skipped_external: list[str] = []
for base in _collect_memory_provider_external_paths():
try:
base.resolve().relative_to(home_dir)
except (ValueError, OSError):
skipped_external.append(str(base))
continue
for fpath in _iter_external_files(base):
try:
rel_to_home = fpath.resolve().relative_to(home_dir)
except (ValueError, OSError):
continue
external_to_add.append((fpath, _EXTERNAL_PREFIX + rel_to_home.as_posix()))
return external_to_add, skipped_external
def run_backup(args) -> None:
"""Create a zip backup of the Hermes home directory."""
hermes_root = get_default_hermes_root()
if not hermes_root.is_dir():
print(f"Error: Hermes home directory not found at {hermes_root}")
sys.exit(1)
try:
with _backup_operation_lock(hermes_root):
_run_backup_locked(args, hermes_root)
except BackupInProgressError as exc:
print(f"Error: {exc}")
raise SystemExit(2) from exc
def _run_backup_locked(args, hermes_root: Path) -> None:
"""Write a full backup while the cross-process backup slot is held."""
out_path = _resolve_backup_output_path(args.output)
# Collect files
scan_started = time.monotonic()
logger.info("backup phase=scan status=started")
print(f"Scanning {display_hermes_home()} ...")
skipped_dirs: set = set()
files_to_add: list[tuple[Path, Path]] = list(_iter_backup_files(hermes_root, out_path, skipped_dirs))
external_to_add, skipped_external = _collect_external_entries()
if not files_to_add and not external_to_add:
logger.info(
"backup phase=scan status=empty duration_ms=%.1f",
(time.monotonic() - scan_started) * 1000,
)
print("No files to back up.")
return
# Create the zip
file_count = len(files_to_add) + len(external_to_add)
logger.info(
"backup phase=scan status=complete duration_ms=%.1f files=%d",
(time.monotonic() - scan_started) * 1000, file_count,
)
logger.info("backup phase=archive status=started files=%d", file_count)
print(f"Backing up {file_count} files ...")
errors = []
t0 = time.monotonic()
def _progress(i: int) -> None:
print(f" {i}/{file_count} files ...")
logger.info("backup phase=archive status=progress completed=%d total=%d", i, file_count)
with _atomic_output_path(out_path) as archive_path, zipfile.ZipFile(
archive_path, "w", zipfile.ZIP_DEFLATED, compresslevel=6
) as zf:
total_bytes = _write_zip_entries(
zf, files_to_add, out_path,
on_db_failure=lambda rel: errors.append(f"{rel}: SQLite safe copy failed"),
on_error=lambda rel, exc: errors.append(f"{rel}: {exc}"),
on_progress=_progress,
track_bytes=True,
)
# External memory-provider state, stored under the ``_external/`` arc
# prefix. These never include ``.db`` files in practice (config/env
# blobs), so a straight zf.write is fine.
for abs_path, arcname in external_to_add:
try:
zf.write(abs_path, arcname=arcname)
total_bytes += abs_path.stat().st_size
except (PermissionError, OSError, ValueError) as exc:
errors.append(f"{arcname}: {exc}")
continue
elapsed = time.monotonic() - t0
zip_size = out_path.stat().st_size
logger.info(
"backup phase=archive status=complete duration_ms=%.1f files=%d errors=%d bytes=%d",
elapsed * 1000, file_count, len(errors), zip_size,
)
# Summary
print()
print(f"Backup {'incomplete' if errors else 'complete'}: {out_path}")
print(f" Files: {file_count}")
print(f" Original: {_format_size(total_bytes)}")
print(f" Compressed: {_format_size(zip_size)}")
print(f" Time: {elapsed:.1f}s")
if external_to_add:
print(
f"\n Included {len(external_to_add)} memory-provider file(s) "
f"stored outside {display_hermes_home()}."
)
if skipped_external:
print(
f"\n Skipped {len(skipped_external)} memory-provider path(s) "
f"outside your home directory (not portable):"
)
print("\n".join(f" {p}" for p in sorted(skipped_external)[:10]))
if skipped_dirs:
print("\n Excluded directories:")
print("\n".join(f" {d}/" for d in sorted(skipped_dirs)))
if errors:
_print_skipped_warnings(errors)
else:
print(f"\nRestore with: hermes import {out_path.name}")
# ---------------------------------------------------------------------------
# Import
# ---------------------------------------------------------------------------
def _validate_backup_zip(zf: zipfile.ZipFile) -> tuple[bool, str]:
"""Check that a zip looks like a Hermes backup."""
names = zf.namelist()
if not names:
return False, "zip archive is empty"
# Telltale files a hermes home has — at the root or one level deep
# (if someone zipped the directory).
if not any(Path(n).name in {"config.yaml", ".env", "state.db"} for n in names):
return False, (
"zip does not appear to be a Hermes backup "
"(no config.yaml, .env, or state databases found)"
)
return True, ""
def _detect_prefix(zf: zipfile.ZipFile) -> str:
"""Detect if the zip has a common directory prefix wrapping all entries."""
names = [n for n in zf.namelist() if not n.endswith("/")]
if not names:
return ""
# All entries share one first directory that looks like a hermes dir name.
first_parts = {Path(n).parts[0] for n in names if len(Path(n).parts) > 1}
if len(first_parts) == 1 and first_parts <= {".hermes", "hermes"}:
return first_parts.pop() + "/"
return ""
def _default_new_file_mode() -> Optional[int]:
"""Return the mode ``open(path, "wb")`` gives a file it has to create.
``tempfile.mkstemp`` always creates at 0600, so staging an import through a temp file would
tighten every *newly created* file to owner-only — the same hazard ``utils._restore_file_mode``
documents for Docker/NAS volume mounts that rely on broader permissions.
"""
try:
current = os.umask(0o077)
os.umask(current)
except OSError:
return None
return 0o666 & ~current
def _extract_member_atomically(
zf: zipfile.ZipFile,
member: str,
target: Path,
new_file_mode: Optional[int] = None,
) -> None:
"""Restore one zip member onto *target* with no truncation window.
``open(target, "wb")`` truncates the user's existing file to zero *before* any replacement bytes
exist.
``atomic_replace`` rather than a bare ``os.replace``: it resolves a symlinked target first, so a
deployment that links ``config.yaml`` into a dotfiles repo keeps the link instead of having it
silently swapped for a regular file (GitHub #16743), and it falls back to copy/fsync/unlink on
``EXDEV``/``EBUSY`` for cross-device and bind-mount installs.
"""
# ``_preserve_file_mode`` returns None when the target does not exist (or
# cannot be stat'd), in which case the umask-derived create-mode applies —
# the same shape as ``atomic_yaml_write``'s ``create_mode`` fallback.
mode = _preserve_file_mode(target)
owner = _preserve_file_owner(target)
if mode is None:
mode = new_file_mode
else:
# Deliberately NOT a faithful mode copy: setuid/setgid are dropped.
# ``_preserve_file_mode`` returns ``stat.S_IMODE``, i.e. all twelve
# bits, and the content replacing this file comes from the archive.
# Carrying the elevated bits across would let archive-controlled bytes
# take over an existing setuid/setgid file, so ``hermes import`` would
# hand whoever produced the zip the identity that file runs as. Nothing
# constrains that to Hermes' own state either: the ``_external/`` branch
# of ``run_import`` publishes members anywhere under ``$HOME``. The
# sticky bit is kept — it is inert on a regular file.
mode &= ~(stat.S_ISUID | stat.S_ISGID)
# Truncate the stem: mkstemp adds ~16 characters, and a member already near
# NAME_MAX would otherwise fail here on a write that used to succeed.
fd, tmp_name = tempfile.mkstemp(
dir=str(target.parent), prefix=f".{target.name[:80]}.", suffix=".partial"
)
try:
with os.fdopen(fd, "wb") as dst:
if mode is not None:
# Apply the mode to the temp file BEFORE the replace so the
# target never transits through mkstemp's 0600, and so
# ``atomic_replace``'s EXDEV/EBUSY ``shutil.copystat`` fallback
# copies the intended bits rather than 0600. fchmod is
# Unix-only; Windows takes the path-based chmod.
if hasattr(os, "fchmod"):
os.fchmod(dst.fileno(), mode)
else:
os.chmod(tmp_name, mode)
# Stream instead of ``src.read()``: a multi-gigabyte state.db member
# must not be held in memory in one piece.
with zf.open(member) as src:
shutil.copyfileobj(src, dst)
dst.flush()
os.fsync(dst.fileno())
real_path = Path(atomic_replace(tmp_name, target))
# Owner first, mode second — the ordering ``atomic_yaml_write`` uses,
# because chown drops setuid/setgid and a mode restore that ran first
# would be partly undone. Here ``mode`` no longer carries those bits,
# so the two agree: neither step can re-elevate the restored file.
_restore_file_owner(real_path, owner)
_restore_file_mode(real_path, mode)
except BaseException:
with suppress(OSError):
os.unlink(tmp_name)
raise
def _confirm_import_overwrite(hermes_root: Path) -> bool:
"""Prompt before importing over an existing installation; True when import may proceed."""
if not any((hermes_root / m).exists() for m in ("config.yaml", ".env")):
return True
print()
print("Warning: Target directory already has Hermes configuration.")
print("Importing will overwrite existing files with backup contents.")
print()
try:
answer = input("Continue? [y/N] ").strip().lower()
except (EOFError, KeyboardInterrupt):
print("\nAborted.")
sys.exit(1)
if answer not in {"y", "yes"}:
print("Aborted.")
return False
return True
def run_import(args) -> None:
"""Restore a Hermes backup from a zip file."""
zip_path = Path(args.zipfile).expanduser().resolve()
if not zip_path.is_file():
print(f"Error: File not found: {zip_path}")
sys.exit(1)
if not zipfile.is_zipfile(zip_path):
print(f"Error: Not a valid zip file: {zip_path}")
sys.exit(1)
# The restore target must be the home the command operates under — the
# same path printed as "Target:" via display_hermes_home(). Resolving
# through get_default_hermes_root() instead maps a profile home
# (<root>/profiles/<name>) back to <root>, silently retargeting the
# restore at the live root while the profile directory stays empty.
hermes_root = get_hermes_home()
with zipfile.ZipFile(zip_path, "r") as zf:
# Validate
ok, reason = _validate_backup_zip(zf)
if not ok:
print(f"Error: {reason}")
sys.exit(1)
prefix = _detect_prefix(zf)
members = [n for n in zf.namelist() if not n.endswith("/")]
file_count = len(members)
print(f"Backup contains {file_count} files")
print(f"Target: {display_hermes_home()}")
if prefix:
print(f"Detected archive prefix: {prefix!r} (will be stripped)")
if not args.force and not _confirm_import_overwrite(hermes_root):
return
# Extract
print(f"\nImporting {file_count} files ...")
hermes_root.mkdir(parents=True, exist_ok=True)
errors = []
restored = 0
restored_external = 0
skipped_runtime: list[str] = []
home_dir = Path.home().resolve()
# Resolved once: every member is published via a temp file, and mkstemp
# would otherwise create newly restored files as 0600.
new_file_mode = _default_new_file_mode()
t0 = time.monotonic()
def _restore_member(
member: str, rel: str, target: Path, root: Path, tighten: bool, *, strict_chmod: bool
) -> bool:
"""Publish one member under *root*; False when blocked or failed (recorded in errors)."""
# Security: reject absolute paths and traversals
try:
target.resolve().relative_to(root)
except ValueError:
errors.append(f"{rel}: path traversal blocked")
return False
try:
target.parent.mkdir(parents=True, exist_ok=True)
_extract_member_atomically(zf, member, target, new_file_mode)
if tighten:
try:
os.chmod(target, 0o600)
except OSError:
if strict_chmod:
raise
except (PermissionError, OSError) as exc:
errors.append(f"{rel}: {exc}")
return False
return True
for member in members:
# External memory-provider state captured under the reserved
# ``_external/`` arc prefix restores to its original home-relative
# location (e.g. ~/.honcho/config.json), NOT under HERMES_HOME.
# Provider configs commonly hold credentials, so they are tightened
# to 0600 best-effort.
external = member.startswith(_EXTERNAL_PREFIX)
if external:
rel = member[len(_EXTERNAL_PREFIX):]
target = home_dir / rel
root = home_dir
tighten = target.suffix in {".json", ".env", ".conf"} or target.name in _SECRET_FILE_NAMES
else:
# Strip prefix if detected
rel = member[len(prefix):] if prefix and member.startswith(prefix) else member
# Never overwrite volatile gateway/process runtime state. These are
# namespaced to the machine/container the backup was taken on;
# clobbering them (especially gateway_state.json) breaks the gateway
# reconciler on the target and disconnects hosted instances from the
# Nous portal. Matched by basename so both the root profile and
# named profiles (profiles/<name>/gateway_state.json) are covered.
if rel and Path(rel).name in _IMPORT_SKIP_NAMES:
skipped_runtime.append(rel)
continue
target = hermes_root / rel
root = hermes_root.resolve()
tighten = target.name in _SECRET_FILE_NAMES
if not rel:
continue
if _restore_member(
member, member if external else rel, target, root, tighten, strict_chmod=not external
):
restored += 1
restored_external += external
if restored % 500 == 0:
print(f" {restored}/{file_count} files ...")
elapsed = time.monotonic() - t0
# Summary
print()
print(f"Import complete: {restored} files restored in {elapsed:.1f}s")
print(f" Target: {display_hermes_home()}")
if restored_external:
print(
f"\n Restored {restored_external} memory-provider file(s) to "
f"their original location(s) outside {display_hermes_home()}."
)
if errors:
_print_skipped_warnings(errors)
if skipped_runtime:
_print_capped(
f"\n Preserved {len(skipped_runtime)} runtime state "
f"file(s) (kept this machine's, not the backup's):",
sorted(skipped_runtime),
" ",
)
restored_profiles = _restore_profile_wrappers(hermes_root)
# Guidance
print()
if not (hermes_root / "hermes-agent").is_dir():
print("Note: The hermes-agent codebase was not included in the backup.")
print(" If this is a fresh install, run: hermes update")
if restored_profiles:
print("\nTo re-enable gateway services for profiles:")
for pname in restored_profiles:
print(f" hermes -p {pname} gateway install")
_revive_gateway_after_import(hermes_root)
print("Done. Your Hermes configuration has been restored.")
def _restore_profile_wrappers(hermes_root: Path) -> List[str]:
"""Re-create shell wrapper scripts for restored named profiles; return the profile names seen."""
profiles_dir = hermes_root / "profiles"
restored_profiles: list[tuple[str, bool]] = []
if not profiles_dir.is_dir():
return []
try:
from hermes_cli.profiles import (
create_wrapper_script, check_alias_collision,
_is_wrapper_dir_in_path, _get_wrapper_dir,
)
for entry in sorted(profiles_dir.iterdir()):
# Only create wrappers for directories with config
if not entry.is_dir() or not any((entry / m).exists() for m in ("config.yaml", ".env")):
continue
profile_name = entry.name
collision = check_alias_collision(profile_name)
if collision:
print(f" Skipped alias '{profile_name}': {collision}")
restored_profiles.append(
(profile_name, not collision and create_wrapper_script(profile_name) is not None)
)
if restored_profiles:
created = [n for n, ok in restored_profiles if ok]
skipped = [n for n, ok in restored_profiles if not ok]
if created:
print(f"\n Profile aliases restored: {', '.join(created)}")
if skipped:
print(f" Profile aliases skipped: {', '.join(skipped)}")
if not _is_wrapper_dir_in_path():
print(f"\n Note: {_get_wrapper_dir()} is not in your PATH.")
print(' Add to your shell config (~/.bashrc or ~/.zshrc):')
print(' export PATH="$HOME/.local/bin:$PATH"')
except ImportError:
# hermes_cli.profiles might not be available (fresh install)
if any(profiles_dir.iterdir()):
print("\n Profiles detected but aliases could not be created.")
print(" Run: hermes profile list (after installing hermes)")
return [n for n, _ in restored_profiles]
def _revive_gateway_after_import(hermes_root: Path) -> None:
"""Bring the restored install to life: install/start the gateway service, best-effort.
The backup may contain bot tokens and registered cron jobs, but they're inert without a
gateway process. A platform-less gateway is a supported mode, so this is safe even for backups
with no messaging config; prompt-free, and failures print a manual fallback, never fail the
import. A restore into a sandbox or profile home must not silently install a second gateway
pointed at it — on the default service name that would shadow or hijack the machine's primary
install — so the service is only revived when the restore landed in the default home, or when
no other install exists on this machine.
"""
native_default = _get_platform_default_hermes_home()
default_has_install = any(
(native_default / marker).exists() for marker in ("config.yaml", ".env", "state.db")
)
if hermes_root != native_default and default_has_install:
print(
"\nRestored into a non-default home; leaving the gateway service "
"alone to avoid clashing with the install at "
f"{native_default}."
)
print("To start a gateway for this home, run: hermes gateway install")
return
try:
from hermes_cli.gateway import ensure_gateway_service, _is_service_running
if not _is_service_running():
print()
ensure_gateway_service(context="import")
except Exception:
print("\nStart the gateway to activate cron jobs and messaging:")
print(" hermes gateway install")
# ---------------------------------------------------------------------------
# Quick state snapshots (used by /snapshot slash command and hermes backup --quick)
# ---------------------------------------------------------------------------
# Critical state files to include in quick snapshots (relative to HERMES_HOME).
# Everything else is either regeneratable (logs, cache) or managed separately
# (skills, repo, sessions/).
#
# Entries may be individual files OR directories. Directories are captured
# recursively; missing entries are silently skipped. Pairing data lives in
# platform-specific JSON blobs outside state.db, so it's listed here explicitly
# — `hermes update` snapshots this set before pulling so approved-user lists
# are recoverable if anything goes wrong (issue #15733).
_QUICK_STATE_FILES = (
"state.db",
"config.yaml",
".env",
"auth.json",
"cron/jobs.json",
"cron/executions.db",
"gateway_state.json",
"channel_directory.json",
"channel_aliases.json",
"processes.json",
"gateway/discord_message_recovery.db", # Discord reconnect replay ledger
# Per-profile user-created stores that live outside the git checkout and
# are therefore destroyed if the update flow removes/replaces the file and
# the post-update schema-init re-creates an empty one (issue #52889). All
# are at $HERMES_HOME/<name> for the default/root profile; on non-root
# profiles the real path is outside HERMES_HOME and the entry is silently
# skipped (best-effort, same as the pairing stores). SQLite DBs are copied
# WAL-safely via _safe_copy_db.
"projects.db", # per-profile project store
"response_store.db", # gateway conversation history / tool payloads
"memory_store.db", # holographic memory facts/entities
"verification_evidence.db", # agent verification audit trail
"kanban.db", # default board (back-compat <root>/kanban.db)
"kanban/boards", # non-default boards: each <slug>/kanban.db + board metadata (workspaces/ + attachments/ are skipped as regenerable)
# Pairing stores (generic + per-platform JSONs outside state.db)
"pairing", # legacy location (gateway/pairing.py)
"platforms/pairing", # new location (gateway/pairing.py)
"feishu_comment_pairing.json", # Feishu comment subscription pairings
)
# ``_QUICK_SNAPSHOTS_DIR`` lives with the exclusion rules at the top of the module.
_QUICK_DEFAULT_KEEP = 20
def _quick_snapshot_root(hermes_home: Optional[Path] = None) -> Path:
home = hermes_home or get_hermes_home()
return home / _QUICK_SNAPSHOTS_DIR
def create_quick_snapshot(
label: Optional[str] = None,
hermes_home: Optional[Path] = None,
keep: Optional[int] = None,
max_file_size: Optional[int] = None,
) -> Optional[str]:
"""Create one atomic quick snapshot while holding the shared backup slot."""
home = hermes_home or get_hermes_home()
with _backup_operation_lock(home):
return _create_quick_snapshot_locked(label, home, keep, max_file_size)
def _quick_snapshot_candidates(home: Path):
"""Yield ``(src, rel_posix, in_dir)`` for every regular file a quick snapshot captures.
Directory entries of ``_QUICK_STATE_FILES`` are walked so restore can treat every file
uniformly; empty dirs are skipped. Heavy, regenerable per-board subtrees (scratch workspaces
and task attachments) are skipped — only the board databases + metadata are needed.
"""
for rel in _QUICK_STATE_FILES:
src = home / rel
if not src.exists():
continue
if src.is_dir():
for sub in src.rglob("*"):
if not sub.is_file():
continue
sub_rel = sub.relative_to(home).as_posix()
if "/workspaces/" in f"/{sub_rel}/" or "/attachments/" in f"/{sub_rel}/":
continue
yield sub, sub_rel, True
elif src.is_file():
yield src, rel, False
def _copy_quick_snapshot_files(
home: Path, staging_dir: Path, max_file_size: Optional[int]
) -> tuple[Dict[str, int], list[str], list[str]]:
"""Copy every quick-snapshot candidate into *staging_dir*.
Returns ``(manifest, failed_dbs, oversized_skipped)``: ``manifest`` maps rel_path -> file size;
``failed_dbs`` lists present ``*.db`` that could not be snapshotted; ``oversized_skipped`` lists
protected DB files skipped for size (#68805) — those are snapshot incompleteness just like a
failed copy, so the caller must suppress pruning to preserve the older complete snapshot that
may contain the only recoverable database.
"""
manifest: Dict[str, int] = {}
failed_dbs: list[str] = []
oversized_skipped: list[str] = []
for src, rel, in_dir in _quick_snapshot_candidates(home):
if max_file_size is not None:
try:
size = src.stat().st_size
except OSError:
size = None
if size is not None and size > max_file_size:
print(
f" ⚠ Snapshot: skipping {rel} "
f"({_format_size(size)} exceeds {_format_size(max_file_size)} limit)"
)
logger.warning(
"Quick snapshot skipped %s: %d bytes exceeds %d byte limit", rel, size, max_file_size
)
if src.suffix == ".db":
oversized_skipped.append(rel)
continue
dst = staging_dir / rel
dst.parent.mkdir(parents=True, exist_ok=True)
try:
# Route SQLite DBs through the WAL-safe backup() path so a DB with
# an open WAL (the gateway may hold it at snapshot time) is
# captured consistently.
if src.suffix == ".db":
if not _safe_copy_db(src, dst):
failed_dbs.append(rel)
print(
f" ⚠ Snapshot: SQLite safe copy FAILED for {rel} "
f"— file may be locked or corrupted"
)
if is_zeroed_sqlite_file(src):
nuls = " of NULs?" if in_dir else ""
print(
f" ⚠ Snapshot: {rel} looks ZEROED "
f"(no SQLite header; {src.stat().st_size} bytes{nuls})"
)
continue
else:
shutil.copy2(src, dst)
manifest[rel] = dst.stat().st_size
except (OSError, PermissionError) as exc:
logger.warning("Could not snapshot %s: %s", rel, exc)
return manifest, failed_dbs, oversized_skipped
def _create_quick_snapshot_locked(
label: Optional[str], hermes_home: Optional[Path], keep: Optional[int], max_file_size: Optional[int]
) -> Optional[str]:
"""Create a quick state snapshot of critical files.
Copies STATE_FILES to a timestamped directory under state-snapshots/ and prunes old snapshots.
``max_file_size`` skips (with a warning) files above that many bytes; the pre-update snapshot
uses it so a multi-GB ``state.db`` can never stall ``hermes update`` while the small
pairing/cron/config files are always captured. ``None`` copies everything.
"""
home = hermes_home or get_hermes_home()
root = _quick_snapshot_root(home)
ts = datetime.now(timezone.utc).strftime("%Y%m%d-%H%M%S")
base_snap_id = f"{ts}-{label}" if label else ts
snap_id = base_snap_id
suffix = 2
while (root / snap_id).exists():
snap_id = f"{base_snap_id}-{suffix}"
suffix += 1
snap_dir = root / snap_id
staging_dir = root / f".{snap_id}.{os.getpid()}.partial"
shutil.rmtree(staging_dir, ignore_errors=True)
staging_dir.mkdir(parents=True, exist_ok=False)
logger.info("quick snapshot phase=copy status=started id=%s", snap_id)
manifest, failed_dbs, oversized_skipped = _copy_quick_snapshot_files(home, staging_dir, max_file_size)
if failed_dbs:
# Critical: update path used to log-and-continue with exit 0, so a
# missing state.db backup looked like a successful pre-update snapshot
# (#68474). Surface this on stdout where operators actually look.
print(" ⚠ CRITICAL: could not snapshot DB file(s): " + ", ".join(failed_dbs))
print(f" ⚠ If sessions disappear after update, check {root} and run: hermes snapshot list")
logger.error("Quick snapshot failed to capture DB file(s): %s", ", ".join(failed_dbs))
if not manifest:
shutil.rmtree(staging_dir, ignore_errors=True)
if failed_dbs:
# Distinguish "nothing to snapshot" from "state.db present but unreadable"
print(f" ⚠ Snapshot aborted: no files captured (failed DBs: {', '.join(failed_dbs)})")
return None
# Write manifest
meta = {
"id": snap_id,
"timestamp": ts,
"label": label,
"file_count": len(manifest),
"total_size": sum(manifest.values()),
"files": manifest,
"failed_dbs": failed_dbs,
"oversized_skipped": oversized_skipped,
}
with open(staging_dir / "manifest.json", "w", encoding="utf-8") as f:
json.dump(meta, f, indent=2)
os.replace(staging_dir, snap_dir)
# Auto-prune. Defaults preserve historical manual /snapshot behavior; callers
# with known high-churn safety snapshots (for example pre-update) can pass a
# smaller keep value so large state.db copies do not accumulate indefinitely.
# #68805 review: skip pruning when a present DB failed to capture OR was
# skipped for size — either way the snapshot is incomplete and the older
# snapshot may contain the only recoverable database.
if not (failed_dbs or oversized_skipped):
_prune_quick_snapshots(root, keep=_QUICK_DEFAULT_KEEP if keep is None else keep)
else:
if oversized_skipped:
print(
" ⚠ Skipping snapshot prune: DB file(s) skipped for size: "
+ ", ".join(oversized_skipped)
)
logger.warning("Quick snapshot skipped oversized DB file(s): %s", ", ".join(oversized_skipped))
logger.warning(
"Skipping snapshot prune because %d DB(s) failed to capture "
"and/or %d were oversized — preserving older snapshots as "
"recovery source",
len(failed_dbs), len(oversized_skipped),
)
logger.info(
"quick snapshot phase=copy status=complete id=%s files=%d bytes=%d",
snap_id, len(manifest), sum(manifest.values()),
)
return snap_id
def _snapshot_dirs(root: Path) -> List[Path]:
"""Published snapshot directories under *root*, newest (by name) first."""
if not root.exists():
return []
return sorted(
(d for d in root.iterdir()
if d.is_dir() and not d.name.startswith(".") and not d.name.endswith(".partial")),
key=lambda d: d.name,
reverse=True,
)
def list_quick_snapshots(limit: int = 20, hermes_home: Optional[Path] = None) -> List[Dict[str, Any]]:
"""List existing quick state snapshots, most recent first."""
results = []
for d in _snapshot_dirs(_quick_snapshot_root(hermes_home)):
manifest_path = d / "manifest.json"
if manifest_path.exists():
try:
with open(manifest_path, encoding="utf-8") as f:
results.append(json.load(f))
except (json.JSONDecodeError, OSError):
results.append({"id": d.name, "file_count": 0, "total_size": 0})
if len(results) >= limit:
break
return results
def restore_quick_snapshot(snapshot_id: str, hermes_home: Optional[Path] = None) -> bool:
"""Restore state from a quick snapshot."""
home = hermes_home or get_hermes_home()
root = _quick_snapshot_root(home)
# Security: reject snapshot_id values that contain path separators or
# traversal sequences so that `root / snapshot_id` stays inside root.
if not snapshot_id or "/" in snapshot_id or "\\" in snapshot_id or snapshot_id in (".", ".."):
logger.error("Invalid snapshot_id: %s", snapshot_id)
return False
snap_dir = root / snapshot_id
# Confirm the resolved path is still inside root (handles symlinks etc.)
try:
snap_dir.resolve().relative_to(root.resolve())
except ValueError:
logger.error("Snapshot path traversal blocked for id: %s", snapshot_id)
return False
manifest_path = snap_dir / "manifest.json"
if not snap_dir.is_dir() or not manifest_path.exists():
return False
with open(manifest_path, encoding="utf-8") as f:
meta = json.load(f)
restored = 0
for rel in meta.get("files", {}):
# Security: reject absolute paths and traversals in manifest entries
src = snap_dir / rel
dst = home / rel
try:
src.resolve().relative_to(snap_dir.resolve())
dst.resolve().relative_to(home.resolve())
except ValueError:
logger.error("Manifest path traversal blocked: %s", rel)
continue
if not src.exists():
continue
dst.parent.mkdir(parents=True, exist_ok=True)
try:
if dst.suffix == ".db":
# Restore through SQLite backup API so live connections
# (gateway, dashboard, another CLI session) see the
# restored data instead of continuing to serve stale
# cached pages from a replaced inode (issue #65942).
_safe_restore_db(src, dst)
else:
shutil.copy2(src, dst)
restored += 1
except (OSError, PermissionError) as exc:
logger.error("Failed to restore %s: %s", rel, exc)
logger.info("Restored %d files from snapshot %s", restored, snapshot_id)
return restored > 0
# Relative path of the cron job database inside HERMES_HOME. Kept in sync with
# the entry in ``_QUICK_STATE_FILES`` and with ``cron/jobs.py``'s ``JOBS_FILE``.
_CRON_JOBS_REL = "cron/jobs.json"
def _count_cron_jobs(path: Path) -> Optional[int]:
"""Return the number of cron jobs stored in ``path``.
Accepts the canonical ``{"jobs": [...]}`` shape and the legacy bare list. Returns ``None`` if
the file is missing or unparseable; callers must treat ``None`` as "unknown", not zero,
since acting on an unreadable file could mask a real corruption the user needs to see.
"""
if not path.is_file():
return None
try:
# utf-8-sig: same dialect as cron/jobs.load_jobs — Windows editors
# may leave a UTF-8 BOM that plain utf-8 json.load rejects. Without
# it a BOM'd jobs.json counts as "unreadable" (None) and the
# post-update cron-loss auto-restore safety net silently disables.
with open(path, "r", encoding="utf-8-sig") as f:
data = json.load(f)
except (OSError, json.JSONDecodeError):
return None
if isinstance(data, dict):
data = data.get("jobs", [])
return len(data) if isinstance(data, list) else None
def restore_cron_jobs_if_emptied(
snapshot_id: str,
hermes_home: Optional[Path] = None,
) -> Optional[Dict[str, Any]]:
"""Safety net for silent cron-job loss across ``hermes update``.
The check is deliberately conservative — it only ever restores when there is unambiguous
evidence of loss (snapshot had more jobs than live file), so a user who genuinely deleted jobs
during/after the update is never second-guessed, and an unreadable live file (count ``None``) is
left untouched so real corruption still surfaces.
"""
if not snapshot_id:
return None
home = hermes_home or get_hermes_home()
live_path = home / _CRON_JOBS_REL
live_count = _count_cron_jobs(live_path)
# ``None`` (missing or unparseable) is intentionally left alone — that's a
# different failure mode the user should see rather than have papered over.
if live_count is None:
return None
snap_path = _quick_snapshot_root(home) / snapshot_id / _CRON_JOBS_REL
snap_count = _count_cron_jobs(snap_path)
if not snap_count: # None or 0 — nothing worth restoring
return None
# Restore when live has FEWER jobs than the pre-update snapshot.
# Catches both total loss (0 vs N) and partial loss (1 vs 19) — the
# desktop scheduler can overwrite jobs.json with its own small set of
# internally-tracked crons after an update/restart.
if live_count >= snap_count:
return None
try:
live_path.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(snap_path, live_path)
except (OSError, PermissionError) as exc:
logger.error("Cron jobs were emptied during update but auto-restore failed: %s", exc)
return None
logger.warning(
"Restored %d cron job(s) from pre-update snapshot %s "
"(live file had %d job(s), snapshot had %d — jobs were lost during migration)",
snap_count, snapshot_id, live_count, snap_count,
)
return {"restored": True, "job_count": snap_count, "snapshot_id": snapshot_id}
def _sibling_profile_homes(invoking_home: Path) -> list[tuple[str, Path]]:
"""(name, home) for every OTHER profile on this install. Never raises.
The update's code swap and gateway fleet restart touch every profile, so the pre-update snapshot
must too (#66140). The invoking profile is excluded — its snapshot is taken by the existing
call.
"""
homes: list[tuple[str, Path]] = []
try:
from hermes_cli.profiles import (
_get_default_hermes_home,
_get_profiles_root,
_PROFILE_ID_RE,
)
invoking = invoking_home.resolve()
default_home = _get_default_hermes_home()
if default_home.is_dir() and default_home.resolve() != invoking:
homes.append(("default", default_home))
root = _get_profiles_root()
if root.is_dir():
for entry in sorted(root.iterdir()):
if (
entry.is_dir()
and entry.name != "default"
and _PROFILE_ID_RE.match(entry.name)
and entry.resolve() != invoking
):
homes.append((entry.name, entry))
except Exception as exc:
logger.debug("Sibling profile enumeration failed: %s", exc)
return homes
def create_pre_update_snapshots_all_profiles(
invoking_home: Optional[Path] = None,
keep: Optional[int] = None,
max_file_size: Optional[int] = None,
) -> Dict[str, str]:
"""Pre-update quick snapshots for every SIBLING profile (#66140).
Same snapshot set, same per-file size cap, same keep policy as the invoking profile's snapshot —
identical semantics per profile, no partial-tier coherence class. Each sibling's snapshot lands
under its OWN ``<home>/state-snapshots/`` so per-profile restore tooling finds it where it
expects.
"""
results: Dict[str, str] = {}
home = invoking_home or get_hermes_home()
for name, profile_home in _sibling_profile_homes(home):
try:
snap_id = create_quick_snapshot(
label="pre-update", hermes_home=profile_home, keep=keep, max_file_size=max_file_size
)
if snap_id:
results[name] = snap_id
except Exception as exc:
logger.debug("Pre-update snapshot for profile %s failed: %s", name, exc)
return results
# Config paths that the update flow must never change (#64160): the model
# routing keys and the Mixture-of-Agents section are consumed machine-wide
# (gateway, cron, desktop), so an update/repair cycle that rewrites them
# silently redirects paid inference. Each entry is a dotted path into the raw
# config.yaml document; a single-element tuple protects the whole section.
_PROTECTED_CONFIG_PATHS: Tuple[Tuple[str, ...], ...] = (
("model", "provider"),
("model", "default"),
("model", "base_url"),
("model", "api_key"),
("moa",),
)
def _read_raw_yaml_dict(path: Path) -> Optional[Dict[str, Any]]:
"""Parse ``path`` as a YAML mapping. ``None`` = missing/unreadable/non-dict."""
if not path.is_file():
return None
try:
import yaml
with open(path, "r", encoding="utf-8") as f:
data = yaml.safe_load(f)
except Exception:
return None
return data if isinstance(data, dict) else None
def _get_config_path_value(data: Dict[str, Any], dotted: Tuple[str, ...]) -> Any:
node: Any = data
for key in dotted:
if not isinstance(node, dict):
return None
node = node.get(key)
return node
def _set_config_path_value(data: Dict[str, Any], dotted: Tuple[str, ...], value: Any) -> None:
node = data
for key in dotted[:-1]:
child = node.get(key)
if not isinstance(child, dict):
child = {}
node[key] = child
node = child
node[dotted[-1]] = value
def restore_config_model_settings_if_rewritten(
snapshot_id: str,
hermes_home: Optional[Path] = None,
) -> Optional[Dict[str, Any]]:
"""Safety net for silent config.yaml model/MoA loss across ``hermes update``.
These keys are consumed by the gateway and unattended cron jobs too, so a rewrite silently
changes paid inference behavior machine-wide.
Mirrors :func:`restore_cron_jobs_if_emptied`: compare the *current* config against the pre-
update snapshot taken minutes earlier by this same update run, and restore only the protected
keys — never the whole file — when a value the user had set was changed or dropped.
"""
if not snapshot_id:
return None
home = hermes_home or get_hermes_home()
live_path = home / "config.yaml"
snap_path = _quick_snapshot_root(home) / snapshot_id / "config.yaml"
snap = _read_raw_yaml_dict(snap_path)
if not snap:
return None # no snapshot copy — nothing to compare against
live = _read_raw_yaml_dict(live_path)
if live is None:
# Missing or unparseable live config is a different failure mode the
# user should see rather than have papered over (matches the cron net).
return None
restored_keys: list[str] = []
for dotted in _PROTECTED_CONFIG_PATHS:
snap_val = _get_config_path_value(snap, dotted)
if snap_val in (None, "", {}, []):
continue # user never set it — nothing to protect
live_val = _get_config_path_value(live, dotted)
if live_val == snap_val:
continue
_set_config_path_value(live, dotted, snap_val)
restored_keys.append(".".join(dotted))
if not restored_keys:
return None
try:
from utils import atomic_yaml_write
atomic_yaml_write(live_path, live)
except (OSError, PermissionError) as exc:
logger.error(
"config.yaml model settings were rewritten during update but "
"auto-restore failed: %s",
exc,
)
return None
logger.warning(
"Restored user config value(s) %s from pre-update snapshot %s — "
"the update flow rewrote them (#64160)",
", ".join(restored_keys),
snapshot_id,
)
return {"restored": True, "keys": restored_keys, "snapshot_id": snapshot_id}
def _restore_all_sibling_profiles(
profile_snapshots: Dict[str, str],
invoking_home: Optional[Path],
restore_fn,
failure_log: str,
) -> list[Dict[str, Any]]:
"""Run a per-profile safety net (``restore_fn(snap_id, hermes_home=...)``) for every sibling.
Each profile's live file is compared against ITS OWN same-generation pre-update snapshot.
Returns one result dict per restored profile, each with a ``profile`` key added. Never raises.
"""
restored: list[Dict[str, Any]] = []
if not profile_snapshots:
return restored
home = invoking_home or get_hermes_home()
by_name = dict(_sibling_profile_homes(home))
for name, snap_id in profile_snapshots.items():
profile_home = by_name.get(name)
if profile_home is None:
continue
try:
result = restore_fn(snap_id, hermes_home=profile_home)
except Exception as exc:
logger.debug(failure_log, name, exc)
continue
if result:
result["profile"] = name
restored.append(result)
return restored
def restore_config_model_settings_all_profiles(
profile_snapshots: Dict[str, str],
invoking_home: Optional[Path] = None,
) -> list[Dict[str, Any]]:
"""Run the config model-settings safety net for every sibling profile.
Same contract as :func:`restore_cron_jobs_all_profiles`: each profile's live ``config.yaml`` is
compared against ITS OWN same-generation pre-update snapshot. Returns one result dict per
restored profile, each with a ``profile`` key added. Never raises.
"""
return _restore_all_sibling_profiles(
profile_snapshots,
invoking_home,
restore_config_model_settings_if_rewritten,
"Config model-settings restore check for profile %s failed: %s",
)
def restore_cron_jobs_all_profiles(
profile_snapshots: Dict[str, str],
invoking_home: Optional[Path] = None,
) -> list[Dict[str, Any]]:
"""Run the cron-jobs safety net for every sibling profile (#66140).
``profile_snapshots`` comes from :func:`create_pre_update_snapshots_all_profiles`; each
profile's live ``cron/jobs.json`` is compared against ITS OWN snapshot, so restores are
same-generation by construction. Returns one result dict per restored profile. Never raises.
"""
return _restore_all_sibling_profiles(
profile_snapshots,
invoking_home,
restore_cron_jobs_if_emptied,
"Cron restore check for profile %s failed: %s",
)
def _prune_oldest(newest_first: List[Path], keep: int, remove, what: str) -> int:
"""``remove(path)`` every entry past the first *keep*; return how many succeeded."""
deleted = 0
for p in newest_first[keep:]:
try:
remove(p)
deleted += 1
except OSError as exc:
logger.warning("Failed to prune %s %s: %s", what, p.name, exc)
return deleted
def _prune_quick_snapshots(root: Path, keep: int = _QUICK_DEFAULT_KEEP) -> int:
"""Remove oldest quick snapshots beyond the keep limit. Returns count deleted."""
return _prune_oldest(_snapshot_dirs(root), keep, shutil.rmtree, "snapshot")
def prune_quick_snapshots(keep: int = _QUICK_DEFAULT_KEEP, hermes_home: Optional[Path] = None) -> int:
"""Manually prune quick snapshots. Returns count deleted."""
return _prune_quick_snapshots(_quick_snapshot_root(hermes_home), keep=keep)
def run_quick_backup(args) -> None:
"""CLI entry point for hermes backup --quick."""
label = getattr(args, "label", None)
snap_id = create_quick_snapshot(label=label)
if snap_id:
print(f"State snapshot created: {snap_id}")
snaps = list_quick_snapshots()
print(f" {len(snaps)} snapshot(s) stored in {display_hermes_home()}/state-snapshots/")
print(f" Restore with: /snapshot restore {snap_id}")
else:
print("No state files found to snapshot.")
# ---------------------------------------------------------------------------
# Shared full-zip backup helper
# ---------------------------------------------------------------------------
def _write_full_zip_backup(out_path: Path, hermes_root: Path) -> Optional[Path]:
"""Write a full zip snapshot of ``hermes_root`` to ``out_path`` while holding the backup slot.
Uses the same exclusion rules and SQLite safe-copy as :func:`run_backup`. Returns the output
path on success, None on failure (nothing to back up, another backup running, or write error —
caller should surface the outcome but not raise).
"""
try:
with _backup_operation_lock(hermes_root):
return _write_full_zip_backup_locked(out_path, hermes_root)
except BackupInProgressError as exc:
logger.warning("Full-zip backup skipped: %s", exc)
return None
def _write_full_zip_backup_locked(out_path: Path, hermes_root: Path) -> Optional[Path]:
scan_started = time.monotonic()
logger.info("automatic backup phase=scan status=started")
try:
files_to_add = list(_iter_backup_files(hermes_root, out_path))
except OSError as exc:
logger.warning("Full-zip backup: walk failed: %s", exc)
return None
if not files_to_add:
return None
logger.info(
"automatic backup phase=scan status=complete duration_ms=%.1f files=%d",
(time.monotonic() - scan_started) * 1000,
len(files_to_add),
)
archive_started = time.monotonic()
def _db_failure(rel_path: Path) -> None:
logger.warning("Full-zip backup aborted: SQLite snapshot failed for %s", rel_path)
raise _SQLiteSnapshotError(str(rel_path))
try:
with _atomic_output_path(out_path) as archive_path, zipfile.ZipFile(
archive_path, "w", zipfile.ZIP_DEFLATED, compresslevel=6
) as zf:
_write_zip_entries(
zf, files_to_add, out_path,
on_db_failure=_db_failure,
on_error=lambda rel, exc: logger.debug("Skipping %s in zip backup: %s", rel, exc),
on_progress=lambda i: logger.info(
"automatic backup phase=archive status=progress completed=%d total=%d",
i, len(files_to_add),
),
track_bytes=False,
)
except (OSError, _SQLiteSnapshotError) as exc:
logger.warning("Full-zip backup: zip write failed: %s", exc)
# ``_atomic_output_path`` already removed the hidden partial. Do not
# unlink ``out_path`` here: it may be a previous valid backup that the
# atomic publisher deliberately preserved.
return None
logger.info(
"automatic backup phase=archive status=complete duration_ms=%.1f files=%d bytes=%d",
(time.monotonic() - archive_started) * 1000,
len(files_to_add),
out_path.stat().st_size,
)
return out_path
# ---------------------------------------------------------------------------
# Pre-update auto-backup
# ---------------------------------------------------------------------------
_PRE_UPDATE_BACKUPS_DIR = "backups"
_PRE_UPDATE_PREFIX = "pre-update-"
_PRE_UPDATE_DEFAULT_KEEP = 5
def _prune_prefixed_zips(backup_dir: Path, prefix: str, keep: int, what: str) -> int:
"""Remove oldest ``<prefix>*.zip`` files in *backup_dir* beyond the keep limit.
Returns the number of files deleted. Only touches files matching the prefix so hand-made zips
or other backup kinds dropped in the same directory are never touched.
Operators who genuinely don't want a backup should set ``updates.pre_update_backup: off`` in
config — that gates creation.
"""
if not backup_dir.exists():
return 0
backups = sorted(
(p for p in backup_dir.iterdir()
if p.is_file() and p.name.startswith(prefix) and p.suffix.lower() == ".zip"),
key=lambda p: p.name,
reverse=True,
)
return _prune_oldest(backups, keep, Path.unlink, what)
def _create_prefixed_full_backup(
hermes_home: Optional[Path], prefix: str, keep: int, what: str, prune_what: str
) -> Optional[Path]:
"""Write ``<HERMES_HOME>/backups/<prefix><timestamp>.zip`` and prune older same-prefix zips.
Returns the created path, or ``None`` if nothing was found to back up or the write failed.
Never raises.
"""
hermes_root = hermes_home or get_default_hermes_root()
if not hermes_root.is_dir():
return None
backup_dir = hermes_root / _PRE_UPDATE_BACKUPS_DIR
try:
backup_dir.mkdir(parents=True, exist_ok=True)
except OSError as exc:
logger.warning("Could not create %s backup dir %s: %s", what, backup_dir, exc)
return None
stamp = datetime.now().strftime("%Y-%m-%d-%H%M%S")
out_path = backup_dir / f"{prefix}{stamp}.zip"
if _write_full_zip_backup(out_path, hermes_root) is None:
return None
_prune_prefixed_zips(backup_dir, prefix, keep, prune_what)
return out_path
def create_pre_update_backup(
hermes_home: Optional[Path] = None,
keep: int = _PRE_UPDATE_DEFAULT_KEEP,
) -> Optional[Path]:
"""Create a full zip backup of HERMES_HOME under ``backups/``.
Mirrors :func:`run_backup` (same exclusion rules, same SQLite safe-copy) but writes to
``<HERMES_HOME>/backups/pre-update-<timestamp>.zip`` and auto-prunes old pre-update backups.
Returns the path to the created zip, or ``None`` if no files were found or the backup could not
be created. Never raises — the caller (``hermes update``) should continue even if the backup
fails.
"""
return _create_prefixed_full_backup(
hermes_home, _PRE_UPDATE_PREFIX, max(keep, 1), "pre-update", "backup"
)
# ---------------------------------------------------------------------------
# Pre-migration auto-backup (used by `hermes claw migrate`)
# ---------------------------------------------------------------------------
_PRE_MIGRATION_PREFIX = "pre-migration-"
_PRE_MIGRATION_DEFAULT_KEEP = 5
def create_pre_migration_backup(
hermes_home: Optional[Path] = None,
keep: int = _PRE_MIGRATION_DEFAULT_KEEP,
) -> Optional[Path]:
"""Create a full zip backup of HERMES_HOME under ``backups/`` before a ``hermes claw migrate`` apply.
Shares implementation with :func:`create_pre_update_backup` via ``_write_full_zip_backup`` —
same exclusions, same SQLite safe-copy, restorable with ``hermes import <archive>``. Writes to
``<HERMES_HOME>/backups/pre-migration-<timestamp>.zip`` (the shared ``backups/`` directory, so
``hermes import`` and the update-backup listing pick up pre-migration archives too) and
auto-prunes old pre-migration backups.
Returns the path to the created zip, or ``None`` if nothing was found to back up (fresh install)
or the write failed. Never raises — the caller decides whether to abort or proceed.
"""
return _create_prefixed_full_backup(
hermes_home, _PRE_MIGRATION_PREFIX, max(keep, 0), "pre-migration", "pre-migration backup"
)