2164 lines
87 KiB
Python
2164 lines
87 KiB
Python
"""Backup and import commands for hermes CLI."""
|
|
|
|
import json
|
|
import logging
|
|
import os
|
|
import shutil
|
|
import sqlite3
|
|
import stat
|
|
import sys
|
|
import tempfile
|
|
import threading
|
|
import time
|
|
import zipfile
|
|
from contextlib import contextmanager, suppress
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
from typing import Any, Dict, List, Optional, Tuple
|
|
|
|
from hermes_constants import (
|
|
_get_platform_default_hermes_home,
|
|
get_default_hermes_root,
|
|
get_hermes_home,
|
|
display_hermes_home,
|
|
)
|
|
from utils import (
|
|
_preserve_file_mode,
|
|
_preserve_file_owner,
|
|
_restore_file_mode,
|
|
_restore_file_owner,
|
|
atomic_replace,
|
|
)
|
|
|
|
# Shared formatter; the private alias is kept because claw.py and the backup
|
|
# tests import ``_format_size`` from this module.
|
|
from hermes_cli.sizefmt import format_bytes as _format_size
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Exclusion rules
|
|
# ---------------------------------------------------------------------------
|
|
|
|
# Where ``hermes backup --quick`` / ``/snapshot`` / the pre-update safety net
|
|
# write their state snapshots (see ``create_quick_snapshot`` below). Defined up
|
|
# here because the exclusion set needs it.
|
|
_QUICK_SNAPSHOTS_DIR = "state-snapshots"
|
|
|
|
# Directory names to skip entirely (matched against each path component)
|
|
# ``hermes-agent`` is special-cased to root level only in ``_should_exclude``
|
|
# so that skill directories like ``skills/autonomous-ai-agents/hermes-agent/``
|
|
# are not accidentally excluded.
|
|
#
|
|
# The dependency/cache entries below matter for more than tidiness: without
|
|
# them a single plugin venv, MCP-server install, or pip/uv cache living under
|
|
# HERMES_HOME gets walked file-by-file, ballooning a backup to hundreds of
|
|
# thousands of entries that crawl for hours — the exact "backup stuck for
|
|
# days / 426543 files" symptom users hit. The dependency/test-env names mostly
|
|
# mirror ``agent.skill_utils.EXCLUDED_SKILL_DIRS`` (the project's canonical
|
|
# "regeneratable dir" set); ``.cache`` is an additional backup-only entry, as
|
|
# it names a broad regeneratable cache convention (pip/uv/etc.) that the skill
|
|
# scanner doesn't need to prune but a backup walk does. We deliberately do NOT
|
|
# exclude ``.archive`` here because the curator's ``skills/.archive/`` holds
|
|
# restorable user skills that must survive a backup.
|
|
_EXCLUDED_DIRS = {
|
|
"hermes-agent", # the codebase repo — re-clone instead
|
|
"__pycache__", # bytecode caches — regenerated on import
|
|
".git", # nested git dirs (profiles shouldn't have these, but safety)
|
|
"node_modules", # js deps — reinstalled on demand
|
|
"backups", # prior auto-backups — don't nest backups exponentially
|
|
_QUICK_SNAPSHOTS_DIR, # quick/pre-update state snapshots — same reason as
|
|
# ``backups``: each holds a full copy of state.db, so
|
|
# zipping them re-ships the DB once per snapshot
|
|
"checkpoints", # session-local trajectory caches — regenerated per-session,
|
|
# session-hash-keyed so they don't port to another machine anyway
|
|
# Live browser profiles (e.g. the CDP Brave profile under browser-profiles/).
|
|
# Chromium holds its SQLite DBs with exclusive locks while running, and
|
|
# sqlite3.Connection.backup() retries SQLITE_BUSY forever instead of honoring
|
|
# the busy timeout — a full backup hangs mid-archive on the first locked DB.
|
|
# Profiles are regenerable (cache + re-login) and unsafe to snapshot live.
|
|
"browser-profiles",
|
|
# Real-profile browsing snapshot (browser.use_real_profile). Holds copies of
|
|
# the user's Cookies / Login Data / Web Data — a credential-bearing store
|
|
# that must NOT enter a backup archive. It is regenerated from the user's
|
|
# live profile on the next consented launch. Singular, distinct from the
|
|
# ``browser-profiles`` CDP dir above; both are excluded.
|
|
"browser-profile",
|
|
# Python dependency trees (plugin / MCP-server venvs under HERMES_HOME) —
|
|
# regenerated by reinstalling; never irreplaceable state.
|
|
".venv",
|
|
"venv",
|
|
"site-packages",
|
|
# Tool / build caches — all regeneratable.
|
|
".cache",
|
|
".tox",
|
|
".nox",
|
|
".pytest_cache",
|
|
".mypy_cache",
|
|
".ruff_cache",
|
|
}
|
|
|
|
# Hermes-managed runtime downloads that only exist at the top of a profile
|
|
# home: local GGUF models, llama.cpp runtime binaries, and the managed Node
|
|
# installation. All of them are re-downloaded on demand (model catalog,
|
|
# runtime bootstrap, node installer) and routinely reach tens to hundreds of
|
|
# GB, so zipping them turns a backup into an hours-long compress of
|
|
# incompressible weights (the "backup stuck at N files" symptom). Matched
|
|
# ONLY at the root of HERMES_HOME and at ``profiles/<name>/`` — a deeper
|
|
# directory that happens to share one of these names (a skill's ``models/``,
|
|
# a user checkout) is user data and stays in the backup.
|
|
_EXCLUDED_ROOT_DIRS = {"models", "runtimes", "node"}
|
|
|
|
|
|
def _in_excluded_root_dir(rel_path: Path) -> bool:
|
|
"""True when *rel_path* (relative to HERMES_HOME) is, or sits inside, a
|
|
Hermes-managed runtime tree at the top of a profile home."""
|
|
parts = rel_path.parts
|
|
# Named profiles are profile homes too: profiles/<name>/models etc.
|
|
return bool(parts) and (
|
|
parts[0] in _EXCLUDED_ROOT_DIRS
|
|
or (len(parts) >= 3 and parts[0] == "profiles" and parts[2] in _EXCLUDED_ROOT_DIRS)
|
|
)
|
|
|
|
|
|
# File-name suffixes to skip
|
|
_EXCLUDED_SUFFIXES = (
|
|
".pyc",
|
|
".pyo",
|
|
# SQLite sidecar files — the backup takes a consistent snapshot of ``*.db``
|
|
# via ``sqlite3.backup()``, so shipping the live WAL / shared-memory /
|
|
# rollback-journal alongside would pair a fresh snapshot with stale sidecar
|
|
# state and produce a torn restore on the next open. They're transient and
|
|
# regenerated on first connection anyway.
|
|
".db-wal",
|
|
".db-shm",
|
|
".db-journal",
|
|
)
|
|
|
|
# File names to skip (runtime state that's meaningless on another machine)
|
|
_EXCLUDED_NAMES = {".backup.lock", "gateway.pid", "cron.pid"}
|
|
|
|
# File-name prefixes to skip. The desktop updater's pre-flight drops
|
|
# ``state.db.pre-update-emergency-<timestamp>.bak`` at the HERMES_HOME root
|
|
# (apps/desktop/electron/main.ts preflightStateDb) — a backup artifact in
|
|
# the same class as ``backups/`` and ``state-snapshots/``, so a full backup
|
|
# must not re-ship it. Matched by prefix because the name carries a
|
|
# timestamp; a plain ``.bak`` suffix rule would drop user files.
|
|
_EXCLUDED_PREFIXES = (
|
|
"state.db.pre-update-emergency-",
|
|
)
|
|
|
|
# File names that ``hermes import`` must never overwrite, matched by basename so
|
|
# they're caught for the root profile (``gateway_state.json``) and for named
|
|
# profiles alike (``profiles/<name>/gateway_state.json``).
|
|
#
|
|
# These hold *volatile gateway/process runtime state that is namespaced to the
|
|
# machine or container the backup was taken on* — PIDs in a dead process
|
|
# namespace, a runtime lock, the process registry, and the gateway's last
|
|
# recorded run/desired state. Restoring them onto a different host (or a hosted
|
|
# container) is at best meaningless and at worst actively harmful:
|
|
#
|
|
# - ``gateway_state.json`` drives the container-boot reconciler
|
|
# (``container_boot._read_desired_state``), which only auto-starts a
|
|
# gateway whose recorded state is ``running``. A backup taken from a
|
|
# machine where the gateway was stopped (or carrying a stale/foreign
|
|
# value) overwrites the container's own state and leaves the gateway
|
|
# stuck "starting"/"cooking", disconnecting it from the Nous portal
|
|
# (NS-508 / the second half of NS-501).
|
|
# - ``gateway.pid`` / ``cron.pid`` / ``gateway.lock`` / ``processes.json``
|
|
# reference PIDs and locks in the *source* machine's process namespace; a
|
|
# numerically-equal PID in the new environment is a different process.
|
|
# These mirror exactly what ``container_boot._STALE_RUNTIME_FILES`` already
|
|
# sweeps on every container boot.
|
|
#
|
|
# Older backups predate the backup-side exclusions, so we filter on import too
|
|
# rather than trusting the archive's contents.
|
|
_IMPORT_SKIP_NAMES = {"gateway_state.json", "gateway.pid", "cron.pid", "gateway.lock", "processes.json"}
|
|
|
|
# zipfile.open() drops Unix mode bits on extract; restore tightens these to 0600.
|
|
_SECRET_FILE_NAMES = {".env", "auth.json", "state.db"}
|
|
|
|
# Reserved archive subtree for provider state that lives OUTSIDE HERMES_HOME
|
|
# (e.g. ~/.honcho, ~/.hindsight). The active memory provider declares these via
|
|
# MemoryProvider.backup_paths(); they're stored under this prefix encoded
|
|
# relative to the user's home directory, and restored to their original
|
|
# home-relative location on import. Anything not under home is skipped.
|
|
_EXTERNAL_PREFIX = "_external/"
|
|
|
|
|
|
class BackupInProgressError(RuntimeError):
|
|
"""Raised when another process already owns the Hermes backup slot."""
|
|
|
|
|
|
class _SQLiteSnapshotError(RuntimeError):
|
|
pass
|
|
|
|
|
|
class _SQLiteBackupTimeout(RuntimeError):
|
|
"""Raised when a SQLite snapshot remains busy past its deadline."""
|
|
|
|
|
|
@contextmanager
|
|
def _backup_operation_lock(hermes_home: Path, timeout_seconds: float = 0.25):
|
|
"""Acquire one cross-process backup slot for full and quick snapshots."""
|
|
lock_path = hermes_home / ".backup.lock"
|
|
lock_path.parent.mkdir(parents=True, exist_ok=True)
|
|
handle = lock_path.open("a+b")
|
|
acquired = False
|
|
deadline = time.monotonic() + max(0.0, timeout_seconds)
|
|
try:
|
|
if os.name == "nt":
|
|
import msvcrt
|
|
|
|
if lock_path.stat().st_size == 0:
|
|
handle.write(b" ")
|
|
handle.flush()
|
|
|
|
def _lock_op(flag: int) -> None:
|
|
handle.seek(0)
|
|
msvcrt.locking(handle.fileno(), flag, 1)
|
|
|
|
lock_flag, unlock_flag = msvcrt.LK_NBLCK, msvcrt.LK_UNLCK
|
|
else:
|
|
import fcntl
|
|
|
|
def _lock_op(flag: int) -> None:
|
|
fcntl.flock(handle.fileno(), flag)
|
|
|
|
lock_flag, unlock_flag = fcntl.LOCK_EX | fcntl.LOCK_NB, fcntl.LOCK_UN
|
|
|
|
while True:
|
|
try:
|
|
_lock_op(lock_flag)
|
|
acquired = True
|
|
break
|
|
except OSError:
|
|
if time.monotonic() >= deadline:
|
|
raise BackupInProgressError("another Hermes backup is already running")
|
|
time.sleep(0.05)
|
|
|
|
yield
|
|
finally:
|
|
if acquired:
|
|
with suppress(OSError):
|
|
_lock_op(unlock_flag)
|
|
handle.close()
|
|
|
|
|
|
@contextmanager
|
|
def _atomic_output_path(final_path: Path):
|
|
"""Yield a hidden sibling path and publish it only after a clean close."""
|
|
partial_path = final_path.with_name(f".{final_path.name}.{os.getpid()}-{threading.get_ident()}.partial")
|
|
partial_path.unlink(missing_ok=True)
|
|
try:
|
|
yield partial_path
|
|
os.replace(partial_path, final_path)
|
|
except BaseException:
|
|
partial_path.unlink(missing_ok=True)
|
|
raise
|
|
|
|
|
|
def _collect_memory_provider_external_paths() -> List[Path]:
|
|
"""Return existing absolute paths the active memory provider stores
|
|
|
|
Reads ``memory.provider``, loads just that provider, and asks it for ``backup_paths()``.
|
|
Returns ``[]`` when no external provider is active or it can't be loaded: backup must never
|
|
fail because of a flaky plugin.
|
|
"""
|
|
try:
|
|
from plugins.memory import _get_active_memory_provider, load_memory_provider
|
|
|
|
active = _get_active_memory_provider()
|
|
provider = load_memory_provider(active) if active else None
|
|
except Exception:
|
|
return []
|
|
if not active or provider is None:
|
|
return []
|
|
|
|
try:
|
|
declared = provider.backup_paths() or []
|
|
except Exception as exc:
|
|
logger.warning("backup_paths() failed for memory provider %r: %s", active, exc)
|
|
return []
|
|
|
|
out: List[Path] = []
|
|
seen: set = set()
|
|
for raw in declared:
|
|
try:
|
|
p = Path(raw).expanduser()
|
|
resolved = p.resolve() if p.exists() else None
|
|
except Exception:
|
|
continue
|
|
if resolved is not None and resolved not in seen:
|
|
seen.add(resolved)
|
|
out.append(p)
|
|
return out
|
|
|
|
|
|
def _iter_external_files(base: Path) -> List[Path]:
|
|
"""Yield regular files under *base* (a file or a directory), skipping
|
|
symlinks, caches, and pyc files. *base* itself may be a file."""
|
|
if base.is_file() and not base.is_symlink():
|
|
return [base]
|
|
files: List[Path] = []
|
|
if not base.is_dir():
|
|
return files
|
|
for dirpath, dirnames, filenames in os.walk(base, followlinks=False):
|
|
dp = Path(dirpath)
|
|
dirnames[:] = [d for d in dirnames if d not in _EXCLUDED_DIRS]
|
|
for fname in filenames:
|
|
fpath = dp / fname
|
|
if fpath.is_symlink() or fname in _EXCLUDED_NAMES or fname.endswith(_EXCLUDED_SUFFIXES):
|
|
continue
|
|
files.append(fpath)
|
|
return files
|
|
|
|
|
|
def _should_exclude(rel_path: Path) -> bool:
|
|
"""Return True if *rel_path* (relative to hermes root) should be skipped."""
|
|
parts = rel_path.parts
|
|
|
|
if _in_excluded_root_dir(rel_path):
|
|
return True
|
|
|
|
# ``hermes-agent`` only matches at the root level (first component).
|
|
# Nested directories with the same name — e.g.
|
|
# ``skills/autonomous-ai-agents/hermes-agent/`` — must be preserved.
|
|
if any(p in _EXCLUDED_DIRS and (p != "hermes-agent" or p == parts[0]) for p in parts):
|
|
return True
|
|
|
|
name = rel_path.name
|
|
return (
|
|
name in _EXCLUDED_NAMES
|
|
or name.startswith(_EXCLUDED_PREFIXES)
|
|
or name.endswith(_EXCLUDED_SUFFIXES)
|
|
)
|
|
|
|
|
|
def _should_skip_backup_file(abs_path: Path, rel_path: Path, out_path: Path) -> bool:
|
|
"""Return True when a candidate file should not be written to a backup zip."""
|
|
if _should_exclude(rel_path):
|
|
return True
|
|
|
|
# zipfile.write() follows file symlinks, so skip links before any archive
|
|
# write can copy data from outside HERMES_HOME.
|
|
if abs_path.is_symlink():
|
|
return True
|
|
|
|
try:
|
|
return abs_path.resolve() == out_path.resolve()
|
|
except (OSError, ValueError):
|
|
return False
|
|
|
|
|
|
def _iter_backup_files(hermes_root: Path, out_path: Path, skipped_dirs: Optional[set] = None):
|
|
"""Yield ``(abs_path, rel_path)`` for every file a full backup should hold.
|
|
|
|
The one owner of the backup walk policy: directory pruning (so os.walk never descends a multi-GB
|
|
excluded tree), the root-only ``hermes-agent`` carve-out, profile-home-root runtime trees, and
|
|
the per-file exclusion rules — shared by the manual ``hermes backup`` path and the automatic
|
|
pre-update/pre-migration path so the two can never drift.
|
|
"""
|
|
for dirpath, dirnames, filenames in os.walk(hermes_root, followlinks=False):
|
|
rel_dir = Path(dirpath).relative_to(hermes_root)
|
|
|
|
# ``hermes-agent`` is only pruned at the root level; nested dirs
|
|
# with the same name (e.g. in skills/) must be preserved. Managed
|
|
# runtime trees (models/, runtimes/, node/) are pruned only at a
|
|
# profile-home root — see _EXCLUDED_ROOT_DIRS.
|
|
is_root = rel_dir == Path(".")
|
|
orig_dirnames = dirnames[:]
|
|
dirnames[:] = [
|
|
d for d in dirnames
|
|
if (d not in _EXCLUDED_DIRS or (d == "hermes-agent" and not is_root))
|
|
and not _in_excluded_root_dir(rel_dir / d)
|
|
]
|
|
if skipped_dirs is not None:
|
|
for removed in set(orig_dirnames) - set(dirnames):
|
|
skipped_dirs.add(str(rel_dir / removed))
|
|
|
|
for fname in filenames:
|
|
rel = rel_dir / fname
|
|
fpath = hermes_root / rel
|
|
if _should_skip_backup_file(fpath, rel, out_path):
|
|
continue
|
|
yield fpath, rel
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# SQLite safe copy
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def _close_quietly(conn: Optional[sqlite3.Connection]) -> None:
|
|
if conn is not None:
|
|
with suppress(Exception):
|
|
conn.close()
|
|
|
|
|
|
def _query_ro_sqlite(path: Path, fn):
|
|
"""Run ``fn(conn)`` on a read-only connection to *path*; return ``(value, None)`` or ``(None, exc)``."""
|
|
conn = None
|
|
try:
|
|
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True, timeout=1.0)
|
|
return fn(conn), None
|
|
except Exception as exc:
|
|
return None, exc
|
|
finally:
|
|
_close_quietly(conn)
|
|
|
|
|
|
def _safe_copy_db(src: Path, dst: Path, *, timeout_seconds: float = 10.0) -> bool:
|
|
"""Copy a SQLite database safely using the backup() API.
|
|
|
|
Handles WAL mode — produces a consistent snapshot even while the DB is being written to. Fail
|
|
closed if a consistent snapshot cannot be created: copying only the live main file can omit
|
|
committed WAL data.
|
|
"""
|
|
conn = None
|
|
backup_conn = None
|
|
try:
|
|
# Disable sqlite3's implicit busy wait so backup() progress callbacks
|
|
# control the full locked-source deadline instead of adding the
|
|
# connection's default timeout before each callback.
|
|
conn = sqlite3.connect(f"file:{src}?mode=ro", uri=True, timeout=0.0)
|
|
backup_conn = sqlite3.connect(str(dst))
|
|
busy_deadline = time.monotonic() + max(0.0, timeout_seconds)
|
|
|
|
def _check_backup_progress(status: int, _remaining: int, _total: int) -> None:
|
|
nonlocal busy_deadline
|
|
now = time.monotonic()
|
|
if status in (sqlite3.SQLITE_BUSY, sqlite3.SQLITE_LOCKED):
|
|
if now >= busy_deadline:
|
|
raise _SQLiteBackupTimeout(f"database remained locked for {timeout_seconds:g} seconds")
|
|
else:
|
|
busy_deadline = now + max(0.0, timeout_seconds)
|
|
|
|
conn.backup(backup_conn, pages=256, progress=_check_backup_progress, sleep=0.1)
|
|
return True
|
|
except Exception as exc:
|
|
logger.warning("SQLite safe copy failed for %s: %s", src, exc)
|
|
# Windows will not remove the partial destination while SQLite still
|
|
# has it open. Close it before fail-closed cleanup; the finally block
|
|
# still owns the source and any close failure.
|
|
_close_quietly(backup_conn)
|
|
backup_conn = None
|
|
with suppress(OSError):
|
|
dst.unlink(missing_ok=True)
|
|
return False
|
|
finally:
|
|
_close_quietly(backup_conn)
|
|
_close_quietly(conn)
|
|
|
|
|
|
def is_zeroed_sqlite_file(path: Path, *, probe_bytes: int = 100, force: bool = False) -> bool:
|
|
"""True when *path* looks like the #68474 zeroed-state.db signature.
|
|
|
|
Only regular files qualify: a special file at the path (FIFO, device, socket) is never "zeroed"
|
|
— and probing one could block indefinitely (opening a FIFO for read waits for a writer), so
|
|
refuse before any I/O.
|
|
"""
|
|
try:
|
|
if not path.is_file():
|
|
return False
|
|
except OSError:
|
|
return False
|
|
from hermes_cli.sqlite_safe_read import has_live_connection, read_header_bytes_preopen
|
|
|
|
if not force and has_live_connection(path):
|
|
return False
|
|
|
|
head = read_header_bytes_preopen(path, length=max(16, probe_bytes), force=force)
|
|
# Empty or all-NUL header => zeroed; a real header (or unreadable) => not.
|
|
return head is not None and not head.startswith(b"SQLite format 3") and not any(head)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# SQLite integrity verification
|
|
# ---------------------------------------------------------------------------
|
|
|
|
_SQLITE_HEADER = b"SQLite format 3\0"
|
|
|
|
# Default ceiling above which ``PRAGMA integrity_check`` is skipped in favour
|
|
# of the (O(1)) header + structural probe. ``integrity_check`` walks every
|
|
# b-tree page in the file, so its cost scales with database size: on a 30 GB
|
|
# state.db it runs for many minutes of pegged CPU with no output, which reads
|
|
# to the user as a hung `hermes update` (#70553 follow-up). Sessions databases
|
|
# in the tens of GB are normal for heavy users, so the size-unbounded check is
|
|
# never an acceptable default on the update path.
|
|
DEFAULT_INTEGRITY_CHECK_MAX_BYTES = 2 << 30 # 2 GiB
|
|
|
|
|
|
def verify_sqlite_integrity(
|
|
path: Path,
|
|
*,
|
|
check_header: bool = True,
|
|
run_pragma: bool = True,
|
|
max_bytes: int = DEFAULT_INTEGRITY_CHECK_MAX_BYTES,
|
|
) -> dict:
|
|
"""Verify that a SQLite database at *path* is intact.
|
|
|
|
Checks, in order: 1. File exists and has an expected minimum size. 2. SQLite header magic bytes
|
|
are present. 3. For files at or under ``max_bytes``, a read-only ``PRAGMA integrity_check``. For
|
|
larger files, a cheap structural probe (schema read) instead — see ``max_bytes``.
|
|
"""
|
|
result: dict = {"valid": False, "message": "", "size": None}
|
|
|
|
def _done(message: str, valid: bool = False) -> dict:
|
|
result["valid"] = valid
|
|
result["message"] = message
|
|
return result
|
|
|
|
try:
|
|
st = path.stat()
|
|
except FileNotFoundError:
|
|
return _done(f"not found: {path}")
|
|
except OSError as exc:
|
|
return _done(f"cannot stat: {exc}")
|
|
|
|
result["size"] = st.st_size
|
|
|
|
if st.st_size < 100: # SQLite minimum viable size (header + 1 page)
|
|
return _done(f"too small ({st.st_size} bytes) to be a valid SQLite database")
|
|
|
|
oversized = max_bytes > 0 and st.st_size > max_bytes
|
|
|
|
if check_header:
|
|
# Byte-level read: refused when a live connection exists, because
|
|
# close() would cancel this process's POSIX locks on the file (see
|
|
# hermes_cli.sqlite_safe_read). Verification targets snapshots and
|
|
# backup artifacts, which are offline by construction.
|
|
from hermes_cli.sqlite_safe_read import read_header_bytes_preopen
|
|
|
|
head = read_header_bytes_preopen(path, length=len(_SQLITE_HEADER))
|
|
if head is None:
|
|
return _done("cannot read header")
|
|
if head != _SQLITE_HEADER:
|
|
return _done(f"missing SQLite header magic (got {head[:16].hex()!r})")
|
|
|
|
if oversized:
|
|
# Too large to page through PRAGMA integrity_check (which is O(file
|
|
# size) and would peg a CPU for minutes on a multi-GB state.db).
|
|
# Fall back to a cheap O(1) structural probe: the header check above
|
|
# catches the #68474 zeroed signature, and opening the DB read-only
|
|
# plus reading sqlite_master + the page geometry catches the
|
|
# malformed-schema and truncated-header-page classes. Both are
|
|
# constant-time — they parse the schema, they do not walk the data.
|
|
_, exc = _query_ro_sqlite(
|
|
path,
|
|
lambda c: (
|
|
c.execute("PRAGMA schema_version").fetchone(),
|
|
c.execute("SELECT count(*) FROM sqlite_master").fetchone(),
|
|
),
|
|
)
|
|
if exc is not None:
|
|
kind = "failed" if isinstance(exc, sqlite3.DatabaseError) else "error"
|
|
return _done(f"schema probe {kind}: {exc}")
|
|
return _done(
|
|
f"size {st.st_size:,} bytes exceeds max_bytes {max_bytes:,}; "
|
|
"skipped PRAGMA integrity_check (header + schema probe passed)",
|
|
valid=True,
|
|
)
|
|
|
|
if run_pragma:
|
|
rows, exc = _query_ro_sqlite(
|
|
path,
|
|
lambda c: [str(r[0]) for r in c.execute("PRAGMA integrity_check").fetchall()],
|
|
)
|
|
if exc is not None:
|
|
kind = "cannot open database" if isinstance(exc, sqlite3.DatabaseError) else "integrity check error"
|
|
return _done(f"{kind}: {exc}")
|
|
if rows == ["ok"]:
|
|
return _done("integrity check passed", valid=True)
|
|
return _done(f"integrity check failed: {'; '.join(rows[:5])}")
|
|
|
|
return _done("header check passed", valid=True)
|
|
|
|
|
|
def _foreign_db_holder_pids(db_path: Path) -> Optional[List[int]]:
|
|
"""PIDs of OTHER processes holding *db_path* or its WAL/SHM open.
|
|
|
|
Linux-only ``/proc/<pid>/fd`` scan (no psutil dependency), preserving the kernel's ``(deleted)``
|
|
suffix so an already-unlinked sidecar generation — the #90950 split-brain fingerprint — still
|
|
counts as held.
|
|
"""
|
|
if not sys.platform.startswith("linux"):
|
|
return None
|
|
|
|
def _canonical(path: str) -> str:
|
|
return os.path.normcase(os.path.abspath(path.removesuffix(" (deleted)")))
|
|
|
|
def _holds_watched(fd_dir: str) -> bool:
|
|
for fd in os.listdir(fd_dir):
|
|
try:
|
|
target = os.readlink(f"{fd_dir}/{fd}")
|
|
except OSError:
|
|
continue
|
|
if _canonical(target) in watched:
|
|
return True
|
|
return False
|
|
|
|
canonical_db = _canonical(os.fspath(db_path))
|
|
watched = {canonical_db, canonical_db + "-wal", canonical_db + "-shm"}
|
|
pids: List[int] = []
|
|
try:
|
|
own_pid = os.getpid()
|
|
for pid_str in os.listdir("/proc"):
|
|
if not pid_str.isdigit() or int(pid_str) == own_pid:
|
|
continue
|
|
try:
|
|
if _holds_watched(f"/proc/{pid_str}/fd"):
|
|
pids.append(int(pid_str))
|
|
except OSError:
|
|
continue
|
|
except OSError:
|
|
return None
|
|
return pids
|
|
|
|
|
|
def _safe_restore_db(src: Path, dst: Path) -> bool:
|
|
"""Restore a SQLite database from snapshot *src* into live *dst*.
|
|
|
|
Uses SQLite's backup() API to write snapshot pages into the live database file, preserving the
|
|
file's inode and WAL state so that any other process still holding the DB open (gateway,
|
|
dashboard, another CLI session) sees the restored data on the next read — instead of continuing
|
|
to serve stale cached pages from a replaced inode.
|
|
|
|
Falls back to the unlink+move approach on failure so restore never blocks on a transient error.
|
|
"""
|
|
try:
|
|
dst_conn = sqlite3.connect(str(dst))
|
|
# Force a WAL checkpoint so the backup starts from a clean
|
|
# state rather than writing on top of a deep WAL.
|
|
with suppress(Exception):
|
|
dst_conn.execute("PRAGMA wal_checkpoint(TRUNCATE)")
|
|
src_conn = sqlite3.connect(f"file:{src}?mode=ro", uri=True)
|
|
try:
|
|
src_conn.backup(dst_conn)
|
|
finally:
|
|
src_conn.close()
|
|
dst_conn.close()
|
|
# Restore original file permissions from the snapshot
|
|
with suppress(Exception):
|
|
dst.chmod(src.stat().st_mode)
|
|
return True
|
|
except Exception as exc:
|
|
logger.warning("SQLite safe restore failed for %s -> %s: %s", src, dst, exc)
|
|
return _unlink_move_restore_db(src, dst)
|
|
|
|
|
|
def _unlink_move_restore_db(src: Path, dst: Path) -> bool:
|
|
"""Fallback restore: unlink+move (the old approach). Works when no process holds the DB open.
|
|
|
|
Replacing the inode under a live holder is the #90950 corruption class: the holder keeps
|
|
writing through a deleted-inode fd (split brain), and removing its sidecars detaches the WAL
|
|
index it is checkpointing through. The backup-API path is the live-safe route; if it failed,
|
|
fail closed rather than corrupt. The foreign-pid scan deliberately excludes THIS process, but
|
|
an in-process SessionDB (the agent's own handle during /snapshot restore, a second SessionDB
|
|
instance, a read pool) is exactly as much of a live holder, so ``offline_file_access`` fails
|
|
CLOSED when any tracked connection to *dst* is live and holds the connection-lifecycle lock
|
|
across the whole swap so no new connection can appear mid-replace.
|
|
"""
|
|
from hermes_cli.sqlite_safe_read import LiveConnectionError, offline_file_access
|
|
|
|
try:
|
|
holders = _foreign_db_holder_pids(dst)
|
|
if holders:
|
|
logger.error(
|
|
"Refusing unlink+move restore of %s: process(es) %s still "
|
|
"hold the database or its WAL open. Stop them and retry.",
|
|
dst, holders,
|
|
)
|
|
return False
|
|
with offline_file_access(dst, what="unlink+move restore of"):
|
|
tmp = dst.parent / f".{dst.name}.snap_restore"
|
|
shutil.copy2(src, tmp)
|
|
dst.unlink(missing_ok=True)
|
|
# Drop the destination's sidecars before installing the snapshot. The
|
|
# snapshot is a checkpointed ``sqlite3.backup()`` image that owns no
|
|
# WAL, so any ``-wal``/``-shm`` still here describes the database we
|
|
# just unlinked (an ungracefully killed gateway leaves them behind —
|
|
# exactly when a restore gets run). SQLite would replay that foreign
|
|
# WAL over the restored file on next open and come up "malformed" (or
|
|
# silently resurrect post-snapshot rows). Same reasoning as
|
|
# ``_EXCLUDED_SUFFIXES``, applied to the restore destination.
|
|
for _sidecar_suffix in ("-wal", "-shm", "-journal"):
|
|
dst.with_name(dst.name + _sidecar_suffix).unlink(missing_ok=True)
|
|
shutil.move(str(tmp), str(dst))
|
|
return True
|
|
except LiveConnectionError as exc2:
|
|
logger.error(
|
|
"Refusing unlink+move restore of %s: %s Close the in-process "
|
|
"database handles (or restart Hermes) and retry.",
|
|
dst, exc2,
|
|
)
|
|
return False
|
|
except Exception as exc2:
|
|
logger.error("Fallback restore also failed for %s -> %s: %s", src, dst, exc2)
|
|
return False
|
|
|
|
|
|
def _zip_sqlite_snapshot(
|
|
zf: zipfile.ZipFile, abs_path: Path, rel_path: Path, out_path: Path
|
|
) -> Optional[int]:
|
|
"""Add a WAL-safe snapshot of *abs_path* to *zf*; return its byte size, or None on failure.
|
|
|
|
The snapshot is staged alongside the output zip so the temp file lives on the same
|
|
filesystem: the system default (/tmp) may be a small tmpfs that cannot hold large databases,
|
|
causing silent backup incompleteness.
|
|
"""
|
|
with tempfile.NamedTemporaryFile(
|
|
suffix=".db", delete=False, dir=str(out_path.parent)
|
|
) as tmp:
|
|
tmp_db = Path(tmp.name)
|
|
try:
|
|
if not _safe_copy_db(abs_path, tmp_db):
|
|
return None
|
|
zf.write(tmp_db, arcname=str(rel_path))
|
|
return tmp_db.stat().st_size
|
|
finally:
|
|
tmp_db.unlink(missing_ok=True)
|
|
|
|
|
|
def _write_zip_entries(
|
|
zf: zipfile.ZipFile,
|
|
files_to_add: List[Tuple[Path, Path]],
|
|
out_path: Path,
|
|
*,
|
|
on_db_failure,
|
|
on_error,
|
|
on_progress,
|
|
track_bytes: bool,
|
|
) -> int:
|
|
"""Add every ``(abs_path, rel_path)`` to *zf*, WAL-safe for ``*.db``; return bytes archived.
|
|
|
|
``on_db_failure(rel_path)`` runs when a SQLite snapshot fails (it may raise to abort);
|
|
``on_error(rel_path, exc)`` records a per-file read failure; ``on_progress(index)`` fires
|
|
every 500 files. ``track_bytes`` stats each archived plain file for the size total.
|
|
"""
|
|
total_bytes = 0
|
|
for i, (abs_path, rel_path) in enumerate(files_to_add, 1):
|
|
try:
|
|
if abs_path.suffix == ".db":
|
|
size = _zip_sqlite_snapshot(zf, abs_path, rel_path, out_path)
|
|
if size is None:
|
|
on_db_failure(rel_path)
|
|
continue
|
|
total_bytes += size
|
|
else:
|
|
zf.write(abs_path, arcname=str(rel_path))
|
|
if track_bytes:
|
|
total_bytes += abs_path.stat().st_size
|
|
except (PermissionError, OSError, ValueError) as exc:
|
|
on_error(rel_path, exc)
|
|
continue
|
|
if i % 500 == 0:
|
|
on_progress(i)
|
|
return total_bytes
|
|
|
|
|
|
def _print_capped(header: str, lines: List[str], indent: str) -> None:
|
|
"""Print *header*, then at most 10 of *lines* (each prefixed by *indent*) and a "... and N more" tail."""
|
|
print(header)
|
|
for line in lines[:10]:
|
|
print(f"{indent}{line}")
|
|
if len(lines) > 10:
|
|
print(f"{indent}... and {len(lines) - 10} more")
|
|
|
|
|
|
def _print_skipped_warnings(errors: List[str]) -> None:
|
|
_print_capped(f"\n Warnings ({len(errors)} files skipped):", errors, " ")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Backup
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def _resolve_backup_output_path(output: Optional[str]) -> Path:
|
|
"""Turn ``--output`` (file, directory, or None) into a ``.zip`` path whose parent exists.
|
|
|
|
A bad/unwritable output path (permission denied, unreadable parent, etc.) gives a clean
|
|
one-line error, not a raw traceback: is_dir() and mkdir() both hit the filesystem.
|
|
"""
|
|
out_path = None
|
|
default_name = f"hermes-backup-{datetime.now().strftime('%Y-%m-%d-%H%M%S')}.zip"
|
|
try:
|
|
if output:
|
|
out_path = Path(output).expanduser().resolve()
|
|
# If user gave a directory, put the zip inside it
|
|
if out_path.is_dir():
|
|
out_path = out_path / default_name
|
|
else:
|
|
out_path = Path.home() / default_name
|
|
if out_path.suffix.lower() != ".zip":
|
|
out_path = out_path.with_suffix(out_path.suffix + ".zip")
|
|
out_path.parent.mkdir(parents=True, exist_ok=True)
|
|
except OSError as exc:
|
|
print(f"Error: cannot write backup to {output or out_path}: {exc}")
|
|
raise SystemExit(1) from exc
|
|
return out_path
|
|
|
|
|
|
def _collect_external_entries() -> tuple[list[tuple[Path, str]], list[str]]:
|
|
"""``([(abs_path, arcname)], [skipped])`` for the active memory provider's external state.
|
|
|
|
Provider state (e.g. ~/.honcho, ~/.hindsight) lives outside HERMES_HOME, so the backup walk
|
|
never sees it; it is staged under the reserved ``_external/`` arc prefix, encoded relative to
|
|
the user's home dir. Only paths under home are captured (security + portability); anything
|
|
else is returned as skipped so the caller can note it.
|
|
"""
|
|
home_dir = Path.home().resolve()
|
|
external_to_add: list[tuple[Path, str]] = []
|
|
skipped_external: list[str] = []
|
|
for base in _collect_memory_provider_external_paths():
|
|
try:
|
|
base.resolve().relative_to(home_dir)
|
|
except (ValueError, OSError):
|
|
skipped_external.append(str(base))
|
|
continue
|
|
for fpath in _iter_external_files(base):
|
|
try:
|
|
rel_to_home = fpath.resolve().relative_to(home_dir)
|
|
except (ValueError, OSError):
|
|
continue
|
|
external_to_add.append((fpath, _EXTERNAL_PREFIX + rel_to_home.as_posix()))
|
|
return external_to_add, skipped_external
|
|
|
|
|
|
def run_backup(args) -> None:
|
|
"""Create a zip backup of the Hermes home directory."""
|
|
hermes_root = get_default_hermes_root()
|
|
|
|
if not hermes_root.is_dir():
|
|
print(f"Error: Hermes home directory not found at {hermes_root}")
|
|
sys.exit(1)
|
|
|
|
try:
|
|
with _backup_operation_lock(hermes_root):
|
|
_run_backup_locked(args, hermes_root)
|
|
except BackupInProgressError as exc:
|
|
print(f"Error: {exc}")
|
|
raise SystemExit(2) from exc
|
|
|
|
|
|
def _run_backup_locked(args, hermes_root: Path) -> None:
|
|
"""Write a full backup while the cross-process backup slot is held."""
|
|
out_path = _resolve_backup_output_path(args.output)
|
|
|
|
# Collect files
|
|
scan_started = time.monotonic()
|
|
logger.info("backup phase=scan status=started")
|
|
print(f"Scanning {display_hermes_home()} ...")
|
|
skipped_dirs: set = set()
|
|
files_to_add: list[tuple[Path, Path]] = list(_iter_backup_files(hermes_root, out_path, skipped_dirs))
|
|
external_to_add, skipped_external = _collect_external_entries()
|
|
|
|
if not files_to_add and not external_to_add:
|
|
logger.info(
|
|
"backup phase=scan status=empty duration_ms=%.1f",
|
|
(time.monotonic() - scan_started) * 1000,
|
|
)
|
|
print("No files to back up.")
|
|
return
|
|
|
|
# Create the zip
|
|
file_count = len(files_to_add) + len(external_to_add)
|
|
logger.info(
|
|
"backup phase=scan status=complete duration_ms=%.1f files=%d",
|
|
(time.monotonic() - scan_started) * 1000, file_count,
|
|
)
|
|
logger.info("backup phase=archive status=started files=%d", file_count)
|
|
print(f"Backing up {file_count} files ...")
|
|
|
|
errors = []
|
|
t0 = time.monotonic()
|
|
|
|
def _progress(i: int) -> None:
|
|
print(f" {i}/{file_count} files ...")
|
|
logger.info("backup phase=archive status=progress completed=%d total=%d", i, file_count)
|
|
|
|
with _atomic_output_path(out_path) as archive_path, zipfile.ZipFile(
|
|
archive_path, "w", zipfile.ZIP_DEFLATED, compresslevel=6
|
|
) as zf:
|
|
total_bytes = _write_zip_entries(
|
|
zf, files_to_add, out_path,
|
|
on_db_failure=lambda rel: errors.append(f"{rel}: SQLite safe copy failed"),
|
|
on_error=lambda rel, exc: errors.append(f"{rel}: {exc}"),
|
|
on_progress=_progress,
|
|
track_bytes=True,
|
|
)
|
|
|
|
# External memory-provider state, stored under the ``_external/`` arc
|
|
# prefix. These never include ``.db`` files in practice (config/env
|
|
# blobs), so a straight zf.write is fine.
|
|
for abs_path, arcname in external_to_add:
|
|
try:
|
|
zf.write(abs_path, arcname=arcname)
|
|
total_bytes += abs_path.stat().st_size
|
|
except (PermissionError, OSError, ValueError) as exc:
|
|
errors.append(f"{arcname}: {exc}")
|
|
continue
|
|
|
|
elapsed = time.monotonic() - t0
|
|
zip_size = out_path.stat().st_size
|
|
logger.info(
|
|
"backup phase=archive status=complete duration_ms=%.1f files=%d errors=%d bytes=%d",
|
|
elapsed * 1000, file_count, len(errors), zip_size,
|
|
)
|
|
|
|
# Summary
|
|
print()
|
|
print(f"Backup {'incomplete' if errors else 'complete'}: {out_path}")
|
|
print(f" Files: {file_count}")
|
|
print(f" Original: {_format_size(total_bytes)}")
|
|
print(f" Compressed: {_format_size(zip_size)}")
|
|
print(f" Time: {elapsed:.1f}s")
|
|
|
|
if external_to_add:
|
|
print(
|
|
f"\n Included {len(external_to_add)} memory-provider file(s) "
|
|
f"stored outside {display_hermes_home()}."
|
|
)
|
|
|
|
if skipped_external:
|
|
print(
|
|
f"\n Skipped {len(skipped_external)} memory-provider path(s) "
|
|
f"outside your home directory (not portable):"
|
|
)
|
|
print("\n".join(f" {p}" for p in sorted(skipped_external)[:10]))
|
|
|
|
if skipped_dirs:
|
|
print("\n Excluded directories:")
|
|
print("\n".join(f" {d}/" for d in sorted(skipped_dirs)))
|
|
|
|
if errors:
|
|
_print_skipped_warnings(errors)
|
|
else:
|
|
print(f"\nRestore with: hermes import {out_path.name}")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Import
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def _validate_backup_zip(zf: zipfile.ZipFile) -> tuple[bool, str]:
|
|
"""Check that a zip looks like a Hermes backup."""
|
|
names = zf.namelist()
|
|
if not names:
|
|
return False, "zip archive is empty"
|
|
|
|
# Telltale files a hermes home has — at the root or one level deep
|
|
# (if someone zipped the directory).
|
|
if not any(Path(n).name in {"config.yaml", ".env", "state.db"} for n in names):
|
|
return False, (
|
|
"zip does not appear to be a Hermes backup "
|
|
"(no config.yaml, .env, or state databases found)"
|
|
)
|
|
|
|
return True, ""
|
|
|
|
|
|
def _detect_prefix(zf: zipfile.ZipFile) -> str:
|
|
"""Detect if the zip has a common directory prefix wrapping all entries."""
|
|
names = [n for n in zf.namelist() if not n.endswith("/")]
|
|
if not names:
|
|
return ""
|
|
# All entries share one first directory that looks like a hermes dir name.
|
|
first_parts = {Path(n).parts[0] for n in names if len(Path(n).parts) > 1}
|
|
if len(first_parts) == 1 and first_parts <= {".hermes", "hermes"}:
|
|
return first_parts.pop() + "/"
|
|
return ""
|
|
|
|
|
|
def _default_new_file_mode() -> Optional[int]:
|
|
"""Return the mode ``open(path, "wb")`` gives a file it has to create.
|
|
|
|
``tempfile.mkstemp`` always creates at 0600, so staging an import through a temp file would
|
|
tighten every *newly created* file to owner-only — the same hazard ``utils._restore_file_mode``
|
|
documents for Docker/NAS volume mounts that rely on broader permissions.
|
|
"""
|
|
try:
|
|
current = os.umask(0o077)
|
|
os.umask(current)
|
|
except OSError:
|
|
return None
|
|
return 0o666 & ~current
|
|
|
|
|
|
def _extract_member_atomically(
|
|
zf: zipfile.ZipFile,
|
|
member: str,
|
|
target: Path,
|
|
new_file_mode: Optional[int] = None,
|
|
) -> None:
|
|
"""Restore one zip member onto *target* with no truncation window.
|
|
|
|
``open(target, "wb")`` truncates the user's existing file to zero *before* any replacement bytes
|
|
exist.
|
|
|
|
``atomic_replace`` rather than a bare ``os.replace``: it resolves a symlinked target first, so a
|
|
deployment that links ``config.yaml`` into a dotfiles repo keeps the link instead of having it
|
|
silently swapped for a regular file (GitHub #16743), and it falls back to copy/fsync/unlink on
|
|
``EXDEV``/``EBUSY`` for cross-device and bind-mount installs.
|
|
"""
|
|
# ``_preserve_file_mode`` returns None when the target does not exist (or
|
|
# cannot be stat'd), in which case the umask-derived create-mode applies —
|
|
# the same shape as ``atomic_yaml_write``'s ``create_mode`` fallback.
|
|
mode = _preserve_file_mode(target)
|
|
owner = _preserve_file_owner(target)
|
|
if mode is None:
|
|
mode = new_file_mode
|
|
else:
|
|
# Deliberately NOT a faithful mode copy: setuid/setgid are dropped.
|
|
# ``_preserve_file_mode`` returns ``stat.S_IMODE``, i.e. all twelve
|
|
# bits, and the content replacing this file comes from the archive.
|
|
# Carrying the elevated bits across would let archive-controlled bytes
|
|
# take over an existing setuid/setgid file, so ``hermes import`` would
|
|
# hand whoever produced the zip the identity that file runs as. Nothing
|
|
# constrains that to Hermes' own state either: the ``_external/`` branch
|
|
# of ``run_import`` publishes members anywhere under ``$HOME``. The
|
|
# sticky bit is kept — it is inert on a regular file.
|
|
mode &= ~(stat.S_ISUID | stat.S_ISGID)
|
|
|
|
# Truncate the stem: mkstemp adds ~16 characters, and a member already near
|
|
# NAME_MAX would otherwise fail here on a write that used to succeed.
|
|
fd, tmp_name = tempfile.mkstemp(
|
|
dir=str(target.parent), prefix=f".{target.name[:80]}.", suffix=".partial"
|
|
)
|
|
try:
|
|
with os.fdopen(fd, "wb") as dst:
|
|
if mode is not None:
|
|
# Apply the mode to the temp file BEFORE the replace so the
|
|
# target never transits through mkstemp's 0600, and so
|
|
# ``atomic_replace``'s EXDEV/EBUSY ``shutil.copystat`` fallback
|
|
# copies the intended bits rather than 0600. fchmod is
|
|
# Unix-only; Windows takes the path-based chmod.
|
|
if hasattr(os, "fchmod"):
|
|
os.fchmod(dst.fileno(), mode)
|
|
else:
|
|
os.chmod(tmp_name, mode)
|
|
# Stream instead of ``src.read()``: a multi-gigabyte state.db member
|
|
# must not be held in memory in one piece.
|
|
with zf.open(member) as src:
|
|
shutil.copyfileobj(src, dst)
|
|
dst.flush()
|
|
os.fsync(dst.fileno())
|
|
real_path = Path(atomic_replace(tmp_name, target))
|
|
# Owner first, mode second — the ordering ``atomic_yaml_write`` uses,
|
|
# because chown drops setuid/setgid and a mode restore that ran first
|
|
# would be partly undone. Here ``mode`` no longer carries those bits,
|
|
# so the two agree: neither step can re-elevate the restored file.
|
|
_restore_file_owner(real_path, owner)
|
|
_restore_file_mode(real_path, mode)
|
|
except BaseException:
|
|
with suppress(OSError):
|
|
os.unlink(tmp_name)
|
|
raise
|
|
|
|
|
|
def _confirm_import_overwrite(hermes_root: Path) -> bool:
|
|
"""Prompt before importing over an existing installation; True when import may proceed."""
|
|
if not any((hermes_root / m).exists() for m in ("config.yaml", ".env")):
|
|
return True
|
|
print()
|
|
print("Warning: Target directory already has Hermes configuration.")
|
|
print("Importing will overwrite existing files with backup contents.")
|
|
print()
|
|
try:
|
|
answer = input("Continue? [y/N] ").strip().lower()
|
|
except (EOFError, KeyboardInterrupt):
|
|
print("\nAborted.")
|
|
sys.exit(1)
|
|
if answer not in {"y", "yes"}:
|
|
print("Aborted.")
|
|
return False
|
|
return True
|
|
|
|
|
|
def run_import(args) -> None:
|
|
"""Restore a Hermes backup from a zip file."""
|
|
zip_path = Path(args.zipfile).expanduser().resolve()
|
|
|
|
if not zip_path.is_file():
|
|
print(f"Error: File not found: {zip_path}")
|
|
sys.exit(1)
|
|
|
|
if not zipfile.is_zipfile(zip_path):
|
|
print(f"Error: Not a valid zip file: {zip_path}")
|
|
sys.exit(1)
|
|
|
|
# The restore target must be the home the command operates under — the
|
|
# same path printed as "Target:" via display_hermes_home(). Resolving
|
|
# through get_default_hermes_root() instead maps a profile home
|
|
# (<root>/profiles/<name>) back to <root>, silently retargeting the
|
|
# restore at the live root while the profile directory stays empty.
|
|
hermes_root = get_hermes_home()
|
|
|
|
with zipfile.ZipFile(zip_path, "r") as zf:
|
|
# Validate
|
|
ok, reason = _validate_backup_zip(zf)
|
|
if not ok:
|
|
print(f"Error: {reason}")
|
|
sys.exit(1)
|
|
|
|
prefix = _detect_prefix(zf)
|
|
members = [n for n in zf.namelist() if not n.endswith("/")]
|
|
file_count = len(members)
|
|
|
|
print(f"Backup contains {file_count} files")
|
|
print(f"Target: {display_hermes_home()}")
|
|
|
|
if prefix:
|
|
print(f"Detected archive prefix: {prefix!r} (will be stripped)")
|
|
|
|
if not args.force and not _confirm_import_overwrite(hermes_root):
|
|
return
|
|
|
|
# Extract
|
|
print(f"\nImporting {file_count} files ...")
|
|
hermes_root.mkdir(parents=True, exist_ok=True)
|
|
|
|
errors = []
|
|
restored = 0
|
|
restored_external = 0
|
|
skipped_runtime: list[str] = []
|
|
home_dir = Path.home().resolve()
|
|
# Resolved once: every member is published via a temp file, and mkstemp
|
|
# would otherwise create newly restored files as 0600.
|
|
new_file_mode = _default_new_file_mode()
|
|
t0 = time.monotonic()
|
|
|
|
def _restore_member(
|
|
member: str, rel: str, target: Path, root: Path, tighten: bool, *, strict_chmod: bool
|
|
) -> bool:
|
|
"""Publish one member under *root*; False when blocked or failed (recorded in errors)."""
|
|
# Security: reject absolute paths and traversals
|
|
try:
|
|
target.resolve().relative_to(root)
|
|
except ValueError:
|
|
errors.append(f"{rel}: path traversal blocked")
|
|
return False
|
|
try:
|
|
target.parent.mkdir(parents=True, exist_ok=True)
|
|
_extract_member_atomically(zf, member, target, new_file_mode)
|
|
if tighten:
|
|
try:
|
|
os.chmod(target, 0o600)
|
|
except OSError:
|
|
if strict_chmod:
|
|
raise
|
|
except (PermissionError, OSError) as exc:
|
|
errors.append(f"{rel}: {exc}")
|
|
return False
|
|
return True
|
|
|
|
for member in members:
|
|
# External memory-provider state captured under the reserved
|
|
# ``_external/`` arc prefix restores to its original home-relative
|
|
# location (e.g. ~/.honcho/config.json), NOT under HERMES_HOME.
|
|
# Provider configs commonly hold credentials, so they are tightened
|
|
# to 0600 best-effort.
|
|
external = member.startswith(_EXTERNAL_PREFIX)
|
|
if external:
|
|
rel = member[len(_EXTERNAL_PREFIX):]
|
|
target = home_dir / rel
|
|
root = home_dir
|
|
tighten = target.suffix in {".json", ".env", ".conf"} or target.name in _SECRET_FILE_NAMES
|
|
else:
|
|
# Strip prefix if detected
|
|
rel = member[len(prefix):] if prefix and member.startswith(prefix) else member
|
|
# Never overwrite volatile gateway/process runtime state. These are
|
|
# namespaced to the machine/container the backup was taken on;
|
|
# clobbering them (especially gateway_state.json) breaks the gateway
|
|
# reconciler on the target and disconnects hosted instances from the
|
|
# Nous portal. Matched by basename so both the root profile and
|
|
# named profiles (profiles/<name>/gateway_state.json) are covered.
|
|
if rel and Path(rel).name in _IMPORT_SKIP_NAMES:
|
|
skipped_runtime.append(rel)
|
|
continue
|
|
target = hermes_root / rel
|
|
root = hermes_root.resolve()
|
|
tighten = target.name in _SECRET_FILE_NAMES
|
|
if not rel:
|
|
continue
|
|
|
|
if _restore_member(
|
|
member, member if external else rel, target, root, tighten, strict_chmod=not external
|
|
):
|
|
restored += 1
|
|
restored_external += external
|
|
|
|
if restored % 500 == 0:
|
|
print(f" {restored}/{file_count} files ...")
|
|
|
|
elapsed = time.monotonic() - t0
|
|
|
|
# Summary
|
|
print()
|
|
print(f"Import complete: {restored} files restored in {elapsed:.1f}s")
|
|
print(f" Target: {display_hermes_home()}")
|
|
|
|
if restored_external:
|
|
print(
|
|
f"\n Restored {restored_external} memory-provider file(s) to "
|
|
f"their original location(s) outside {display_hermes_home()}."
|
|
)
|
|
|
|
if errors:
|
|
_print_skipped_warnings(errors)
|
|
|
|
if skipped_runtime:
|
|
_print_capped(
|
|
f"\n Preserved {len(skipped_runtime)} runtime state "
|
|
f"file(s) (kept this machine's, not the backup's):",
|
|
sorted(skipped_runtime),
|
|
" ",
|
|
)
|
|
|
|
restored_profiles = _restore_profile_wrappers(hermes_root)
|
|
|
|
# Guidance
|
|
print()
|
|
if not (hermes_root / "hermes-agent").is_dir():
|
|
print("Note: The hermes-agent codebase was not included in the backup.")
|
|
print(" If this is a fresh install, run: hermes update")
|
|
|
|
if restored_profiles:
|
|
print("\nTo re-enable gateway services for profiles:")
|
|
for pname in restored_profiles:
|
|
print(f" hermes -p {pname} gateway install")
|
|
|
|
_revive_gateway_after_import(hermes_root)
|
|
print("Done. Your Hermes configuration has been restored.")
|
|
|
|
|
|
def _restore_profile_wrappers(hermes_root: Path) -> List[str]:
|
|
"""Re-create shell wrapper scripts for restored named profiles; return the profile names seen."""
|
|
profiles_dir = hermes_root / "profiles"
|
|
restored_profiles: list[tuple[str, bool]] = []
|
|
if not profiles_dir.is_dir():
|
|
return []
|
|
try:
|
|
from hermes_cli.profiles import (
|
|
create_wrapper_script, check_alias_collision,
|
|
_is_wrapper_dir_in_path, _get_wrapper_dir,
|
|
)
|
|
for entry in sorted(profiles_dir.iterdir()):
|
|
# Only create wrappers for directories with config
|
|
if not entry.is_dir() or not any((entry / m).exists() for m in ("config.yaml", ".env")):
|
|
continue
|
|
profile_name = entry.name
|
|
collision = check_alias_collision(profile_name)
|
|
if collision:
|
|
print(f" Skipped alias '{profile_name}': {collision}")
|
|
restored_profiles.append(
|
|
(profile_name, not collision and create_wrapper_script(profile_name) is not None)
|
|
)
|
|
|
|
if restored_profiles:
|
|
created = [n for n, ok in restored_profiles if ok]
|
|
skipped = [n for n, ok in restored_profiles if not ok]
|
|
if created:
|
|
print(f"\n Profile aliases restored: {', '.join(created)}")
|
|
if skipped:
|
|
print(f" Profile aliases skipped: {', '.join(skipped)}")
|
|
if not _is_wrapper_dir_in_path():
|
|
print(f"\n Note: {_get_wrapper_dir()} is not in your PATH.")
|
|
print(' Add to your shell config (~/.bashrc or ~/.zshrc):')
|
|
print(' export PATH="$HOME/.local/bin:$PATH"')
|
|
except ImportError:
|
|
# hermes_cli.profiles might not be available (fresh install)
|
|
if any(profiles_dir.iterdir()):
|
|
print("\n Profiles detected but aliases could not be created.")
|
|
print(" Run: hermes profile list (after installing hermes)")
|
|
return [n for n, _ in restored_profiles]
|
|
|
|
|
|
def _revive_gateway_after_import(hermes_root: Path) -> None:
|
|
"""Bring the restored install to life: install/start the gateway service, best-effort.
|
|
|
|
The backup may contain bot tokens and registered cron jobs, but they're inert without a
|
|
gateway process. A platform-less gateway is a supported mode, so this is safe even for backups
|
|
with no messaging config; prompt-free, and failures print a manual fallback, never fail the
|
|
import. A restore into a sandbox or profile home must not silently install a second gateway
|
|
pointed at it — on the default service name that would shadow or hijack the machine's primary
|
|
install — so the service is only revived when the restore landed in the default home, or when
|
|
no other install exists on this machine.
|
|
"""
|
|
native_default = _get_platform_default_hermes_home()
|
|
default_has_install = any(
|
|
(native_default / marker).exists() for marker in ("config.yaml", ".env", "state.db")
|
|
)
|
|
if hermes_root != native_default and default_has_install:
|
|
print(
|
|
"\nRestored into a non-default home; leaving the gateway service "
|
|
"alone to avoid clashing with the install at "
|
|
f"{native_default}."
|
|
)
|
|
print("To start a gateway for this home, run: hermes gateway install")
|
|
return
|
|
try:
|
|
from hermes_cli.gateway import ensure_gateway_service, _is_service_running
|
|
|
|
if not _is_service_running():
|
|
print()
|
|
ensure_gateway_service(context="import")
|
|
except Exception:
|
|
print("\nStart the gateway to activate cron jobs and messaging:")
|
|
print(" hermes gateway install")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Quick state snapshots (used by /snapshot slash command and hermes backup --quick)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
# Critical state files to include in quick snapshots (relative to HERMES_HOME).
|
|
# Everything else is either regeneratable (logs, cache) or managed separately
|
|
# (skills, repo, sessions/).
|
|
#
|
|
# Entries may be individual files OR directories. Directories are captured
|
|
# recursively; missing entries are silently skipped. Pairing data lives in
|
|
# platform-specific JSON blobs outside state.db, so it's listed here explicitly
|
|
# — `hermes update` snapshots this set before pulling so approved-user lists
|
|
# are recoverable if anything goes wrong (issue #15733).
|
|
_QUICK_STATE_FILES = (
|
|
"state.db",
|
|
"config.yaml",
|
|
".env",
|
|
"auth.json",
|
|
"cron/jobs.json",
|
|
"cron/executions.db",
|
|
"gateway_state.json",
|
|
"channel_directory.json",
|
|
"channel_aliases.json",
|
|
"processes.json",
|
|
"gateway/discord_message_recovery.db", # Discord reconnect replay ledger
|
|
# Per-profile user-created stores that live outside the git checkout and
|
|
# are therefore destroyed if the update flow removes/replaces the file and
|
|
# the post-update schema-init re-creates an empty one (issue #52889). All
|
|
# are at $HERMES_HOME/<name> for the default/root profile; on non-root
|
|
# profiles the real path is outside HERMES_HOME and the entry is silently
|
|
# skipped (best-effort, same as the pairing stores). SQLite DBs are copied
|
|
# WAL-safely via _safe_copy_db.
|
|
"projects.db", # per-profile project store
|
|
"response_store.db", # gateway conversation history / tool payloads
|
|
"memory_store.db", # holographic memory facts/entities
|
|
"verification_evidence.db", # agent verification audit trail
|
|
"kanban.db", # default board (back-compat <root>/kanban.db)
|
|
"kanban/boards", # non-default boards: each <slug>/kanban.db + board metadata (workspaces/ + attachments/ are skipped as regenerable)
|
|
# Pairing stores (generic + per-platform JSONs outside state.db)
|
|
"pairing", # legacy location (gateway/pairing.py)
|
|
"platforms/pairing", # new location (gateway/pairing.py)
|
|
"feishu_comment_pairing.json", # Feishu comment subscription pairings
|
|
)
|
|
|
|
# ``_QUICK_SNAPSHOTS_DIR`` lives with the exclusion rules at the top of the module.
|
|
_QUICK_DEFAULT_KEEP = 20
|
|
|
|
|
|
def _quick_snapshot_root(hermes_home: Optional[Path] = None) -> Path:
|
|
home = hermes_home or get_hermes_home()
|
|
return home / _QUICK_SNAPSHOTS_DIR
|
|
|
|
|
|
def create_quick_snapshot(
|
|
label: Optional[str] = None,
|
|
hermes_home: Optional[Path] = None,
|
|
keep: Optional[int] = None,
|
|
max_file_size: Optional[int] = None,
|
|
) -> Optional[str]:
|
|
"""Create one atomic quick snapshot while holding the shared backup slot."""
|
|
home = hermes_home or get_hermes_home()
|
|
with _backup_operation_lock(home):
|
|
return _create_quick_snapshot_locked(label, home, keep, max_file_size)
|
|
|
|
|
|
def _quick_snapshot_candidates(home: Path):
|
|
"""Yield ``(src, rel_posix, in_dir)`` for every regular file a quick snapshot captures.
|
|
|
|
Directory entries of ``_QUICK_STATE_FILES`` are walked so restore can treat every file
|
|
uniformly; empty dirs are skipped. Heavy, regenerable per-board subtrees (scratch workspaces
|
|
and task attachments) are skipped — only the board databases + metadata are needed.
|
|
"""
|
|
for rel in _QUICK_STATE_FILES:
|
|
src = home / rel
|
|
if not src.exists():
|
|
continue
|
|
if src.is_dir():
|
|
for sub in src.rglob("*"):
|
|
if not sub.is_file():
|
|
continue
|
|
sub_rel = sub.relative_to(home).as_posix()
|
|
if "/workspaces/" in f"/{sub_rel}/" or "/attachments/" in f"/{sub_rel}/":
|
|
continue
|
|
yield sub, sub_rel, True
|
|
elif src.is_file():
|
|
yield src, rel, False
|
|
|
|
|
|
def _copy_quick_snapshot_files(
|
|
home: Path, staging_dir: Path, max_file_size: Optional[int]
|
|
) -> tuple[Dict[str, int], list[str], list[str]]:
|
|
"""Copy every quick-snapshot candidate into *staging_dir*.
|
|
|
|
Returns ``(manifest, failed_dbs, oversized_skipped)``: ``manifest`` maps rel_path -> file size;
|
|
``failed_dbs`` lists present ``*.db`` that could not be snapshotted; ``oversized_skipped`` lists
|
|
protected DB files skipped for size (#68805) — those are snapshot incompleteness just like a
|
|
failed copy, so the caller must suppress pruning to preserve the older complete snapshot that
|
|
may contain the only recoverable database.
|
|
"""
|
|
manifest: Dict[str, int] = {}
|
|
failed_dbs: list[str] = []
|
|
oversized_skipped: list[str] = []
|
|
|
|
for src, rel, in_dir in _quick_snapshot_candidates(home):
|
|
if max_file_size is not None:
|
|
try:
|
|
size = src.stat().st_size
|
|
except OSError:
|
|
size = None
|
|
if size is not None and size > max_file_size:
|
|
print(
|
|
f" ⚠ Snapshot: skipping {rel} "
|
|
f"({_format_size(size)} exceeds {_format_size(max_file_size)} limit)"
|
|
)
|
|
logger.warning(
|
|
"Quick snapshot skipped %s: %d bytes exceeds %d byte limit", rel, size, max_file_size
|
|
)
|
|
if src.suffix == ".db":
|
|
oversized_skipped.append(rel)
|
|
continue
|
|
|
|
dst = staging_dir / rel
|
|
dst.parent.mkdir(parents=True, exist_ok=True)
|
|
try:
|
|
# Route SQLite DBs through the WAL-safe backup() path so a DB with
|
|
# an open WAL (the gateway may hold it at snapshot time) is
|
|
# captured consistently.
|
|
if src.suffix == ".db":
|
|
if not _safe_copy_db(src, dst):
|
|
failed_dbs.append(rel)
|
|
print(
|
|
f" ⚠ Snapshot: SQLite safe copy FAILED for {rel} "
|
|
f"— file may be locked or corrupted"
|
|
)
|
|
if is_zeroed_sqlite_file(src):
|
|
nuls = " of NULs?" if in_dir else ""
|
|
print(
|
|
f" ⚠ Snapshot: {rel} looks ZEROED "
|
|
f"(no SQLite header; {src.stat().st_size} bytes{nuls})"
|
|
)
|
|
continue
|
|
else:
|
|
shutil.copy2(src, dst)
|
|
manifest[rel] = dst.stat().st_size
|
|
except (OSError, PermissionError) as exc:
|
|
logger.warning("Could not snapshot %s: %s", rel, exc)
|
|
return manifest, failed_dbs, oversized_skipped
|
|
|
|
|
|
def _create_quick_snapshot_locked(
|
|
label: Optional[str], hermes_home: Optional[Path], keep: Optional[int], max_file_size: Optional[int]
|
|
) -> Optional[str]:
|
|
"""Create a quick state snapshot of critical files.
|
|
|
|
Copies STATE_FILES to a timestamped directory under state-snapshots/ and prunes old snapshots.
|
|
``max_file_size`` skips (with a warning) files above that many bytes; the pre-update snapshot
|
|
uses it so a multi-GB ``state.db`` can never stall ``hermes update`` while the small
|
|
pairing/cron/config files are always captured. ``None`` copies everything.
|
|
"""
|
|
home = hermes_home or get_hermes_home()
|
|
root = _quick_snapshot_root(home)
|
|
|
|
ts = datetime.now(timezone.utc).strftime("%Y%m%d-%H%M%S")
|
|
base_snap_id = f"{ts}-{label}" if label else ts
|
|
snap_id = base_snap_id
|
|
suffix = 2
|
|
while (root / snap_id).exists():
|
|
snap_id = f"{base_snap_id}-{suffix}"
|
|
suffix += 1
|
|
snap_dir = root / snap_id
|
|
staging_dir = root / f".{snap_id}.{os.getpid()}.partial"
|
|
shutil.rmtree(staging_dir, ignore_errors=True)
|
|
staging_dir.mkdir(parents=True, exist_ok=False)
|
|
logger.info("quick snapshot phase=copy status=started id=%s", snap_id)
|
|
|
|
manifest, failed_dbs, oversized_skipped = _copy_quick_snapshot_files(home, staging_dir, max_file_size)
|
|
|
|
if failed_dbs:
|
|
# Critical: update path used to log-and-continue with exit 0, so a
|
|
# missing state.db backup looked like a successful pre-update snapshot
|
|
# (#68474). Surface this on stdout where operators actually look.
|
|
print(" ⚠ CRITICAL: could not snapshot DB file(s): " + ", ".join(failed_dbs))
|
|
print(f" ⚠ If sessions disappear after update, check {root} and run: hermes snapshot list")
|
|
logger.error("Quick snapshot failed to capture DB file(s): %s", ", ".join(failed_dbs))
|
|
|
|
if not manifest:
|
|
shutil.rmtree(staging_dir, ignore_errors=True)
|
|
if failed_dbs:
|
|
# Distinguish "nothing to snapshot" from "state.db present but unreadable"
|
|
print(f" ⚠ Snapshot aborted: no files captured (failed DBs: {', '.join(failed_dbs)})")
|
|
return None
|
|
|
|
# Write manifest
|
|
meta = {
|
|
"id": snap_id,
|
|
"timestamp": ts,
|
|
"label": label,
|
|
"file_count": len(manifest),
|
|
"total_size": sum(manifest.values()),
|
|
"files": manifest,
|
|
"failed_dbs": failed_dbs,
|
|
"oversized_skipped": oversized_skipped,
|
|
}
|
|
with open(staging_dir / "manifest.json", "w", encoding="utf-8") as f:
|
|
json.dump(meta, f, indent=2)
|
|
|
|
os.replace(staging_dir, snap_dir)
|
|
|
|
# Auto-prune. Defaults preserve historical manual /snapshot behavior; callers
|
|
# with known high-churn safety snapshots (for example pre-update) can pass a
|
|
# smaller keep value so large state.db copies do not accumulate indefinitely.
|
|
# #68805 review: skip pruning when a present DB failed to capture OR was
|
|
# skipped for size — either way the snapshot is incomplete and the older
|
|
# snapshot may contain the only recoverable database.
|
|
if not (failed_dbs or oversized_skipped):
|
|
_prune_quick_snapshots(root, keep=_QUICK_DEFAULT_KEEP if keep is None else keep)
|
|
else:
|
|
if oversized_skipped:
|
|
print(
|
|
" ⚠ Skipping snapshot prune: DB file(s) skipped for size: "
|
|
+ ", ".join(oversized_skipped)
|
|
)
|
|
logger.warning("Quick snapshot skipped oversized DB file(s): %s", ", ".join(oversized_skipped))
|
|
logger.warning(
|
|
"Skipping snapshot prune because %d DB(s) failed to capture "
|
|
"and/or %d were oversized — preserving older snapshots as "
|
|
"recovery source",
|
|
len(failed_dbs), len(oversized_skipped),
|
|
)
|
|
|
|
logger.info(
|
|
"quick snapshot phase=copy status=complete id=%s files=%d bytes=%d",
|
|
snap_id, len(manifest), sum(manifest.values()),
|
|
)
|
|
return snap_id
|
|
|
|
|
|
def _snapshot_dirs(root: Path) -> List[Path]:
|
|
"""Published snapshot directories under *root*, newest (by name) first."""
|
|
if not root.exists():
|
|
return []
|
|
return sorted(
|
|
(d for d in root.iterdir()
|
|
if d.is_dir() and not d.name.startswith(".") and not d.name.endswith(".partial")),
|
|
key=lambda d: d.name,
|
|
reverse=True,
|
|
)
|
|
|
|
|
|
def list_quick_snapshots(limit: int = 20, hermes_home: Optional[Path] = None) -> List[Dict[str, Any]]:
|
|
"""List existing quick state snapshots, most recent first."""
|
|
results = []
|
|
for d in _snapshot_dirs(_quick_snapshot_root(hermes_home)):
|
|
manifest_path = d / "manifest.json"
|
|
if manifest_path.exists():
|
|
try:
|
|
with open(manifest_path, encoding="utf-8") as f:
|
|
results.append(json.load(f))
|
|
except (json.JSONDecodeError, OSError):
|
|
results.append({"id": d.name, "file_count": 0, "total_size": 0})
|
|
if len(results) >= limit:
|
|
break
|
|
|
|
return results
|
|
|
|
|
|
def restore_quick_snapshot(snapshot_id: str, hermes_home: Optional[Path] = None) -> bool:
|
|
"""Restore state from a quick snapshot."""
|
|
home = hermes_home or get_hermes_home()
|
|
root = _quick_snapshot_root(home)
|
|
|
|
# Security: reject snapshot_id values that contain path separators or
|
|
# traversal sequences so that `root / snapshot_id` stays inside root.
|
|
if not snapshot_id or "/" in snapshot_id or "\\" in snapshot_id or snapshot_id in (".", ".."):
|
|
logger.error("Invalid snapshot_id: %s", snapshot_id)
|
|
return False
|
|
|
|
snap_dir = root / snapshot_id
|
|
|
|
# Confirm the resolved path is still inside root (handles symlinks etc.)
|
|
try:
|
|
snap_dir.resolve().relative_to(root.resolve())
|
|
except ValueError:
|
|
logger.error("Snapshot path traversal blocked for id: %s", snapshot_id)
|
|
return False
|
|
|
|
manifest_path = snap_dir / "manifest.json"
|
|
if not snap_dir.is_dir() or not manifest_path.exists():
|
|
return False
|
|
|
|
with open(manifest_path, encoding="utf-8") as f:
|
|
meta = json.load(f)
|
|
|
|
restored = 0
|
|
for rel in meta.get("files", {}):
|
|
# Security: reject absolute paths and traversals in manifest entries
|
|
src = snap_dir / rel
|
|
dst = home / rel
|
|
try:
|
|
src.resolve().relative_to(snap_dir.resolve())
|
|
dst.resolve().relative_to(home.resolve())
|
|
except ValueError:
|
|
logger.error("Manifest path traversal blocked: %s", rel)
|
|
continue
|
|
if not src.exists():
|
|
continue
|
|
dst.parent.mkdir(parents=True, exist_ok=True)
|
|
try:
|
|
if dst.suffix == ".db":
|
|
# Restore through SQLite backup API so live connections
|
|
# (gateway, dashboard, another CLI session) see the
|
|
# restored data instead of continuing to serve stale
|
|
# cached pages from a replaced inode (issue #65942).
|
|
_safe_restore_db(src, dst)
|
|
else:
|
|
shutil.copy2(src, dst)
|
|
restored += 1
|
|
except (OSError, PermissionError) as exc:
|
|
logger.error("Failed to restore %s: %s", rel, exc)
|
|
|
|
logger.info("Restored %d files from snapshot %s", restored, snapshot_id)
|
|
return restored > 0
|
|
|
|
|
|
# Relative path of the cron job database inside HERMES_HOME. Kept in sync with
|
|
# the entry in ``_QUICK_STATE_FILES`` and with ``cron/jobs.py``'s ``JOBS_FILE``.
|
|
_CRON_JOBS_REL = "cron/jobs.json"
|
|
|
|
|
|
def _count_cron_jobs(path: Path) -> Optional[int]:
|
|
"""Return the number of cron jobs stored in ``path``.
|
|
|
|
Accepts the canonical ``{"jobs": [...]}`` shape and the legacy bare list. Returns ``None`` if
|
|
the file is missing or unparseable; callers must treat ``None`` as "unknown", not zero,
|
|
since acting on an unreadable file could mask a real corruption the user needs to see.
|
|
"""
|
|
if not path.is_file():
|
|
return None
|
|
try:
|
|
# utf-8-sig: same dialect as cron/jobs.load_jobs — Windows editors
|
|
# may leave a UTF-8 BOM that plain utf-8 json.load rejects. Without
|
|
# it a BOM'd jobs.json counts as "unreadable" (None) and the
|
|
# post-update cron-loss auto-restore safety net silently disables.
|
|
with open(path, "r", encoding="utf-8-sig") as f:
|
|
data = json.load(f)
|
|
except (OSError, json.JSONDecodeError):
|
|
return None
|
|
if isinstance(data, dict):
|
|
data = data.get("jobs", [])
|
|
return len(data) if isinstance(data, list) else None
|
|
|
|
|
|
def restore_cron_jobs_if_emptied(
|
|
snapshot_id: str,
|
|
hermes_home: Optional[Path] = None,
|
|
) -> Optional[Dict[str, Any]]:
|
|
"""Safety net for silent cron-job loss across ``hermes update``.
|
|
|
|
The check is deliberately conservative — it only ever restores when there is unambiguous
|
|
evidence of loss (snapshot had more jobs than live file), so a user who genuinely deleted jobs
|
|
during/after the update is never second-guessed, and an unreadable live file (count ``None``) is
|
|
left untouched so real corruption still surfaces.
|
|
"""
|
|
if not snapshot_id:
|
|
return None
|
|
|
|
home = hermes_home or get_hermes_home()
|
|
live_path = home / _CRON_JOBS_REL
|
|
|
|
live_count = _count_cron_jobs(live_path)
|
|
# ``None`` (missing or unparseable) is intentionally left alone — that's a
|
|
# different failure mode the user should see rather than have papered over.
|
|
if live_count is None:
|
|
return None
|
|
|
|
snap_path = _quick_snapshot_root(home) / snapshot_id / _CRON_JOBS_REL
|
|
snap_count = _count_cron_jobs(snap_path)
|
|
if not snap_count: # None or 0 — nothing worth restoring
|
|
return None
|
|
|
|
# Restore when live has FEWER jobs than the pre-update snapshot.
|
|
# Catches both total loss (0 vs N) and partial loss (1 vs 19) — the
|
|
# desktop scheduler can overwrite jobs.json with its own small set of
|
|
# internally-tracked crons after an update/restart.
|
|
if live_count >= snap_count:
|
|
return None
|
|
|
|
try:
|
|
live_path.parent.mkdir(parents=True, exist_ok=True)
|
|
shutil.copy2(snap_path, live_path)
|
|
except (OSError, PermissionError) as exc:
|
|
logger.error("Cron jobs were emptied during update but auto-restore failed: %s", exc)
|
|
return None
|
|
|
|
logger.warning(
|
|
"Restored %d cron job(s) from pre-update snapshot %s "
|
|
"(live file had %d job(s), snapshot had %d — jobs were lost during migration)",
|
|
snap_count, snapshot_id, live_count, snap_count,
|
|
)
|
|
return {"restored": True, "job_count": snap_count, "snapshot_id": snapshot_id}
|
|
|
|
|
|
def _sibling_profile_homes(invoking_home: Path) -> list[tuple[str, Path]]:
|
|
"""(name, home) for every OTHER profile on this install. Never raises.
|
|
|
|
The update's code swap and gateway fleet restart touch every profile, so the pre-update snapshot
|
|
must too (#66140). The invoking profile is excluded — its snapshot is taken by the existing
|
|
call.
|
|
"""
|
|
homes: list[tuple[str, Path]] = []
|
|
try:
|
|
from hermes_cli.profiles import (
|
|
_get_default_hermes_home,
|
|
_get_profiles_root,
|
|
_PROFILE_ID_RE,
|
|
)
|
|
|
|
invoking = invoking_home.resolve()
|
|
default_home = _get_default_hermes_home()
|
|
if default_home.is_dir() and default_home.resolve() != invoking:
|
|
homes.append(("default", default_home))
|
|
root = _get_profiles_root()
|
|
if root.is_dir():
|
|
for entry in sorted(root.iterdir()):
|
|
if (
|
|
entry.is_dir()
|
|
and entry.name != "default"
|
|
and _PROFILE_ID_RE.match(entry.name)
|
|
and entry.resolve() != invoking
|
|
):
|
|
homes.append((entry.name, entry))
|
|
except Exception as exc:
|
|
logger.debug("Sibling profile enumeration failed: %s", exc)
|
|
return homes
|
|
|
|
|
|
def create_pre_update_snapshots_all_profiles(
|
|
invoking_home: Optional[Path] = None,
|
|
keep: Optional[int] = None,
|
|
max_file_size: Optional[int] = None,
|
|
) -> Dict[str, str]:
|
|
"""Pre-update quick snapshots for every SIBLING profile (#66140).
|
|
|
|
Same snapshot set, same per-file size cap, same keep policy as the invoking profile's snapshot —
|
|
identical semantics per profile, no partial-tier coherence class. Each sibling's snapshot lands
|
|
under its OWN ``<home>/state-snapshots/`` so per-profile restore tooling finds it where it
|
|
expects.
|
|
"""
|
|
results: Dict[str, str] = {}
|
|
home = invoking_home or get_hermes_home()
|
|
for name, profile_home in _sibling_profile_homes(home):
|
|
try:
|
|
snap_id = create_quick_snapshot(
|
|
label="pre-update", hermes_home=profile_home, keep=keep, max_file_size=max_file_size
|
|
)
|
|
if snap_id:
|
|
results[name] = snap_id
|
|
except Exception as exc:
|
|
logger.debug("Pre-update snapshot for profile %s failed: %s", name, exc)
|
|
return results
|
|
|
|
|
|
# Config paths that the update flow must never change (#64160): the model
|
|
# routing keys and the Mixture-of-Agents section are consumed machine-wide
|
|
# (gateway, cron, desktop), so an update/repair cycle that rewrites them
|
|
# silently redirects paid inference. Each entry is a dotted path into the raw
|
|
# config.yaml document; a single-element tuple protects the whole section.
|
|
_PROTECTED_CONFIG_PATHS: Tuple[Tuple[str, ...], ...] = (
|
|
("model", "provider"),
|
|
("model", "default"),
|
|
("model", "base_url"),
|
|
("model", "api_key"),
|
|
("moa",),
|
|
)
|
|
|
|
|
|
def _read_raw_yaml_dict(path: Path) -> Optional[Dict[str, Any]]:
|
|
"""Parse ``path`` as a YAML mapping. ``None`` = missing/unreadable/non-dict."""
|
|
if not path.is_file():
|
|
return None
|
|
try:
|
|
import yaml
|
|
|
|
with open(path, "r", encoding="utf-8") as f:
|
|
data = yaml.safe_load(f)
|
|
except Exception:
|
|
return None
|
|
return data if isinstance(data, dict) else None
|
|
|
|
|
|
def _get_config_path_value(data: Dict[str, Any], dotted: Tuple[str, ...]) -> Any:
|
|
node: Any = data
|
|
for key in dotted:
|
|
if not isinstance(node, dict):
|
|
return None
|
|
node = node.get(key)
|
|
return node
|
|
|
|
|
|
def _set_config_path_value(data: Dict[str, Any], dotted: Tuple[str, ...], value: Any) -> None:
|
|
node = data
|
|
for key in dotted[:-1]:
|
|
child = node.get(key)
|
|
if not isinstance(child, dict):
|
|
child = {}
|
|
node[key] = child
|
|
node = child
|
|
node[dotted[-1]] = value
|
|
|
|
|
|
def restore_config_model_settings_if_rewritten(
|
|
snapshot_id: str,
|
|
hermes_home: Optional[Path] = None,
|
|
) -> Optional[Dict[str, Any]]:
|
|
"""Safety net for silent config.yaml model/MoA loss across ``hermes update``.
|
|
|
|
These keys are consumed by the gateway and unattended cron jobs too, so a rewrite silently
|
|
changes paid inference behavior machine-wide.
|
|
|
|
Mirrors :func:`restore_cron_jobs_if_emptied`: compare the *current* config against the pre-
|
|
update snapshot taken minutes earlier by this same update run, and restore only the protected
|
|
keys — never the whole file — when a value the user had set was changed or dropped.
|
|
"""
|
|
if not snapshot_id:
|
|
return None
|
|
|
|
home = hermes_home or get_hermes_home()
|
|
live_path = home / "config.yaml"
|
|
snap_path = _quick_snapshot_root(home) / snapshot_id / "config.yaml"
|
|
|
|
snap = _read_raw_yaml_dict(snap_path)
|
|
if not snap:
|
|
return None # no snapshot copy — nothing to compare against
|
|
live = _read_raw_yaml_dict(live_path)
|
|
if live is None:
|
|
# Missing or unparseable live config is a different failure mode the
|
|
# user should see rather than have papered over (matches the cron net).
|
|
return None
|
|
|
|
restored_keys: list[str] = []
|
|
for dotted in _PROTECTED_CONFIG_PATHS:
|
|
snap_val = _get_config_path_value(snap, dotted)
|
|
if snap_val in (None, "", {}, []):
|
|
continue # user never set it — nothing to protect
|
|
live_val = _get_config_path_value(live, dotted)
|
|
if live_val == snap_val:
|
|
continue
|
|
_set_config_path_value(live, dotted, snap_val)
|
|
restored_keys.append(".".join(dotted))
|
|
|
|
if not restored_keys:
|
|
return None
|
|
|
|
try:
|
|
from utils import atomic_yaml_write
|
|
|
|
atomic_yaml_write(live_path, live)
|
|
except (OSError, PermissionError) as exc:
|
|
logger.error(
|
|
"config.yaml model settings were rewritten during update but "
|
|
"auto-restore failed: %s",
|
|
exc,
|
|
)
|
|
return None
|
|
|
|
logger.warning(
|
|
"Restored user config value(s) %s from pre-update snapshot %s — "
|
|
"the update flow rewrote them (#64160)",
|
|
", ".join(restored_keys),
|
|
snapshot_id,
|
|
)
|
|
return {"restored": True, "keys": restored_keys, "snapshot_id": snapshot_id}
|
|
|
|
|
|
def _restore_all_sibling_profiles(
|
|
profile_snapshots: Dict[str, str],
|
|
invoking_home: Optional[Path],
|
|
restore_fn,
|
|
failure_log: str,
|
|
) -> list[Dict[str, Any]]:
|
|
"""Run a per-profile safety net (``restore_fn(snap_id, hermes_home=...)``) for every sibling.
|
|
|
|
Each profile's live file is compared against ITS OWN same-generation pre-update snapshot.
|
|
Returns one result dict per restored profile, each with a ``profile`` key added. Never raises.
|
|
"""
|
|
restored: list[Dict[str, Any]] = []
|
|
if not profile_snapshots:
|
|
return restored
|
|
home = invoking_home or get_hermes_home()
|
|
by_name = dict(_sibling_profile_homes(home))
|
|
for name, snap_id in profile_snapshots.items():
|
|
profile_home = by_name.get(name)
|
|
if profile_home is None:
|
|
continue
|
|
try:
|
|
result = restore_fn(snap_id, hermes_home=profile_home)
|
|
except Exception as exc:
|
|
logger.debug(failure_log, name, exc)
|
|
continue
|
|
if result:
|
|
result["profile"] = name
|
|
restored.append(result)
|
|
return restored
|
|
|
|
|
|
def restore_config_model_settings_all_profiles(
|
|
profile_snapshots: Dict[str, str],
|
|
invoking_home: Optional[Path] = None,
|
|
) -> list[Dict[str, Any]]:
|
|
"""Run the config model-settings safety net for every sibling profile.
|
|
|
|
Same contract as :func:`restore_cron_jobs_all_profiles`: each profile's live ``config.yaml`` is
|
|
compared against ITS OWN same-generation pre-update snapshot. Returns one result dict per
|
|
restored profile, each with a ``profile`` key added. Never raises.
|
|
"""
|
|
return _restore_all_sibling_profiles(
|
|
profile_snapshots,
|
|
invoking_home,
|
|
restore_config_model_settings_if_rewritten,
|
|
"Config model-settings restore check for profile %s failed: %s",
|
|
)
|
|
|
|
|
|
def restore_cron_jobs_all_profiles(
|
|
profile_snapshots: Dict[str, str],
|
|
invoking_home: Optional[Path] = None,
|
|
) -> list[Dict[str, Any]]:
|
|
"""Run the cron-jobs safety net for every sibling profile (#66140).
|
|
|
|
``profile_snapshots`` comes from :func:`create_pre_update_snapshots_all_profiles`; each
|
|
profile's live ``cron/jobs.json`` is compared against ITS OWN snapshot, so restores are
|
|
same-generation by construction. Returns one result dict per restored profile. Never raises.
|
|
"""
|
|
return _restore_all_sibling_profiles(
|
|
profile_snapshots,
|
|
invoking_home,
|
|
restore_cron_jobs_if_emptied,
|
|
"Cron restore check for profile %s failed: %s",
|
|
)
|
|
|
|
|
|
def _prune_oldest(newest_first: List[Path], keep: int, remove, what: str) -> int:
|
|
"""``remove(path)`` every entry past the first *keep*; return how many succeeded."""
|
|
deleted = 0
|
|
for p in newest_first[keep:]:
|
|
try:
|
|
remove(p)
|
|
deleted += 1
|
|
except OSError as exc:
|
|
logger.warning("Failed to prune %s %s: %s", what, p.name, exc)
|
|
return deleted
|
|
|
|
|
|
def _prune_quick_snapshots(root: Path, keep: int = _QUICK_DEFAULT_KEEP) -> int:
|
|
"""Remove oldest quick snapshots beyond the keep limit. Returns count deleted."""
|
|
return _prune_oldest(_snapshot_dirs(root), keep, shutil.rmtree, "snapshot")
|
|
|
|
|
|
def prune_quick_snapshots(keep: int = _QUICK_DEFAULT_KEEP, hermes_home: Optional[Path] = None) -> int:
|
|
"""Manually prune quick snapshots. Returns count deleted."""
|
|
return _prune_quick_snapshots(_quick_snapshot_root(hermes_home), keep=keep)
|
|
|
|
|
|
def run_quick_backup(args) -> None:
|
|
"""CLI entry point for hermes backup --quick."""
|
|
label = getattr(args, "label", None)
|
|
snap_id = create_quick_snapshot(label=label)
|
|
if snap_id:
|
|
print(f"State snapshot created: {snap_id}")
|
|
snaps = list_quick_snapshots()
|
|
print(f" {len(snaps)} snapshot(s) stored in {display_hermes_home()}/state-snapshots/")
|
|
print(f" Restore with: /snapshot restore {snap_id}")
|
|
else:
|
|
print("No state files found to snapshot.")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Shared full-zip backup helper
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def _write_full_zip_backup(out_path: Path, hermes_root: Path) -> Optional[Path]:
|
|
"""Write a full zip snapshot of ``hermes_root`` to ``out_path`` while holding the backup slot.
|
|
|
|
Uses the same exclusion rules and SQLite safe-copy as :func:`run_backup`. Returns the output
|
|
path on success, None on failure (nothing to back up, another backup running, or write error —
|
|
caller should surface the outcome but not raise).
|
|
"""
|
|
try:
|
|
with _backup_operation_lock(hermes_root):
|
|
return _write_full_zip_backup_locked(out_path, hermes_root)
|
|
except BackupInProgressError as exc:
|
|
logger.warning("Full-zip backup skipped: %s", exc)
|
|
return None
|
|
|
|
|
|
def _write_full_zip_backup_locked(out_path: Path, hermes_root: Path) -> Optional[Path]:
|
|
scan_started = time.monotonic()
|
|
logger.info("automatic backup phase=scan status=started")
|
|
try:
|
|
files_to_add = list(_iter_backup_files(hermes_root, out_path))
|
|
except OSError as exc:
|
|
logger.warning("Full-zip backup: walk failed: %s", exc)
|
|
return None
|
|
|
|
if not files_to_add:
|
|
return None
|
|
|
|
logger.info(
|
|
"automatic backup phase=scan status=complete duration_ms=%.1f files=%d",
|
|
(time.monotonic() - scan_started) * 1000,
|
|
len(files_to_add),
|
|
)
|
|
|
|
archive_started = time.monotonic()
|
|
|
|
def _db_failure(rel_path: Path) -> None:
|
|
logger.warning("Full-zip backup aborted: SQLite snapshot failed for %s", rel_path)
|
|
raise _SQLiteSnapshotError(str(rel_path))
|
|
|
|
try:
|
|
with _atomic_output_path(out_path) as archive_path, zipfile.ZipFile(
|
|
archive_path, "w", zipfile.ZIP_DEFLATED, compresslevel=6
|
|
) as zf:
|
|
_write_zip_entries(
|
|
zf, files_to_add, out_path,
|
|
on_db_failure=_db_failure,
|
|
on_error=lambda rel, exc: logger.debug("Skipping %s in zip backup: %s", rel, exc),
|
|
on_progress=lambda i: logger.info(
|
|
"automatic backup phase=archive status=progress completed=%d total=%d",
|
|
i, len(files_to_add),
|
|
),
|
|
track_bytes=False,
|
|
)
|
|
except (OSError, _SQLiteSnapshotError) as exc:
|
|
logger.warning("Full-zip backup: zip write failed: %s", exc)
|
|
# ``_atomic_output_path`` already removed the hidden partial. Do not
|
|
# unlink ``out_path`` here: it may be a previous valid backup that the
|
|
# atomic publisher deliberately preserved.
|
|
return None
|
|
|
|
logger.info(
|
|
"automatic backup phase=archive status=complete duration_ms=%.1f files=%d bytes=%d",
|
|
(time.monotonic() - archive_started) * 1000,
|
|
len(files_to_add),
|
|
out_path.stat().st_size,
|
|
)
|
|
|
|
return out_path
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Pre-update auto-backup
|
|
# ---------------------------------------------------------------------------
|
|
|
|
_PRE_UPDATE_BACKUPS_DIR = "backups"
|
|
_PRE_UPDATE_PREFIX = "pre-update-"
|
|
_PRE_UPDATE_DEFAULT_KEEP = 5
|
|
|
|
|
|
def _prune_prefixed_zips(backup_dir: Path, prefix: str, keep: int, what: str) -> int:
|
|
"""Remove oldest ``<prefix>*.zip`` files in *backup_dir* beyond the keep limit.
|
|
|
|
Returns the number of files deleted. Only touches files matching the prefix so hand-made zips
|
|
or other backup kinds dropped in the same directory are never touched.
|
|
|
|
Operators who genuinely don't want a backup should set ``updates.pre_update_backup: off`` in
|
|
config — that gates creation.
|
|
"""
|
|
if not backup_dir.exists():
|
|
return 0
|
|
|
|
backups = sorted(
|
|
(p for p in backup_dir.iterdir()
|
|
if p.is_file() and p.name.startswith(prefix) and p.suffix.lower() == ".zip"),
|
|
key=lambda p: p.name,
|
|
reverse=True,
|
|
)
|
|
return _prune_oldest(backups, keep, Path.unlink, what)
|
|
|
|
|
|
def _create_prefixed_full_backup(
|
|
hermes_home: Optional[Path], prefix: str, keep: int, what: str, prune_what: str
|
|
) -> Optional[Path]:
|
|
"""Write ``<HERMES_HOME>/backups/<prefix><timestamp>.zip`` and prune older same-prefix zips.
|
|
|
|
Returns the created path, or ``None`` if nothing was found to back up or the write failed.
|
|
Never raises.
|
|
"""
|
|
hermes_root = hermes_home or get_default_hermes_root()
|
|
if not hermes_root.is_dir():
|
|
return None
|
|
|
|
backup_dir = hermes_root / _PRE_UPDATE_BACKUPS_DIR
|
|
try:
|
|
backup_dir.mkdir(parents=True, exist_ok=True)
|
|
except OSError as exc:
|
|
logger.warning("Could not create %s backup dir %s: %s", what, backup_dir, exc)
|
|
return None
|
|
|
|
stamp = datetime.now().strftime("%Y-%m-%d-%H%M%S")
|
|
out_path = backup_dir / f"{prefix}{stamp}.zip"
|
|
|
|
if _write_full_zip_backup(out_path, hermes_root) is None:
|
|
return None
|
|
|
|
_prune_prefixed_zips(backup_dir, prefix, keep, prune_what)
|
|
return out_path
|
|
|
|
|
|
def create_pre_update_backup(
|
|
hermes_home: Optional[Path] = None,
|
|
keep: int = _PRE_UPDATE_DEFAULT_KEEP,
|
|
) -> Optional[Path]:
|
|
"""Create a full zip backup of HERMES_HOME under ``backups/``.
|
|
|
|
Mirrors :func:`run_backup` (same exclusion rules, same SQLite safe-copy) but writes to
|
|
``<HERMES_HOME>/backups/pre-update-<timestamp>.zip`` and auto-prunes old pre-update backups.
|
|
|
|
Returns the path to the created zip, or ``None`` if no files were found or the backup could not
|
|
be created. Never raises — the caller (``hermes update``) should continue even if the backup
|
|
fails.
|
|
"""
|
|
return _create_prefixed_full_backup(
|
|
hermes_home, _PRE_UPDATE_PREFIX, max(keep, 1), "pre-update", "backup"
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Pre-migration auto-backup (used by `hermes claw migrate`)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
_PRE_MIGRATION_PREFIX = "pre-migration-"
|
|
_PRE_MIGRATION_DEFAULT_KEEP = 5
|
|
|
|
|
|
def create_pre_migration_backup(
|
|
hermes_home: Optional[Path] = None,
|
|
keep: int = _PRE_MIGRATION_DEFAULT_KEEP,
|
|
) -> Optional[Path]:
|
|
"""Create a full zip backup of HERMES_HOME under ``backups/`` before a ``hermes claw migrate`` apply.
|
|
|
|
Shares implementation with :func:`create_pre_update_backup` via ``_write_full_zip_backup`` —
|
|
same exclusions, same SQLite safe-copy, restorable with ``hermes import <archive>``. Writes to
|
|
``<HERMES_HOME>/backups/pre-migration-<timestamp>.zip`` (the shared ``backups/`` directory, so
|
|
``hermes import`` and the update-backup listing pick up pre-migration archives too) and
|
|
auto-prunes old pre-migration backups.
|
|
|
|
Returns the path to the created zip, or ``None`` if nothing was found to back up (fresh install)
|
|
or the write failed. Never raises — the caller decides whether to abort or proceed.
|
|
"""
|
|
return _create_prefixed_full_backup(
|
|
hermes_home, _PRE_MIGRATION_PREFIX, max(keep, 0), "pre-migration", "pre-migration backup"
|
|
)
|