diff --git a/gateway/run.py b/gateway/run.py index 2a352c2c56..8a587becde 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -4777,7 +4777,7 @@ def _housekeeping_checkpoint_prune() -> None: """Checkpoint store retention + size cap on a live timer; ``auto_prune_from_config`` gates on ``checkpoints.auto_prune`` and the 24h ``.last_prune`` marker. Off the startup path because its ``git gc`` can block for tens of seconds on a large store.""" - from tools.checkpoint_manager import auto_prune_from_config + from tools.checkpoint_maintenance import auto_prune_from_config auto_prune_from_config() diff --git a/hermes_cli/checkpoints.py b/hermes_cli/checkpoints.py index d0ce78c3df..3a05191e2b 100644 --- a/hermes_cli/checkpoints.py +++ b/hermes_cli/checkpoints.py @@ -27,7 +27,7 @@ def _fmt_age(ts: Any) -> str: def cmd_status(args: argparse.Namespace) -> int: - from tools.checkpoint_manager import store_status + from tools.checkpoint_maintenance import store_status info = store_status() base = info["base"] @@ -64,7 +64,7 @@ def _print_archives(archives) -> None: def cmd_prune(args: argparse.Namespace) -> int: - from tools.checkpoint_manager import prune_checkpoints, store_status + from tools.checkpoint_maintenance import prune_checkpoints, store_status delete_orphans = not args.keep_orphans @@ -135,7 +135,8 @@ def _confirmed(args: argparse.Namespace, prompt: str) -> bool: def cmd_clear(args: argparse.Namespace) -> int: - from tools.checkpoint_manager import CHECKPOINT_BASE, clear_all, store_status + from tools.checkpoint_manager import CHECKPOINT_BASE + from tools.checkpoint_maintenance import clear_all, store_status info = store_status() if info["total_size_bytes"] == 0 and not Path(CHECKPOINT_BASE).exists(): @@ -160,7 +161,7 @@ def cmd_clear(args: argparse.Namespace) -> int: def cmd_clear_legacy(args: argparse.Namespace) -> int: - from tools.checkpoint_manager import clear_legacy, store_status + from tools.checkpoint_maintenance import clear_legacy, store_status info = store_status() legacy = info.get("legacy_archives", []) diff --git a/hermes_cli/cli_auto_maintenance.py b/hermes_cli/cli_auto_maintenance.py index dcc14dc7ee..f291231329 100644 --- a/hermes_cli/cli_auto_maintenance.py +++ b/hermes_cli/cli_auto_maintenance.py @@ -70,5 +70,5 @@ def _run_checkpoint_auto_maintenance() -> None: """Checkpoint store retention on a daemon thread: its ``git gc`` can block for tens of seconds on a large store, which used to stall the prompt once a day. ``auto_prune_from_config`` owns the config gate and the 24h marker and never raises.""" - from tools.checkpoint_manager import auto_prune_from_config + from tools.checkpoint_maintenance import auto_prune_from_config threading.Thread(target=auto_prune_from_config, name="checkpoint-auto-prune", daemon=True).start() diff --git a/hermes_cli/doctor_state.py b/hermes_cli/doctor_state.py index 9a7e1fc0e1..be60aee019 100644 --- a/hermes_cli/doctor_state.py +++ b/hermes_cli/doctor_state.py @@ -504,7 +504,7 @@ def _check_state_db(should_fix: bool, f: Finding) -> None: @doctor_check() def _check_checkpoint_store(should_fix: bool, f: Finding) -> None: """/rollback store footprint: warn when checkpoints are on and the store sits above its cap.""" - from tools.checkpoint_manager import checkpoint_footprint_notice + from tools.checkpoint_maintenance import checkpoint_footprint_notice notice = checkpoint_footprint_notice() if notice: check_warn(notice) diff --git a/hermes_cli/update_cmd_maint.py b/hermes_cli/update_cmd_maint.py index bd03f5655c..02fba4cee2 100644 --- a/hermes_cli/update_cmd_maint.py +++ b/hermes_cli/update_cmd_maint.py @@ -871,7 +871,7 @@ def _refresh_cua_driver_after_update() -> None: def _print_checkpoint_footprint_notice() -> None: """Surface a GB-scale /rollback store the user may not know is on (see the helper's docstring).""" - from tools.checkpoint_manager import checkpoint_footprint_notice + from tools.checkpoint_maintenance import checkpoint_footprint_notice notice = checkpoint_footprint_notice() if notice: print(f"\n\033[1;33mℹ {notice}\033[0m") diff --git a/tests/gateway/test_checkpoint_prune_housekeeping.py b/tests/gateway/test_checkpoint_prune_housekeeping.py index 813dc9e0e5..e7186bda21 100644 --- a/tests/gateway/test_checkpoint_prune_housekeeping.py +++ b/tests/gateway/test_checkpoint_prune_housekeeping.py @@ -21,7 +21,7 @@ class _OneTickStopEvent: def test_gateway_housekeeping_runs_the_checkpoint_prune(monkeypatch): - import tools.checkpoint_manager as cm + import tools.checkpoint_maintenance as cm calls = [] monkeypatch.setattr(cm, "auto_prune_from_config", lambda: calls.append(True) or {"skipped": False}) diff --git a/tests/hermes_cli/test_checkpoints_prune.py b/tests/hermes_cli/test_checkpoints_prune.py index 909c1746a2..82d048336e 100644 --- a/tests/hermes_cli/test_checkpoints_prune.py +++ b/tests/hermes_cli/test_checkpoints_prune.py @@ -49,7 +49,7 @@ _MIXED_STATUS = { def _patch_checkpoint_manager(monkeypatch, status: dict, prune_calls: list): - import tools.checkpoint_manager as ckpt_mgr + import tools.checkpoint_maintenance as ckpt_mgr monkeypatch.setattr(ckpt_mgr, "store_status", lambda *a, **k: status) diff --git a/tests/hermes_cli/test_console_engine.py b/tests/hermes_cli/test_console_engine.py index 0aefdb57e5..04117638cf 100644 --- a/tests/hermes_cli/test_console_engine.py +++ b/tests/hermes_cli/test_console_engine.py @@ -285,7 +285,7 @@ _ORPHAN_STORE_STATUS = { def _patch_checkpoint_manager(monkeypatch, prune_calls: list) -> None: """Report one orphan project and record the resulting prune call.""" - import tools.checkpoint_manager as ckpt_mgr + import tools.checkpoint_maintenance as ckpt_mgr monkeypatch.setattr(ckpt_mgr, "store_status", lambda *a, **k: _ORPHAN_STORE_STATUS) diff --git a/tests/test_utils_rmtree_readonly.py b/tests/test_utils_rmtree_readonly.py index 7f84c254f6..effdfa4876 100644 --- a/tests/test_utils_rmtree_readonly.py +++ b/tests/test_utils_rmtree_readonly.py @@ -29,9 +29,9 @@ def _read_only_object_dir(root: Path) -> Path: def test_checkpoint_clear_all_removes_tree_with_read_only_object(tmp_path): - """Driven through a production call site (#117170): ``checkpoint_manager.clear_all`` must + """Driven through a production call site (#117170): ``checkpoint_maintenance.clear_all`` must reach ``rmtree_readonly`` — a bare ``shutil.rmtree`` there reports ``deleted=False``.""" - from tools.checkpoint_manager import clear_all + from tools.checkpoint_maintenance import clear_all root = tmp_path / "checkpoints" obj_dir = _read_only_object_dir(root) diff --git a/tests/tools/test_checkpoint_footprint_notice.py b/tests/tools/test_checkpoint_footprint_notice.py index 3008d4a030..4222e28aa3 100644 --- a/tests/tools/test_checkpoint_footprint_notice.py +++ b/tests/tools/test_checkpoint_footprint_notice.py @@ -11,7 +11,8 @@ import os import hermes_yaml as yaml from hermes_constants import get_hermes_home -from tools.checkpoint_manager import CheckpointManager, checkpoint_footprint_notice +from tools.checkpoint_manager import CheckpointManager +from tools.checkpoint_maintenance import checkpoint_footprint_notice def _write_config(enabled: bool, cap_mb: int) -> None: home = get_hermes_home() diff --git a/tests/tools/test_checkpoint_manager.py b/tests/tools/test_checkpoint_manager.py index e6cd68c1fc..0186afa43e 100644 --- a/tests/tools/test_checkpoint_manager.py +++ b/tests/tools/test_checkpoint_manager.py @@ -19,11 +19,9 @@ from tools.checkpoint_manager import ( _ref_name, _project_meta_path, _touch_project, - prune_checkpoints, - maybe_auto_prune_checkpoints, - store_status, - clear_all, - clear_legacy, +) +from tools.checkpoint_maintenance import ( + clear_all, clear_legacy, maybe_auto_prune_checkpoints, prune_checkpoints, store_status, ) diff --git a/tests/tools/test_checkpoint_size_cap.py b/tests/tools/test_checkpoint_size_cap.py index b19d473928..f81c422569 100644 --- a/tests/tools/test_checkpoint_size_cap.py +++ b/tests/tools/test_checkpoint_size_cap.py @@ -5,7 +5,8 @@ import os import pytest -from tools import checkpoint_manager as checkpoints +import tools.checkpoint_manager as checkpoints +from tools import checkpoint_maintenance @pytest.fixture @@ -31,7 +32,7 @@ def test_maintenance_size_cap_retains_every_snapshot_after_the_only_large_one(hi base, project, manager = history store = checkpoints._store_path(base) assert checkpoints._dir_size_bytes(store) > 1024 * 1024 - result = checkpoints.prune_checkpoints(retention_days=0, checkpoint_base=base, max_total_size_mb=1) + result = checkpoint_maintenance.prune_checkpoints(retention_days=0, checkpoint_base=base, max_total_size_mb=1) assert result["errors"] == 0 assert [row["reason"] for row in manager.list_checkpoints(str(project))] == [f"small-{i}" for i in reversed(range(5))] assert checkpoints._dir_size_bytes(store) <= 1024 * 1024 @@ -66,7 +67,7 @@ def test_snapshot_over_cap_drops_one_round_and_leaves_the_gc_to_the_prune(histor # Objects are still in the pack until the prune runs. assert checkpoints._dir_size_bytes(store) > 1024 * 1024 - result = checkpoints.prune_checkpoints(retention_days=0, checkpoint_base=base, max_total_size_mb=1) + result = checkpoint_maintenance.prune_checkpoints(retention_days=0, checkpoint_base=base, max_total_size_mb=1) assert result["errors"] == 0 assert not (store / GC_PENDING_NAME).exists() assert checkpoints._dir_size_bytes(store) <= 1024 * 1024 @@ -84,7 +85,7 @@ def test_reclaim_stops_before_touching_another_projects_history(history): manager.new_turn() assert manager.ensure_checkpoint(str(other), f"other-{index}") before = manager.list_checkpoints(str(other)) - result = checkpoints.prune_checkpoints(retention_days=0, checkpoint_base=base, max_total_size_mb=1) + result = checkpoint_maintenance.prune_checkpoints(retention_days=0, checkpoint_base=base, max_total_size_mb=1) assert result["errors"] == 0 assert manager.list_checkpoints(str(other)) == before assert len(manager.list_checkpoints(str(first))) == 5 @@ -95,7 +96,7 @@ def test_maintenance_stops_on_failed_git_without_losing_a_second_snapshot(histor import json base, project, manager = history - original = checkpoints._run_git + original = checkpoint_maintenance._run_git rewrites = [] retention = 0 if failure == "delete-ref": @@ -114,8 +115,8 @@ def test_maintenance_stops_on_failed_git_without_losing_a_second_snapshot(histor return result with monkeypatch.context() as patcher: - patcher.setattr(checkpoints, "_run_git", fail_one) - result = checkpoints.prune_checkpoints(retention_days=retention, checkpoint_base=base, max_total_size_mb=1) + patcher.setattr(checkpoint_maintenance, "_run_git", fail_one) + result = checkpoint_maintenance.prune_checkpoints(retention_days=retention, checkpoint_base=base, max_total_size_mb=1) assert result["errors"] > 0 assert len(rewrites) <= 1 assert len(manager.list_checkpoints(str(project))) >= 5 diff --git a/tests/tools/test_checkpoint_store_ownership.py b/tests/tools/test_checkpoint_store_ownership.py index 371407fe50..509e4d46e0 100644 --- a/tests/tools/test_checkpoint_store_ownership.py +++ b/tests/tools/test_checkpoint_store_ownership.py @@ -8,7 +8,8 @@ import subprocess import sys from threading import Thread -from tools import checkpoint_manager as checkpoints +import tools.checkpoint_manager as checkpoints +from tools import checkpoint_maintenance def test_active_snapshot_owns_store_until_its_ref_is_published(tmp_path, monkeypatch): @@ -45,9 +46,9 @@ assert c.CheckpointManager(enabled=True, max_total_size_mb=0).ensure_checkpoint( reader.start() try: assert ready.get(timeout=20).strip() == "tree-written" - result = checkpoints.prune_checkpoints(retention_days=0, checkpoint_base=base) + result = checkpoint_maintenance.prune_checkpoints(retention_days=0, checkpoint_base=base) assert result["errors"] > 0, "maintenance must refuse an active store writer" - assert not checkpoints.clear_all(base)["deleted"] + assert not checkpoint_maintenance.clear_all(base)["deleted"] finally: output, error = child.communicate(input="publish\n", timeout=30) reader.join(timeout=5) diff --git a/tools/checkpoint_maintenance.py b/tools/checkpoint_maintenance.py new file mode 100644 index 0000000000..73c072dc2e --- /dev/null +++ b/tools/checkpoint_maintenance.py @@ -0,0 +1,608 @@ +"""Checkpoint store retention, orphan pruning, status and cleanup.""" + +import logging +import os +import time +from pathlib import Path +from typing import Dict, Optional + +from hermes_cli.gitlock import clear_stale_tmp_packs +from utils import rmtree_readonly +from tools.checkpoint_manager import ( + _GIT_TIMEOUT, _LEGACY_PREFIX, _PRUNE_MARKER_NAME, _REFS_PREFIX, _STORE_DIRNAME, + _dir_size_bytes, _index_path, _list_projects, _pre_v2_shadow_repos, + _project_meta_path, _ref_name, _resolve_checkpoint_base, _run_git, _store_path, +) + +logger = logging.getLogger(__name__) + +# --------------------------------------------------------------------------- +# Auto-maintenance +# --------------------------------------------------------------------------- +# +# v2 rewrite. The sweep now operates on per-project refs inside the shared +# store rather than per-project shadow repos. Legacy-archive dirs +# (``legacy-/``) are swept with the same retention policy. + +def _delete_ref(store: Path, ref: str) -> bool: + """Delete a ref from the store. Returns True on success.""" + ok, _, _ = _run_git( + ["update-ref", "-d", ref], store, str(store.parent), + allowed_returncodes={128}, + ) + return ok + + +def _workdir_is_observably_gone( + workdir: str, + parent_dev: Optional[int] = None, + parent_ino: Optional[int] = None, + require_parent_identity: bool = True, +) -> bool: + """True only when we can positively observe that ``workdir`` was removed. + + ``Path.exists()`` returns False for a deleted directory AND for one whose + storage simply is not attached right now — an unplugged external drive, a + network share behind a downed VPN, a bind-mount absent from this + container, an offline Windows mapped drive. Orphan pruning deletes the + project's entire checkpoint history, so treating that ambiguity as + "deleted" throws away the user's restore points over a transient mount + state, unattended, at startup. + + Require corroboration, in three steps. + + First, the parent directory must be present, so the absence of the project + inside it is something we actually observed. When the parent is missing + too, the volume is not there and we know nothing. + + Second, the present parent must be the directory we knew — not merely a + directory at the same path. Unmounting swaps the directory visible at a + mount point: while the volume is attached the path resolves to the + mounted filesystem's root; after detach it resolves to the *underlying* + (underlay) directory, which may carry entries of its own (a ``.keep`` + placeholder, sibling mount points). Those entries were never next to the + project and prove nothing about the volume being attached. So the + parent's ``(st_dev, st_ino)`` must match the identity recorded in the + project's metadata while the project was observably live + (``parent_dev``/``parent_ino``). A mismatch means a different directory + is visible at that path — a detached volume, not an observed deletion. + When no identity was ever recorded (metadata written by an older + version) and ``require_parent_identity`` is True, stay conservative and + do not classify as orphan. Callers that have no identity channel at all + (the frozen pre-v2 layout) pass ``require_parent_identity=False`` to + keep the structural checks only. + + Third, the (identity-confirmed) parent must actually carry information. + Unmounting leaves classic static mount points (``/mnt/volume/proj``, an + fstab entry, a container bind-mount) behind as *empty* directories, so an + empty parent is the signature of a detached volume just as much as of a + deleted project. Prune only when the parent holds something else (we + observed a populated directory that does not contain the project) or is + itself a live mount point (the volume is demonstrably attached and the + project is demonstrably not on it). + + Genuinely abandoned projects are still reclaimed by the retention/stale + rule, which runs off ``last_touch`` rather than a filesystem probe. + """ + if not workdir: + return False + path = Path(workdir) + try: + if path.exists(): + return False + parent = path.parent + # A path whose parent is itself (a filesystem root) gives us nothing + # to corroborate against. + if parent == path: + return False + if not parent.is_dir(): + return False + if parent_dev is not None and parent_ino is not None: + st = parent.stat() + if (st.st_dev, st.st_ino) != (parent_dev, parent_ino): + # A different directory is visible at the parent's path than + # the one the project lived in — the volume is detached (its + # underlay showing through) or was swapped. Not a deletion. + return False + elif require_parent_identity: + # No recorded identity to check against — we cannot tell the + # project's real parent from an underlay directory exposed by an + # unmount. Unsure never deletes; retention still reclaims. + return False + if _dir_has_any_entry(parent): + return True + # Empty parent: only evidence if that directory is a mount point, i.e. + # the volume is attached right now and simply does not hold the + # project. An empty plain directory is an unmounted mount point as + # readily as an emptied project root. + return os.path.ismount(parent) + except OSError: + # Probe failed (permission, I/O error) — not evidence of deletion. + return False + + +def _dir_has_any_entry(directory: Path) -> bool: + """True when ``directory`` contains at least one entry. + + Stops after the first entry rather than materializing the listing; a + project root can hold a large tree. + """ + with os.scandir(directory) as entries: + for _ in entries: + return True + return False + + +def prune_checkpoints( + retention_days: int = 7, + delete_orphans: bool = True, + checkpoint_base: Optional[Path] = None, + max_total_size_mb: int = 0, + orphan_allowlist: Optional[set] = None, +) -> Dict[str, int]: + """Delete stale/orphan checkpoints and reclaim store space. + + A project entry is deleted when either: + + * ``delete_orphans=True`` and its ``workdir`` no longer exists on disk + (the original project was deleted / moved); OR + * its ``last_touch`` is older than ``retention_days`` days. + + ``orphan_allowlist``, when not ``None``, restricts orphan deletion to + the given identities (v2 project ``_hash`` strings and/or pre-v2 shadow + repo paths as ``str``). This lets a caller that showed the user a + confirmation preview (built from ``store_status()``) bind the resulting + deletion to exactly what was displayed — a project that only becomes + orphaned *after* the preview (e.g. its workdir vanishes while the human + is answering the prompt) is skipped rather than swept up under the + earlier confirmation. Pass ``None`` (the default) to delete every + currently-orphaned project, e.g. for ``--force`` or unattended callers + that never show a preview. + + Additionally, if ``max_total_size_mb > 0`` and the store exceeds that + after orphan/stale pruning, the oldest commit per remaining project is + dropped until the store is under the cap. + + Legacy-archive dirs (``legacy-*``) older than ``retention_days`` are + also deleted. + + Returns a dict with counts ``{"scanned", "deleted_orphan", + "deleted_stale", "errors", "bytes_freed"}``. + + Never raises — maintenance must never block interactive startup. + """ + base = checkpoint_base or _resolve_checkpoint_base() + result = { + "scanned": 0, + "deleted_orphan": 0, + "deleted_stale": 0, + "errors": 0, + "bytes_freed": 0, + } + if not base.exists(): + return result + + from tools.checkpoint_pruning import PruneError, store_lock + + try: + with store_lock(base): + return _prune_checkpoints(base, result, retention_days, delete_orphans, max_total_size_mb, orphan_allowlist) + except (PruneError, OSError) as exc: + result["errors"] += 1 + logger.warning("Checkpoint maintenance stopped: %s", exc) + return result + + +def _prune_checkpoints( + base: Path, result: Dict[str, int], retention_days: int, delete_orphans: bool, + max_total_size_mb: int, orphan_allowlist: Optional[set], +) -> Dict[str, int]: + size_before = _dir_size_bytes(base) + + # --- Legacy pre-v2 per-project shadow repos (kept directly under base) --- + # Pre-v2 layout: ``base//HEAD`` etc. We treat these exactly as the + # v1 pruner did so behaviour is unchanged for anyone still on that layout + # or sitting on a mid-migration system. + cutoff = 0.0 + if retention_days > 0: + cutoff = time.time() - retention_days * 86400 + + for child in base.iterdir(): + if not child.is_dir(): + continue + if child.name == _STORE_DIRNAME: + continue + if child.name.startswith(_LEGACY_PREFIX): + # Legacy archive: prune by dir mtime using same retention rule. + if retention_days <= 0: + continue + try: + m = child.stat().st_mtime + except OSError: + continue + if m >= cutoff: + continue + try: + size = _dir_size_bytes(child) + rmtree_readonly(child) + result["bytes_freed"] += size + result["deleted_stale"] += 1 + except OSError as exc: + result["errors"] += 1 + logger.warning("Failed to delete legacy archive %s: %s", child, exc) + + # Pre-v2 per-project shadow repos. Scanned via the same helper + # `store_status()` uses for its orphan preview, so a confirmation prompt + # built from that preview always matches what gets deleted here. + for repo in _pre_v2_shadow_repos(base): + child = repo["path"] + result["scanned"] += 1 + reason: Optional[str] = None + if ( + delete_orphans + and not repo["marker_unreadable"] + and ( + repo["workdir"] is None + # The frozen pre-v2 layout has no metadata channel to carry a + # recorded parent identity, so only the structural checks + # (parent present + populated / live mount point) apply here. + or _workdir_is_observably_gone( + repo["workdir"], require_parent_identity=False, + ) + ) + and (orphan_allowlist is None or str(child) in orphan_allowlist) + ): + reason = "orphan" + if reason is None and retention_days > 0: + newest = 0.0 + try: + for p in child.rglob("*"): + try: + mt = p.stat().st_mtime + newest = max(newest, mt) + except OSError: + continue + except OSError: + pass + if newest > 0 and newest < cutoff: + reason = "stale" + if reason is None: + continue + try: + size = _dir_size_bytes(child) + rmtree_readonly(child) + result["bytes_freed"] += size + if reason == "orphan": + result["deleted_orphan"] += 1 + else: + result["deleted_stale"] += 1 + except OSError as exc: + result["errors"] += 1 + logger.warning("Failed to prune checkpoint repo %s: %s", child.name, exc) + + # --- v2 shared store: per-project ref pruning via metadata --- + store = _store_path(base) + if (store / "HEAD").exists(): + # A gc killed by the store timeout strands tmp_pack_* files that gc.auto=0 means git + # itself never reclaims; sweep them even when no ref moved (a sweep is a directory + # listing, unlike the pack-rewriting gc gated on refs below). + clear_stale_tmp_packs(store) + for meta in _list_projects(store): + dir_hash = meta.get("_hash") or "" + workdir = meta.get("workdir") or "" + if not dir_hash: + continue + result["scanned"] += 1 + reason = None + parent_dev = meta.get("workdir_parent_dev") + parent_ino = meta.get("workdir_parent_ino") + if not isinstance(parent_dev, int) or isinstance(parent_dev, bool): + parent_dev = None + if not isinstance(parent_ino, int) or isinstance(parent_ino, bool): + parent_ino = None + if ( + delete_orphans + and ( + not workdir + or _workdir_is_observably_gone( + workdir, + parent_dev=parent_dev, + parent_ino=parent_ino, + ) + ) + and (orphan_allowlist is None or dir_hash in orphan_allowlist) + ): + reason = "orphan" + elif retention_days > 0: + last_touch = float(meta.get("last_touch", 0) or 0) + if last_touch > 0 and last_touch < cutoff: + reason = "stale" + if reason is None: + continue + ref = _ref_name(dir_hash) + if not _delete_ref(store, ref): + result["errors"] += 1 + return result + # Drop per-project index and metadata. + try: + idx = _index_path(store, dir_hash) + if idx.exists(): + idx.unlink() + except OSError: + pass + try: + mp = _project_meta_path(store, dir_hash) + if mp.exists(): + mp.unlink() + except OSError: + pass + if reason == "orphan": + result["deleted_orphan"] += 1 + else: + result["deleted_stale"] += 1 + + from tools.checkpoint_pruning import Pruner, PruneError + + pruner = Pruner(_run_git, store, str(base), _GIT_TIMEOUT, _dir_size_bytes, _REFS_PREFIX) + try: + if result["deleted_orphan"] + result["deleted_stale"] or pruner.gc_pending(): + pruner.reclaim() + if not pruner.enforce_size(max_total_size_mb * 1024 * 1024): + result["errors"] += 1 + logger.warning("Checkpoint store remains over its size cap; minimum history retained") + except (PruneError, OSError) as exc: + result["errors"] += 1 + logger.warning("Checkpoint pruning stopped: %s", exc) + + size_after = _dir_size_bytes(base) + delta = size_before - size_after + result["bytes_freed"] = max(result["bytes_freed"], delta) + + return result + + +def maybe_auto_prune_checkpoints( + retention_days: int = 7, + min_interval_hours: int = 24, + delete_orphans: bool = True, + checkpoint_base: Optional[Path] = None, + max_total_size_mb: int = 0, +) -> Dict[str, object]: + """Idempotent wrapper around ``prune_checkpoints`` for startup hooks. + + Writes ``CHECKPOINT_BASE/.last_prune`` on completion so subsequent + calls within ``min_interval_hours`` short-circuit. + + Returns ``{"skipped": bool, "result": prune_checkpoints-dict, + "error": optional str}``. + """ + base = checkpoint_base or _resolve_checkpoint_base() + out: Dict[str, object] = {"skipped": False} + + try: + if not base.exists(): + out["result"] = { + "scanned": 0, "deleted_orphan": 0, "deleted_stale": 0, + "errors": 0, "bytes_freed": 0, + } + return out + + marker = base / _PRUNE_MARKER_NAME + now = time.time() + try: + if marker.exists() and now - float(marker.read_text(encoding="utf-8-sig").strip()) < min_interval_hours * 3600: + out["skipped"] = True + return out + except (OSError, ValueError): + pass # corrupt marker — treat as no prior run + # Claim the interval before pruning: callers run on a periodic tick, and a prune that + # dies mid-way must cost one skipped day, not a git gc every tick. + try: + marker.write_text(str(now), encoding="utf-8") + except OSError as exc: + logger.debug("Could not write checkpoint prune marker: %s", exc) + result = out["result"] = prune_checkpoints(retention_days=retention_days, delete_orphans=delete_orphans, + checkpoint_base=base, max_total_size_mb=max_total_size_mb) + + total = result["deleted_orphan"] + result["deleted_stale"] + if total > 0: + logger.info( + "checkpoint auto-maintenance: pruned %d entry(ies) " + "(%d orphan, %d stale), reclaimed %.1f MB", + total, + result["deleted_orphan"], + result["deleted_stale"], + result["bytes_freed"] / (1024 * 1024), + ) + except Exception as exc: + logger.warning("checkpoint auto-maintenance failed: %s", exc) + out["error"] = str(exc) + + return out + + +def auto_prune_from_config() -> Dict[str, object]: + """``maybe_auto_prune_checkpoints`` driven by the ``checkpoints:`` config section — the one + startup/housekeeping entry point for the CLI and the gateway. ``delete_orphans`` is never + honoured unattended: a missing workdir is ambiguous (deleted vs. unmounted share); orphan + cleanup is only via explicit ``hermes checkpoints prune``. Never raises.""" + try: + from hermes_cli.config import load_config + cfg = load_config().get("checkpoints") or {} + if not cfg.get("auto_prune", False): + return {"skipped": True} + return maybe_auto_prune_checkpoints( + retention_days=int(cfg.get("retention_days", 7)), + min_interval_hours=int(cfg.get("min_interval_hours", 24)), + delete_orphans=False, + max_total_size_mb=int(cfg.get("max_total_size_mb", 500))) + except Exception as exc: + logger.debug("checkpoint auto-maintenance skipped: %s", exc) + return {"skipped": True, "error": str(exc)} + + +def checkpoint_footprint_notice() -> Optional[str]: + """One-line notice when ``/rollback`` checkpoints are on and their store sits at or above + ``checkpoints.max_total_size_mb``, else None. Checkpoints were on by default for a while + (Mar–May 2026) and that ``enabled: true`` persisted into user configs; many users carry a + GB-scale store for a feature they never invoke. The cap is a floor of one snapshot per + project, so a big store is expected, not broken — the notice names the opt-out. Never raises.""" + try: + from hermes_cli.config import load_config + cfg = load_config().get("checkpoints") or {} + if not cfg.get("enabled", False): + return None + cap_mb = int(cfg.get("max_total_size_mb", 500) or 0) + status = store_status() + size = int(status["total_size_bytes"]) + if cap_mb <= 0 or size < cap_mb * 1024 * 1024: + return None + from hermes_cli.sizefmt import format_bytes + return (f"Filesystem checkpoints (/rollback) are on: {format_bytes(size)} across " + f"{status['project_count']} project(s), above the {cap_mb} MB cap (one snapshot per project is " + f"always kept). Not using /rollback? `hermes config set checkpoints.enabled false` then " + f"`hermes checkpoints clear`; or lower `checkpoints.retention_days`.") + except Exception as exc: + logger.debug("checkpoint footprint notice skipped: %s", exc) + return None + + +# --------------------------------------------------------------------------- +# Public helpers for `hermes checkpoints` CLI +# --------------------------------------------------------------------------- + +def store_status(checkpoint_base: Optional[Path] = None) -> Dict: + """Return a summary of the shadow store. + + ``{"base": path, "store_size_bytes": N, "legacy_size_bytes": N, + "total_size_bytes": N, "project_count": N, "projects": [...], + "pre_v2_projects": [...], "legacy_archives": [...]}`` + + ``pre_v2_projects`` covers shadow repos still on the pre-v2 per-project + layout (``base//HEAD``) — distinct from ``legacy_archives``, which + are already-migrated ``legacy-/`` dirs. Callers that preview an + orphan-deletion sweep must include both ``projects`` and + ``pre_v2_projects``, since ``prune_checkpoints`` deletes orphans from + both layouts. + """ + base = checkpoint_base or _resolve_checkpoint_base() + out: Dict = { + "base": str(base), + "store_size_bytes": 0, + "legacy_size_bytes": 0, + "total_size_bytes": 0, + "project_count": 0, + "projects": [], + "pre_v2_projects": [], + "legacy_archives": [], + } + if not base.exists(): + return out + + store = _store_path(base) + if store.exists(): + out["store_size_bytes"] = _dir_size_bytes(store) + if (store / "HEAD").exists(): + for meta in _list_projects(store): + dir_hash = meta.get("_hash") or "" + workdir = meta.get("workdir") or "" + ref = _ref_name(dir_hash) + ok, count_out, _ = _run_git( + ["rev-list", "--count", ref], store, str(base), + allowed_returncodes={128}, + ) + try: + commits = int(count_out) if ok else 0 + except ValueError: + commits = 0 + out["projects"].append({ + "hash": dir_hash, + "workdir": workdir, + "exists": bool(workdir) and Path(workdir).exists(), + "created_at": meta.get("created_at"), + "last_touch": meta.get("last_touch"), + "commits": commits, + }) + out["project_count"] = len(out["projects"]) + + out["pre_v2_projects"] = [ + { + "path": str(r["path"]), + "workdir": r["workdir"], + "exists": r["exists"], + } + for r in _pre_v2_shadow_repos(base) + ] + + for child in base.iterdir(): + if child.is_dir() and child.name.startswith(_LEGACY_PREFIX): + try: + size = _dir_size_bytes(child) + except OSError: + size = 0 + out["legacy_size_bytes"] += size + try: + mt = child.stat().st_mtime + except OSError: + mt = 0 + out["legacy_archives"].append({ + "name": child.name, + "size_bytes": size, + "mtime": mt, + }) + + out["total_size_bytes"] = _dir_size_bytes(base) + return out + + +def _rmtree_force(path: Path) -> None: + rmtree_readonly(path) + + +def clear_all(checkpoint_base: Optional[Path] = None) -> Dict[str, int]: + """Nuke the entire checkpoint base (store + legacy). Irreversible. + + Returns ``{"bytes_freed": N, "deleted": bool}``. + """ + base = checkpoint_base or _resolve_checkpoint_base() + out = {"bytes_freed": 0, "deleted": False} + if not base.exists(): + return out + size = _dir_size_bytes(base) + try: + from tools.checkpoint_pruning import store_lock + + with store_lock(base): + _rmtree_force(base) + out["bytes_freed"] = size + out["deleted"] = True + except (OSError, RuntimeError) as exc: + logger.warning("Could not clear checkpoint base %s: %s", base, exc) + return out + + +def clear_legacy(checkpoint_base: Optional[Path] = None) -> Dict[str, int]: + """Delete all ``legacy-*`` archive directories and report any failures.""" + base = checkpoint_base or _resolve_checkpoint_base() + out = {"bytes_freed": 0, "deleted": 0, "errors": 0} + if not base.exists(): + return out + from tools.checkpoint_pruning import store_lock + + try: + with store_lock(base): + for child in list(base.iterdir()): + if not child.is_dir() or not child.name.startswith(_LEGACY_PREFIX): + continue + try: + size = _dir_size_bytes(child) + _rmtree_force(child) + out["bytes_freed"] += size + out["deleted"] += 1 + except OSError as exc: + out["errors"] += 1 + logger.warning("Could not delete legacy archive %s: %s", child, exc) + except (OSError, RuntimeError) as exc: + out["errors"] += 1 + logger.warning("Could not clear legacy archives: %s", exc) + return out diff --git a/tools/checkpoint_manager.py b/tools/checkpoint_manager.py index e99913bc85..16e0a96991 100644 --- a/tools/checkpoint_manager.py +++ b/tools/checkpoint_manager.py @@ -38,8 +38,8 @@ a new worktree costs near-zero. The shadow store uses ``GIT_DIR`` + ``GIT_WORK_TREE`` + ``GIT_INDEX_FILE`` so no git state leaks into the user's project directory. -Auto-maintenance ----------------- +Auto-maintenance (``tools.checkpoint_maintenance``) +--------------------------------------------------- Shadow state accumulates over time. ``prune_checkpoints`` deletes refs whose recorded working directory no longer exists (orphan) or whose last touch is @@ -88,6 +88,7 @@ _INDEXES_DIRNAME = "indexes" _PROJECTS_DIRNAME = "projects" _LEDGERS_DIRNAME = "ledgers" _LEGACY_PREFIX = "legacy-" +_PRUNE_MARKER_NAME = ".last_prune" # Agent-write ledger cap: newest entries retained per project. _LEDGER_MAX_ENTRIES = 2000 @@ -1535,598 +1536,3 @@ def format_checkpoint_list(checkpoints: List[Dict], directory: str) -> str: lines.append(" /rollback diff preview changes since checkpoint N") lines.append(" /rollback restore a single file from checkpoint N") return "\n".join(lines) - - -# --------------------------------------------------------------------------- -# Auto-maintenance -# --------------------------------------------------------------------------- -# -# v2 rewrite. The sweep now operates on per-project refs inside the shared -# store rather than per-project shadow repos. Legacy-archive dirs -# (``legacy-/``) are swept with the same retention policy. - -_PRUNE_MARKER_NAME = ".last_prune" - - -def _delete_ref(store: Path, ref: str) -> bool: - """Delete a ref from the store. Returns True on success.""" - ok, _, _ = _run_git( - ["update-ref", "-d", ref], store, str(store.parent), - allowed_returncodes={128}, - ) - return ok - - -def _workdir_is_observably_gone( - workdir: str, - parent_dev: Optional[int] = None, - parent_ino: Optional[int] = None, - require_parent_identity: bool = True, -) -> bool: - """True only when we can positively observe that ``workdir`` was removed. - - ``Path.exists()`` returns False for a deleted directory AND for one whose - storage simply is not attached right now — an unplugged external drive, a - network share behind a downed VPN, a bind-mount absent from this - container, an offline Windows mapped drive. Orphan pruning deletes the - project's entire checkpoint history, so treating that ambiguity as - "deleted" throws away the user's restore points over a transient mount - state, unattended, at startup. - - Require corroboration, in three steps. - - First, the parent directory must be present, so the absence of the project - inside it is something we actually observed. When the parent is missing - too, the volume is not there and we know nothing. - - Second, the present parent must be the directory we knew — not merely a - directory at the same path. Unmounting swaps the directory visible at a - mount point: while the volume is attached the path resolves to the - mounted filesystem's root; after detach it resolves to the *underlying* - (underlay) directory, which may carry entries of its own (a ``.keep`` - placeholder, sibling mount points). Those entries were never next to the - project and prove nothing about the volume being attached. So the - parent's ``(st_dev, st_ino)`` must match the identity recorded in the - project's metadata while the project was observably live - (``parent_dev``/``parent_ino``). A mismatch means a different directory - is visible at that path — a detached volume, not an observed deletion. - When no identity was ever recorded (metadata written by an older - version) and ``require_parent_identity`` is True, stay conservative and - do not classify as orphan. Callers that have no identity channel at all - (the frozen pre-v2 layout) pass ``require_parent_identity=False`` to - keep the structural checks only. - - Third, the (identity-confirmed) parent must actually carry information. - Unmounting leaves classic static mount points (``/mnt/volume/proj``, an - fstab entry, a container bind-mount) behind as *empty* directories, so an - empty parent is the signature of a detached volume just as much as of a - deleted project. Prune only when the parent holds something else (we - observed a populated directory that does not contain the project) or is - itself a live mount point (the volume is demonstrably attached and the - project is demonstrably not on it). - - Genuinely abandoned projects are still reclaimed by the retention/stale - rule, which runs off ``last_touch`` rather than a filesystem probe. - """ - if not workdir: - return False - path = Path(workdir) - try: - if path.exists(): - return False - parent = path.parent - # A path whose parent is itself (a filesystem root) gives us nothing - # to corroborate against. - if parent == path: - return False - if not parent.is_dir(): - return False - if parent_dev is not None and parent_ino is not None: - st = parent.stat() - if (st.st_dev, st.st_ino) != (parent_dev, parent_ino): - # A different directory is visible at the parent's path than - # the one the project lived in — the volume is detached (its - # underlay showing through) or was swapped. Not a deletion. - return False - elif require_parent_identity: - # No recorded identity to check against — we cannot tell the - # project's real parent from an underlay directory exposed by an - # unmount. Unsure never deletes; retention still reclaims. - return False - if _dir_has_any_entry(parent): - return True - # Empty parent: only evidence if that directory is a mount point, i.e. - # the volume is attached right now and simply does not hold the - # project. An empty plain directory is an unmounted mount point as - # readily as an emptied project root. - return os.path.ismount(parent) - except OSError: - # Probe failed (permission, I/O error) — not evidence of deletion. - return False - - -def _dir_has_any_entry(directory: Path) -> bool: - """True when ``directory`` contains at least one entry. - - Stops after the first entry rather than materializing the listing; a - project root can hold a large tree. - """ - with os.scandir(directory) as entries: - for _ in entries: - return True - return False - - -def prune_checkpoints( - retention_days: int = 7, - delete_orphans: bool = True, - checkpoint_base: Optional[Path] = None, - max_total_size_mb: int = 0, - orphan_allowlist: Optional[set] = None, -) -> Dict[str, int]: - """Delete stale/orphan checkpoints and reclaim store space. - - A project entry is deleted when either: - - * ``delete_orphans=True`` and its ``workdir`` no longer exists on disk - (the original project was deleted / moved); OR - * its ``last_touch`` is older than ``retention_days`` days. - - ``orphan_allowlist``, when not ``None``, restricts orphan deletion to - the given identities (v2 project ``_hash`` strings and/or pre-v2 shadow - repo paths as ``str``). This lets a caller that showed the user a - confirmation preview (built from ``store_status()``) bind the resulting - deletion to exactly what was displayed — a project that only becomes - orphaned *after* the preview (e.g. its workdir vanishes while the human - is answering the prompt) is skipped rather than swept up under the - earlier confirmation. Pass ``None`` (the default) to delete every - currently-orphaned project, e.g. for ``--force`` or unattended callers - that never show a preview. - - Additionally, if ``max_total_size_mb > 0`` and the store exceeds that - after orphan/stale pruning, the oldest commit per remaining project is - dropped until the store is under the cap. - - Legacy-archive dirs (``legacy-*``) older than ``retention_days`` are - also deleted. - - Returns a dict with counts ``{"scanned", "deleted_orphan", - "deleted_stale", "errors", "bytes_freed"}``. - - Never raises — maintenance must never block interactive startup. - """ - base = checkpoint_base or _resolve_checkpoint_base() - result = { - "scanned": 0, - "deleted_orphan": 0, - "deleted_stale": 0, - "errors": 0, - "bytes_freed": 0, - } - if not base.exists(): - return result - - from tools.checkpoint_pruning import PruneError, store_lock - - try: - with store_lock(base): - return _prune_checkpoints(base, result, retention_days, delete_orphans, max_total_size_mb, orphan_allowlist) - except (PruneError, OSError) as exc: - result["errors"] += 1 - logger.warning("Checkpoint maintenance stopped: %s", exc) - return result - - -def _prune_checkpoints( - base: Path, result: Dict[str, int], retention_days: int, delete_orphans: bool, - max_total_size_mb: int, orphan_allowlist: Optional[set], -) -> Dict[str, int]: - size_before = _dir_size_bytes(base) - - # --- Legacy pre-v2 per-project shadow repos (kept directly under base) --- - # Pre-v2 layout: ``base//HEAD`` etc. We treat these exactly as the - # v1 pruner did so behaviour is unchanged for anyone still on that layout - # or sitting on a mid-migration system. - cutoff = 0.0 - if retention_days > 0: - cutoff = time.time() - retention_days * 86400 - - for child in base.iterdir(): - if not child.is_dir(): - continue - if child.name == _STORE_DIRNAME: - continue - if child.name.startswith(_LEGACY_PREFIX): - # Legacy archive: prune by dir mtime using same retention rule. - if retention_days <= 0: - continue - try: - m = child.stat().st_mtime - except OSError: - continue - if m >= cutoff: - continue - try: - size = _dir_size_bytes(child) - rmtree_readonly(child) - result["bytes_freed"] += size - result["deleted_stale"] += 1 - except OSError as exc: - result["errors"] += 1 - logger.warning("Failed to delete legacy archive %s: %s", child, exc) - - # Pre-v2 per-project shadow repos. Scanned via the same helper - # `store_status()` uses for its orphan preview, so a confirmation prompt - # built from that preview always matches what gets deleted here. - for repo in _pre_v2_shadow_repos(base): - child = repo["path"] - result["scanned"] += 1 - reason: Optional[str] = None - if ( - delete_orphans - and not repo["marker_unreadable"] - and ( - repo["workdir"] is None - # The frozen pre-v2 layout has no metadata channel to carry a - # recorded parent identity, so only the structural checks - # (parent present + populated / live mount point) apply here. - or _workdir_is_observably_gone( - repo["workdir"], require_parent_identity=False, - ) - ) - and (orphan_allowlist is None or str(child) in orphan_allowlist) - ): - reason = "orphan" - if reason is None and retention_days > 0: - newest = 0.0 - try: - for p in child.rglob("*"): - try: - mt = p.stat().st_mtime - newest = max(newest, mt) - except OSError: - continue - except OSError: - pass - if newest > 0 and newest < cutoff: - reason = "stale" - if reason is None: - continue - try: - size = _dir_size_bytes(child) - rmtree_readonly(child) - result["bytes_freed"] += size - if reason == "orphan": - result["deleted_orphan"] += 1 - else: - result["deleted_stale"] += 1 - except OSError as exc: - result["errors"] += 1 - logger.warning("Failed to prune checkpoint repo %s: %s", child.name, exc) - - # --- v2 shared store: per-project ref pruning via metadata --- - store = _store_path(base) - if (store / "HEAD").exists(): - # A gc killed by the store timeout strands tmp_pack_* files that gc.auto=0 means git - # itself never reclaims; sweep them even when no ref moved (a sweep is a directory - # listing, unlike the pack-rewriting gc gated on refs below). - clear_stale_tmp_packs(store) - for meta in _list_projects(store): - dir_hash = meta.get("_hash") or "" - workdir = meta.get("workdir") or "" - if not dir_hash: - continue - result["scanned"] += 1 - reason = None - parent_dev = meta.get("workdir_parent_dev") - parent_ino = meta.get("workdir_parent_ino") - if not isinstance(parent_dev, int) or isinstance(parent_dev, bool): - parent_dev = None - if not isinstance(parent_ino, int) or isinstance(parent_ino, bool): - parent_ino = None - if ( - delete_orphans - and ( - not workdir - or _workdir_is_observably_gone( - workdir, - parent_dev=parent_dev, - parent_ino=parent_ino, - ) - ) - and (orphan_allowlist is None or dir_hash in orphan_allowlist) - ): - reason = "orphan" - elif retention_days > 0: - last_touch = float(meta.get("last_touch", 0) or 0) - if last_touch > 0 and last_touch < cutoff: - reason = "stale" - if reason is None: - continue - ref = _ref_name(dir_hash) - if not _delete_ref(store, ref): - result["errors"] += 1 - return result - # Drop per-project index and metadata. - try: - idx = _index_path(store, dir_hash) - if idx.exists(): - idx.unlink() - except OSError: - pass - try: - mp = _project_meta_path(store, dir_hash) - if mp.exists(): - mp.unlink() - except OSError: - pass - if reason == "orphan": - result["deleted_orphan"] += 1 - else: - result["deleted_stale"] += 1 - - from tools.checkpoint_pruning import Pruner, PruneError - - pruner = Pruner(_run_git, store, str(base), _GIT_TIMEOUT, _dir_size_bytes, _REFS_PREFIX) - try: - if result["deleted_orphan"] + result["deleted_stale"] or pruner.gc_pending(): - pruner.reclaim() - if not pruner.enforce_size(max_total_size_mb * 1024 * 1024): - result["errors"] += 1 - logger.warning("Checkpoint store remains over its size cap; minimum history retained") - except (PruneError, OSError) as exc: - result["errors"] += 1 - logger.warning("Checkpoint pruning stopped: %s", exc) - - size_after = _dir_size_bytes(base) - delta = size_before - size_after - result["bytes_freed"] = max(result["bytes_freed"], delta) - - return result - - -def maybe_auto_prune_checkpoints( - retention_days: int = 7, - min_interval_hours: int = 24, - delete_orphans: bool = True, - checkpoint_base: Optional[Path] = None, - max_total_size_mb: int = 0, -) -> Dict[str, object]: - """Idempotent wrapper around ``prune_checkpoints`` for startup hooks. - - Writes ``CHECKPOINT_BASE/.last_prune`` on completion so subsequent - calls within ``min_interval_hours`` short-circuit. - - Returns ``{"skipped": bool, "result": prune_checkpoints-dict, - "error": optional str}``. - """ - base = checkpoint_base or _resolve_checkpoint_base() - out: Dict[str, object] = {"skipped": False} - - try: - if not base.exists(): - out["result"] = { - "scanned": 0, "deleted_orphan": 0, "deleted_stale": 0, - "errors": 0, "bytes_freed": 0, - } - return out - - marker = base / _PRUNE_MARKER_NAME - now = time.time() - try: - if marker.exists() and now - float(marker.read_text(encoding="utf-8-sig").strip()) < min_interval_hours * 3600: - out["skipped"] = True - return out - except (OSError, ValueError): - pass # corrupt marker — treat as no prior run - # Claim the interval before pruning: callers run on a periodic tick, and a prune that - # dies mid-way must cost one skipped day, not a git gc every tick. - try: - marker.write_text(str(now), encoding="utf-8") - except OSError as exc: - logger.debug("Could not write checkpoint prune marker: %s", exc) - result = out["result"] = prune_checkpoints(retention_days=retention_days, delete_orphans=delete_orphans, - checkpoint_base=base, max_total_size_mb=max_total_size_mb) - - total = result["deleted_orphan"] + result["deleted_stale"] - if total > 0: - logger.info( - "checkpoint auto-maintenance: pruned %d entry(ies) " - "(%d orphan, %d stale), reclaimed %.1f MB", - total, - result["deleted_orphan"], - result["deleted_stale"], - result["bytes_freed"] / (1024 * 1024), - ) - except Exception as exc: - logger.warning("checkpoint auto-maintenance failed: %s", exc) - out["error"] = str(exc) - - return out - - -def auto_prune_from_config() -> Dict[str, object]: - """``maybe_auto_prune_checkpoints`` driven by the ``checkpoints:`` config section — the one - startup/housekeeping entry point for the CLI and the gateway. ``delete_orphans`` is never - honoured unattended: a missing workdir is ambiguous (deleted vs. unmounted share); orphan - cleanup is only via explicit ``hermes checkpoints prune``. Never raises.""" - try: - from hermes_cli.config import load_config - cfg = load_config().get("checkpoints") or {} - if not cfg.get("auto_prune", False): - return {"skipped": True} - return maybe_auto_prune_checkpoints( - retention_days=int(cfg.get("retention_days", 7)), - min_interval_hours=int(cfg.get("min_interval_hours", 24)), - delete_orphans=False, - max_total_size_mb=int(cfg.get("max_total_size_mb", 500))) - except Exception as exc: - logger.debug("checkpoint auto-maintenance skipped: %s", exc) - return {"skipped": True, "error": str(exc)} - - -def checkpoint_footprint_notice() -> Optional[str]: - """One-line notice when ``/rollback`` checkpoints are on and their store sits at or above - ``checkpoints.max_total_size_mb``, else None. Checkpoints were on by default for a while - (Mar–May 2026) and that ``enabled: true`` persisted into user configs; many users carry a - GB-scale store for a feature they never invoke. The cap is a floor of one snapshot per - project, so a big store is expected, not broken — the notice names the opt-out. Never raises.""" - try: - from hermes_cli.config import load_config - cfg = load_config().get("checkpoints") or {} - if not cfg.get("enabled", False): - return None - cap_mb = int(cfg.get("max_total_size_mb", 500) or 0) - status = store_status() - size = int(status["total_size_bytes"]) - if cap_mb <= 0 or size < cap_mb * 1024 * 1024: - return None - from hermes_cli.sizefmt import format_bytes - return (f"Filesystem checkpoints (/rollback) are on: {format_bytes(size)} across " - f"{status['project_count']} project(s), above the {cap_mb} MB cap (one snapshot per project is " - f"always kept). Not using /rollback? `hermes config set checkpoints.enabled false` then " - f"`hermes checkpoints clear`; or lower `checkpoints.retention_days`.") - except Exception as exc: - logger.debug("checkpoint footprint notice skipped: %s", exc) - return None - - -# --------------------------------------------------------------------------- -# Public helpers for `hermes checkpoints` CLI -# --------------------------------------------------------------------------- - -def store_status(checkpoint_base: Optional[Path] = None) -> Dict: - """Return a summary of the shadow store. - - ``{"base": path, "store_size_bytes": N, "legacy_size_bytes": N, - "total_size_bytes": N, "project_count": N, "projects": [...], - "pre_v2_projects": [...], "legacy_archives": [...]}`` - - ``pre_v2_projects`` covers shadow repos still on the pre-v2 per-project - layout (``base//HEAD``) — distinct from ``legacy_archives``, which - are already-migrated ``legacy-/`` dirs. Callers that preview an - orphan-deletion sweep must include both ``projects`` and - ``pre_v2_projects``, since ``prune_checkpoints`` deletes orphans from - both layouts. - """ - base = checkpoint_base or _resolve_checkpoint_base() - out: Dict = { - "base": str(base), - "store_size_bytes": 0, - "legacy_size_bytes": 0, - "total_size_bytes": 0, - "project_count": 0, - "projects": [], - "pre_v2_projects": [], - "legacy_archives": [], - } - if not base.exists(): - return out - - store = _store_path(base) - if store.exists(): - out["store_size_bytes"] = _dir_size_bytes(store) - if (store / "HEAD").exists(): - for meta in _list_projects(store): - dir_hash = meta.get("_hash") or "" - workdir = meta.get("workdir") or "" - ref = _ref_name(dir_hash) - ok, count_out, _ = _run_git( - ["rev-list", "--count", ref], store, str(base), - allowed_returncodes={128}, - ) - try: - commits = int(count_out) if ok else 0 - except ValueError: - commits = 0 - out["projects"].append({ - "hash": dir_hash, - "workdir": workdir, - "exists": bool(workdir) and Path(workdir).exists(), - "created_at": meta.get("created_at"), - "last_touch": meta.get("last_touch"), - "commits": commits, - }) - out["project_count"] = len(out["projects"]) - - out["pre_v2_projects"] = [ - { - "path": str(r["path"]), - "workdir": r["workdir"], - "exists": r["exists"], - } - for r in _pre_v2_shadow_repos(base) - ] - - for child in base.iterdir(): - if child.is_dir() and child.name.startswith(_LEGACY_PREFIX): - try: - size = _dir_size_bytes(child) - except OSError: - size = 0 - out["legacy_size_bytes"] += size - try: - mt = child.stat().st_mtime - except OSError: - mt = 0 - out["legacy_archives"].append({ - "name": child.name, - "size_bytes": size, - "mtime": mt, - }) - - out["total_size_bytes"] = _dir_size_bytes(base) - return out - - -def _rmtree_force(path: Path) -> None: - rmtree_readonly(path) - - -def clear_all(checkpoint_base: Optional[Path] = None) -> Dict[str, int]: - """Nuke the entire checkpoint base (store + legacy). Irreversible. - - Returns ``{"bytes_freed": N, "deleted": bool}``. - """ - base = checkpoint_base or _resolve_checkpoint_base() - out = {"bytes_freed": 0, "deleted": False} - if not base.exists(): - return out - size = _dir_size_bytes(base) - try: - from tools.checkpoint_pruning import store_lock - - with store_lock(base): - _rmtree_force(base) - out["bytes_freed"] = size - out["deleted"] = True - except (OSError, RuntimeError) as exc: - logger.warning("Could not clear checkpoint base %s: %s", base, exc) - return out - - -def clear_legacy(checkpoint_base: Optional[Path] = None) -> Dict[str, int]: - """Delete all ``legacy-*`` archive directories and report any failures.""" - base = checkpoint_base or _resolve_checkpoint_base() - out = {"bytes_freed": 0, "deleted": 0, "errors": 0} - if not base.exists(): - return out - from tools.checkpoint_pruning import store_lock - - try: - with store_lock(base): - for child in list(base.iterdir()): - if not child.is_dir() or not child.name.startswith(_LEGACY_PREFIX): - continue - try: - size = _dir_size_bytes(child) - _rmtree_force(child) - out["bytes_freed"] += size - out["deleted"] += 1 - except OSError as exc: - out["errors"] += 1 - logger.warning("Could not delete legacy archive %s: %s", child, exc) - except (OSError, RuntimeError) as exc: - out["errors"] += 1 - logger.warning("Could not clear legacy archives: %s", exc) - return out diff --git a/tools/environments/local.py b/tools/environments/local.py index 132e101d30..df94391845 100644 --- a/tools/environments/local.py +++ b/tools/environments/local.py @@ -314,40 +314,13 @@ def _sanitize_subprocess_env(base_env: dict | None, extra_env: dict | None = Non def hermes_subprocess_env( *, inherit_credentials: bool = False, base_env: dict[str, str] | None = None ) -> dict[str, str]: - """Build a sanitized environment dict for a spawned subprocess. + """Sanitize a non-terminal child's environment (no skill passthrough). - Centralized helper for the **non-terminal** spawn surface (browser, - ACP/CLI executors, computer-use driver, dep-ensure, TUI Node host, - detached gateway). Use this instead of copying ``os.environ`` directly - so strip-by-default is the uniform policy across every spawn site, with a - single source of truth (``_HERMES_PROVIDER_ENV_BLOCKLIST``). The terminal - / execute_code path keeps using :func:`_sanitize_subprocess_env`, which is - skill-aware (``env_passthrough``); this helper is for spawns that have no - skill-passthrough concept. - - Two-tier stripping: - - * **Tier 1 (always):** ``_ALWAYS_STRIP_KEYS`` — gateway bot tokens, GitHub - auth, and remote-compute secrets are removed regardless of - ``inherit_credentials``. No child Hermes spawns legitimately needs them. - * **Tier 2 (conditional):** the rest of ``_HERMES_PROVIDER_ENV_BLOCKLIST`` - (LLM provider API keys, tool secrets) is removed unless the caller passes - ``inherit_credentials=True``. - - Pass ``inherit_credentials=True`` **only** when the child legitimately - needs LLM provider credentials — a user-blessed ``claude`` / ``codex`` / - ``gemini`` CLI executor, or the TUI Node host that makes model calls. The - flag is grep-able for audit: ``grep -rn 'inherit_credentials=True'`` lists - every spawn site that still receives provider credentials. - - Callers that need a *specific* non-provider secret (e.g. the browser worker - needs ``BROWSERBASE_API_KEY`` / ``FIRECRAWL_API_KEY``) should call with - ``inherit_credentials=False`` and copy just those keys back from - ``os.environ`` into the returned dict. - - ``base_env`` swaps the starting environment (default ``os.environ``) — - for callers that already hold a curated env (pm's sanitized uv env) and - want the strip policy applied on top of it. + Bot, GitHub and remote-compute secrets never pass through; provider/tool + credentials pass only with ``inherit_credentials=True`` for children that + need them. Callers needing one other secret should add only that key back. + ``base_env`` lets an already curated environment use the same policy. + Terminal and execute_code spawns use the skill-aware sanitizer instead. """ env = dict(base_env) if base_env is not None else os.environ.copy() env = _scrub_credentials(env, inherit_credentials=inherit_credentials)