242 lines
9.1 KiB
Python
242 lines
9.1 KiB
Python
"""Cross-process mutual exclusion for in-flight Hermes updates.
|
|
|
|
Until now only the Tauri updater published an "update in progress" marker (``UpdateMarkerGuard`` in
|
|
``apps/bootstrap-installer/src-tauri/src/update.rs``), and only the Electron desktop consumed it
|
|
(``electron/update-marker.ts``, to gate local backend startup).
|
|
|
|
This module makes that same marker the single lock for **all** update entrypoints instead of adding
|
|
a fourth mechanism. Format and location are unchanged and remain byte-compatible with the Rust and
|
|
Electron readers:
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import os
|
|
import time
|
|
from dataclasses import dataclass
|
|
from pathlib import Path
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Keep in sync with UPDATE_MARKER_MAX_AGE_MS in
|
|
# apps/desktop/electron/update-marker.ts — the same marker is read by both, and
|
|
# a shorter ceiling here would let Python steal a lock Electron still considers
|
|
# live. A full update (git pull + uv sync + desktop rebuild) is minutes.
|
|
UPDATE_MARKER_MAX_AGE_SECONDS = 20 * 60
|
|
|
|
MARKER_NAME = ".hermes-update-in-progress"
|
|
|
|
# Set by an orchestrating updater (the Tauri `hermes-setup --update` flow) to
|
|
# its own pid before spawning `hermes update` as a child stage. The parent
|
|
# holds the marker for its whole run, so without this the child refuses its
|
|
# own parent's lock and the GUI update can never complete. See update_child_env
|
|
# in apps/bootstrap-installer/src-tauri/src/update.rs — keep the name in sync.
|
|
HANDOFF_PID_ENV = "HERMES_UPDATE_HANDOFF_PID"
|
|
|
|
# Exit code meaning "another updater/instance owns this install right now".
|
|
# Already the de-facto contract: the Windows shim + venv-holder guards in
|
|
# _cmd_update_impl exit 2, and the Tauri updater matches on it
|
|
# (UPDATE_EXIT_CONCURRENT in apps/bootstrap-installer/src-tauri/src/update.rs)
|
|
# to show "Hermes is still running" instead of a generic failure. Naming it
|
|
# here keeps the concurrent-update refusal on that same understood contract.
|
|
UPDATE_EXIT_CONCURRENT = 2
|
|
|
|
|
|
def update_marker_path() -> Path:
|
|
"""Path of the shared update marker.
|
|
|
|
Uses the *process* Hermes home (never the context-local profile override): the Rust updater
|
|
resolves ``$HERMES_HOME`` or the platform default, and the desktop pins that same value into the
|
|
updater's env. A profile-scoped path here would put the lock somewhere the other two owners
|
|
never look.
|
|
"""
|
|
from hermes_constants import get_process_hermes_home
|
|
|
|
return get_process_hermes_home() / MARKER_NAME
|
|
|
|
|
|
def _pid_alive(pid: int) -> bool:
|
|
"""True when a process with ``pid`` currently exists.
|
|
|
|
Delegates to :func:`gateway.status._pid_exists`, the project's existing no-kill probe. Do NOT
|
|
hand-roll this with ``os.kill(pid, 0)``: on Windows that is not a no-op — CPython routes
|
|
``sig=0`` to ``GenerateConsoleCtrlEvent``, which Ctrl+C's the target's whole console process
|
|
group (bpo-14484).
|
|
|
|
Any pid we cannot evaluate counts as dead: a corrupt marker must not wedge the lock forever.
|
|
"""
|
|
if pid <= 0:
|
|
return False
|
|
try:
|
|
from gateway.status import _pid_exists
|
|
|
|
return bool(_pid_exists(pid))
|
|
except Exception as exc:
|
|
# Import failure or an unusable pid (e.g. larger than the platform's
|
|
# pid_t). Treat the marker as stale rather than blocking updates.
|
|
logger.debug("Could not probe pid %s: %s", pid, exc)
|
|
return False
|
|
|
|
|
|
def _handoff_pid() -> int | None:
|
|
"""Pid of the orchestrating updater that spawned us, if any.
|
|
|
|
Read from :data:`HANDOFF_PID_ENV`. Malformed values count as absent — a broken handoff must fall
|
|
back to the normal refusal, never crash.
|
|
"""
|
|
raw = os.environ.get(HANDOFF_PID_ENV, "").strip()
|
|
if not raw:
|
|
return None
|
|
try:
|
|
pid = int(raw)
|
|
except ValueError:
|
|
return None
|
|
return pid if pid > 0 else None
|
|
|
|
|
|
def _is_ancestor_pid(pid: int) -> bool:
|
|
"""True when ``pid`` is a live ancestor (parent chain) of this process.
|
|
|
|
The orchestrating updater spawns ``hermes update`` as a (grand)child, so a live marker owned by
|
|
one of our ancestors can only be the claim we are already running under — an unrelated
|
|
concurrent updater is never in our parent chain.
|
|
|
|
Never includes our own pid, and any failure counts as "not an ancestor": an unprovable ancestry
|
|
must fall back to the normal refusal.
|
|
"""
|
|
if pid <= 0:
|
|
return False
|
|
try:
|
|
import psutil
|
|
|
|
return any(parent.pid == pid for parent in psutil.Process().parents())
|
|
except Exception as exc:
|
|
logger.debug("Could not walk process ancestry for pid %s: %s", pid, exc)
|
|
return False
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class UpdateHolder:
|
|
"""A confirmed-live update currently holding the lock."""
|
|
|
|
pid: int
|
|
age_seconds: float
|
|
|
|
|
|
def read_live_update(*, path: Path | None = None) -> UpdateHolder | None:
|
|
"""Return the live update holding the lock, or ``None``.
|
|
|
|
Mirrors ``readLiveUpdateMarker`` in ``electron/update-marker.ts``: absent, unreadable,
|
|
malformed, dead-pid, and past-the-ceiling all mean "no live update", and a stale marker file is
|
|
deleted so it can't strand future runs. Never raises.
|
|
"""
|
|
marker = path or update_marker_path()
|
|
try:
|
|
raw = marker.read_text(encoding="utf-8")
|
|
except OSError:
|
|
return None # absent or unreadable => no live update
|
|
|
|
lines = raw.splitlines()
|
|
try:
|
|
pid = int(lines[0].strip())
|
|
except (IndexError, ValueError):
|
|
pid = -1
|
|
try:
|
|
started_at = float(lines[1].strip())
|
|
except (IndexError, ValueError):
|
|
started_at = float("-inf")
|
|
|
|
age = time.time() - started_at
|
|
if not _pid_alive(pid) or age > UPDATE_MARKER_MAX_AGE_SECONDS:
|
|
try:
|
|
marker.unlink()
|
|
except OSError:
|
|
pass
|
|
return None
|
|
|
|
return UpdateHolder(pid=pid, age_seconds=age)
|
|
|
|
|
|
def describe_holder(holder: UpdateHolder) -> str:
|
|
"""One-line, user-facing explanation of who holds the update lock."""
|
|
minutes, seconds = divmod(int(max(holder.age_seconds, 0)), 60)
|
|
elapsed = f"{minutes}m {seconds}s" if minutes else f"{seconds}s"
|
|
return (
|
|
f"✗ Another Hermes update is already running (PID {holder.pid}, "
|
|
f"started {elapsed} ago).\n"
|
|
"\n"
|
|
" Two updates mutating the same checkout corrupt it: one rewrites\n"
|
|
" source while the other is mid-install. Wait for it to finish, or\n"
|
|
" close the window/dashboard tab that started it, then retry."
|
|
)
|
|
|
|
|
|
class UpdateLock:
|
|
"""Context manager owning the shared update marker for this process.
|
|
|
|
``acquired`` is False when another live update holds it; callers decide between hard refusal
|
|
(CLI/dashboard) and waiting. Release only removes the marker when *we* still own it, so a marker
|
|
rewritten by a handoff partner (the Tauri updater writes its own pid) is never deleted from
|
|
under its new owner.
|
|
"""
|
|
|
|
def __init__(self, *, path: Path | None = None) -> None:
|
|
self.path = path or update_marker_path()
|
|
self.acquired = False
|
|
self.holder: UpdateHolder | None = None
|
|
|
|
def acquire(self) -> bool:
|
|
"""Claim the lock. Returns False (and sets ``holder``) if it's taken.
|
|
|
|
A live holder whose pid matches :data:`HANDOFF_PID_ENV` — or is an ancestor of ours — is our
|
|
own orchestrating parent (e.g. the Tauri updater staging ``hermes update``): run under ITS
|
|
claim and leave its marker untouched on release. The ancestry path covers staged updaters
|
|
older than the env-var export.
|
|
"""
|
|
existing = read_live_update(path=self.path)
|
|
if existing is not None:
|
|
if existing.pid == _handoff_pid() or _is_ancestor_pid(existing.pid):
|
|
return True
|
|
self.holder = existing
|
|
return False
|
|
try:
|
|
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
self.path.write_text(
|
|
f"{os.getpid()}\n{int(time.time())}\n", encoding="utf-8"
|
|
)
|
|
except OSError as exc:
|
|
# Best-effort, exactly like the Rust guard: an unwritable marker
|
|
# must not block the update itself (that would be a worse failure
|
|
# than the race it prevents). Degrade to the pre-lock behavior.
|
|
logger.debug("Could not write update marker %s: %s", self.path, exc)
|
|
return True
|
|
self.acquired = True
|
|
return True
|
|
|
|
def release(self) -> None:
|
|
"""Drop the marker if this process still owns it. Never raises."""
|
|
if not self.acquired:
|
|
return
|
|
self.acquired = False
|
|
try:
|
|
raw = self.path.read_text(encoding="utf-8")
|
|
owner = int(raw.splitlines()[0].strip())
|
|
except (OSError, IndexError, ValueError):
|
|
return
|
|
if owner != os.getpid():
|
|
# A handoff partner took ownership (e.g. the Tauri updater wrote
|
|
# its own pid). Leave it alone — it's still a live update.
|
|
return
|
|
try:
|
|
self.path.unlink()
|
|
except OSError:
|
|
pass
|
|
|
|
def __enter__(self) -> "UpdateLock":
|
|
self.acquire()
|
|
return self
|
|
|
|
def __exit__(self, *_exc) -> None:
|
|
self.release()
|