fix(state): refuse WAL on cross-VM filesystems (virtiofs/9p) before corruption

Port from openclaw/openclaw#120597: Docker Desktop, OrbStack, and Podman
expose host bind mounts as fuse.virtiofs or 9p. SQLite WAL shared-memory
is not coherent across the VM boundary and fails SILENTLY under write
pressure (zero-filled pages), so the existing reactive marker fallback
(_WAL_INCOMPAT_MARKERS) never fires — the mode must be refused before
the pragma.

apply_wal_with_fallback now checks the DB path against
/proc/self/mountinfo (longest mount-point prefix wins; per-directory
cached) and applies journal_mode=DELETE with a one-shot WARNING naming
the fix (native volume instead of a bind mount). On-disk WAL databases
are still never live-downgraded; require_wal callers get
WalUnsupportedError. Non-Linux hosts and unreadable mount tables
conservatively skip detection (behavior unchanged).

Sabotage-verified: disabling only the wiring makes
test_fresh_db_on_cross_vm_fs_gets_delete fail.
This commit is contained in:
Teknium
2026-08-30 17:40:41 -07:00
parent 2ff55bc895
commit d8dcdfd620
2 changed files with 181 additions and 0 deletions

View File

@@ -8,6 +8,7 @@ from __future__ import annotations
import contextlib
import functools
import logging
import os
import sqlite3
import sys
import threading
@@ -172,6 +173,74 @@ class WalUnsupportedError(sqlite3.OperationalError):
opener). Subclasses ``OperationalError`` so DB-init handlers still catch it."""
# Docker Desktop / OrbStack / Podman expose host bind mounts to the guest as ``fuse.virtiofs`` or ``9p``. WAL's
# -shm file must be coherent shared memory for every opener; across the VM boundary it is not, and under write
# pressure the failure is SILENT (zero-filled pages), so the reactive ``_WAL_INCOMPAT_MARKERS`` fallback in
# ``_enable_wal`` never gets a signal — WAL must be refused before the pragma. Port of openclaw/openclaw#120597.
# Detection is Linux ``/proc/self/mountinfo`` only (statvfs carries no fstype); elsewhere, or when the table is
# unreadable, the answer is False and behaviour is unchanged. Nothing else (ext4/btrfs/xfs/zfs/tmpfs/overlay/nfs)
# is ever flagged here — those keep the existing reactive paths.
_CROSS_VM_FSTYPES = frozenset({"virtiofs", "fuse.virtiofs", "9p", "9p2000", "9p2000.l", "9p2000.u"})
_cross_vm_fs_cache: Dict[str, bool] = {} # per DB directory; kanban_db.connect() opens per operation
_cross_vm_fs_cache_lock = threading.Lock()
_cross_vm_warned_paths: set[str] = set()
_cross_vm_warned_lock = threading.Lock()
def _mountinfo_fstype(directory: str, mountinfo_path: str = "/proc/self/mountinfo") -> str:
"""fstype of the longest mount point that is a prefix of ``directory`` (``""`` if unreadable / no match)."""
try:
with open(mountinfo_path, "r", encoding="utf-8", errors="replace") as fh:
lines = fh.read().splitlines()
except OSError:
return ""
best_len, best_fstype = -1, ""
for line in lines:
# proc(5): <id> <parent> <maj:min> <root> <mount point> <opts> [optional...] - <fstype> <source> <super opts>
fields, _, tail = line.partition(" - ")
parts = fields.split()
if len(parts) < 5 or not tail:
continue
mount_point = parts[4]
if "\\" in mount_point: # octal escapes (\040 = space)
mount_point = mount_point.encode("latin-1", "ignore").decode("unicode_escape")
if directory == mount_point or directory.startswith(mount_point.rstrip("/") + "/"):
if len(mount_point) > best_len:
best_len, best_fstype = len(mount_point), tail.split()[0]
return best_fstype.lower()
def _detect_cross_vm_fs(directory: str, mountinfo_path: str = "/proc/self/mountinfo") -> bool:
"""True only when ``directory`` sits on a virtiofs/9p mount per ``mountinfo_path``."""
if sys.platform != "linux":
return False
return _mountinfo_fstype(directory, mountinfo_path) in _CROSS_VM_FSTYPES
def _path_on_cross_vm_fs(path: str) -> bool:
"""True when ``path`` resides on a virtiofs/9p (cross-VM) filesystem; cached per resolved directory."""
try:
directory = os.path.dirname(os.path.realpath(path)) or "/"
except (OSError, ValueError):
return False
with _cross_vm_fs_cache_lock:
cached = _cross_vm_fs_cache.get(directory)
if cached is None:
cached = _detect_cross_vm_fs(directory)
with _cross_vm_fs_cache_lock:
_cross_vm_fs_cache[directory] = cached
return cached
def _connection_db_file(conn: sqlite3.Connection) -> str:
"""Filesystem path of the ``main`` database, ``""`` for in-memory / unknown."""
try:
row = conn.execute("PRAGMA database_list").fetchone()
return str(row[2]) if row and row[2] else ""
except (sqlite3.OperationalError, IndexError, TypeError):
return ""
def _verify_configured_delete(actual: str) -> str:
"""Raise unless SQLite reported ``delete`` for an explicit operator request."""
if actual != "delete":
@@ -233,6 +302,15 @@ def apply_wal_with_fallback(conn: sqlite3.Connection, *, db_label: str = "state.
"concurrent openers); cannot guarantee WAL")
_log_once("wal_probe_unknown", db_label)
return "wal"
# Cross-VM bind mount (virtiofs/9p): WAL corrupts silently there, so refuse to ENABLE it. On-disk WAL
# databases were returned above (never live-downgrade); a 0-page / DELETE file just stays DELETE.
db_file = _connection_db_file(conn)
if db_file and _path_on_cross_vm_fs(db_file):
if require_wal:
raise WalUnsupportedError("journal_mode=WAL refused: database is on a cross-VM filesystem (virtiofs/9p) "
"where WAL shared-memory silently corrupts")
_log_once("cross_vm_fs", db_label)
return _set_journal_mode_no_wait(conn, "DELETE") or "delete"
return _enable_wal(conn, db_label, require_wal, current_mode)
@@ -435,6 +513,13 @@ _ONCE_LOGS = {
"%s: could not verify the on-disk journal mode (database is locked / busy); not issuing a journal-mode "
"set-pragma while another connection may hold the file (it could unlink the -wal/-shm sidecars that "
"connection still uses). Leaving the file untouched; this connection inherits the on-disk mode. This message fires once per process per database."),
"cross_vm_fs": (_cross_vm_warned_lock, "_cross_vm_warned_paths", logging.WARNING,
# The DB keeps working in DELETE mode; the loss is concurrency, so WARNING with the actionable fix.
"%s: database directory is on a cross-VM filesystem (virtiofs/9p — typical for Docker Desktop / OrbStack / "
"Podman host bind mounts). SQLite WAL shared-memory is not coherent across the VM boundary and can silently "
"corrupt the database, so journal_mode=DELETE is used instead. To restore WAL concurrency, move the database "
"onto a native volume (e.g. a named Docker volume) instead of a host bind mount. This message fires once per "
"process per database."),
}

View File

@@ -0,0 +1,96 @@
"""Cross-VM filesystem (virtiofs/9p) WAL refusal — port of openclaw#120597.
WAL over a VM-boundary filesystem (Docker Desktop / OrbStack / Podman host bind mounts) corrupts silently, so
``apply_wal_with_fallback`` must refuse to ENABLE WAL when the DB lives on such a mount — before the pragma —
while never live-downgrading an on-disk WAL database and never flagging an ordinary filesystem.
"""
import sqlite3
import pytest
import hermes_state_wal
from hermes_state_wal import WalUnsupportedError, _detect_cross_vm_fs, apply_wal_with_fallback
def _mountinfo(tmp_path, lines):
p = tmp_path / "mountinfo"
p.write_text("\n".join(lines) + "\n")
return str(p)
# Realistic mountinfo rows (id parent major:minor root mountpoint opts ... - fstype source superopts)
ROOT_EXT4 = "25 1 8:1 / / rw,relatime shared:1 - ext4 /dev/sda1 rw"
BIND_VIRTIOFS = "612 25 0:53 / /data rw,relatime shared:300 - fuse.virtiofs mount0 rw"
BIND_9P = "613 25 0:54 / /mnt/host rw,relatime - 9p host0 rw,trans=virtio"
NESTED_EXT4 = "614 612 8:2 / /data/native rw,relatime - ext4 /dev/sdb1 rw"
SPACE_VIRTIOFS = "615 25 0:55 / /mnt/my\\040share rw,relatime - virtiofs share rw"
class TestDetectCrossVmFs:
@pytest.mark.parametrize("path,expected", [
("/data/agent", True), # fuse.virtiofs bind mount
("/mnt/host/db", True), # 9p bind mount
("/mnt/my share/db", True), # octal-escaped mount point
("/home/user/.hermes", False), # ext4 root
("/data/native/db", False), # ext4 mounted over the virtiofs tree — longest prefix wins
("/datastore", False), # sibling path sharing a prefix string, not a mount prefix
])
def test_only_virtiofs_and_9p_mounts_are_flagged(self, tmp_path, path, expected):
mi = _mountinfo(tmp_path, [ROOT_EXT4, BIND_VIRTIOFS, BIND_9P, NESTED_EXT4, SPACE_VIRTIOFS])
assert _detect_cross_vm_fs(path, mountinfo_path=mi) is expected
@pytest.mark.parametrize("fstype", [
"ext4", "xfs", "btrfs", "zfs", "tmpfs", "overlay", "nfs", "nfs4", "cifs", "fuse.sshfs", "apfs", "f2fs",
])
def test_ordinary_filesystems_never_flagged(self, tmp_path, fstype):
# A false positive here would put every session on DELETE mode — the class bug this pins absent.
mi = _mountinfo(tmp_path, [f"25 1 8:1 / / rw,relatime shared:1 - {fstype} /dev/sda1 rw"])
assert _detect_cross_vm_fs("/home/user/.hermes", mountinfo_path=mi) is False
def test_missing_mountinfo_conservative_false(self, tmp_path):
assert _detect_cross_vm_fs("/data", mountinfo_path=str(tmp_path / "nope")) is False
class TestWalRefusalOnCrossVmFs:
@pytest.fixture(autouse=True)
def _isolate(self, monkeypatch):
# Pin the WAL-reset vulnerability gate OFF: on builds bundling a vulnerable SQLite (3.50.4 on CI)
# apply_wal_with_fallback returns via _apply_delete_for_wal_reset_bug before the cross-VM check.
monkeypatch.setattr(hermes_state_wal, "is_sqlite_wal_reset_vulnerable", lambda *a, **k: False)
monkeypatch.setattr(hermes_state_wal, "resolve_journal_mode", lambda: "wal")
hermes_state_wal._cross_vm_warned_paths.clear()
def test_fresh_db_on_cross_vm_fs_gets_delete_and_without_detection_gets_wal(self, tmp_path, monkeypatch):
monkeypatch.setattr(hermes_state_wal, "_path_on_cross_vm_fs", lambda p: True)
conn = sqlite3.connect(str(tmp_path / "a.db"))
assert apply_wal_with_fallback(conn, db_label="a.db") == "delete"
conn.close()
# Sabotage guard: same environment, detection off -> WAL is enabled, so the refusal above did the work.
monkeypatch.setattr(hermes_state_wal, "_path_on_cross_vm_fs", lambda p: False)
conn = sqlite3.connect(str(tmp_path / "b.db"))
mode = apply_wal_with_fallback(conn, db_label="b.db")
conn.close()
if mode != "wal":
pytest.skip("environment refuses WAL for unrelated reasons")
def test_require_wal_raises_on_cross_vm_fs(self, tmp_path, monkeypatch):
monkeypatch.setattr(hermes_state_wal, "_path_on_cross_vm_fs", lambda p: True)
conn = sqlite3.connect(str(tmp_path / "state.db"))
with pytest.raises(WalUnsupportedError, match=r"cross-VM"):
apply_wal_with_fallback(conn, db_label="state.db", require_wal=True)
conn.close()
def test_on_disk_wal_db_is_never_downgraded(self, tmp_path, monkeypatch):
db = tmp_path / "already-wal.db"
seed = sqlite3.connect(str(db))
if str(seed.execute("PRAGMA journal_mode=WAL").fetchone()[0]).lower() != "wal":
seed.close()
pytest.skip("environment refuses WAL")
seed.execute("CREATE TABLE t (x)")
seed.commit()
seed.close()
monkeypatch.setattr(hermes_state_wal, "_path_on_cross_vm_fs", lambda p: True)
conn = sqlite3.connect(str(db))
assert apply_wal_with_fallback(conn, db_label=str(db)) == "wal"
conn.close()