refactor(state): dict-dispatch persistence-error classifier, unify WHERE/placeholder builders, compact docstrings across state modules

This commit is contained in:
Teknium
2026-09-02 20:18:08 -07:00
parent 3c9a51bbf1
commit ff3ebf509f
5 changed files with 372 additions and 549 deletions

View File

@@ -5,14 +5,12 @@ strings as well as live sqlite3 exceptions."""
import errno
import sqlite3
# ---------------------------------------------------------------------------
# Malformed-schema recovery: ``sqlite_master`` itself is inconsistent (typically
# a DUPLICATE ``CREATE VIRTUAL TABLE messages_fts`` row). SQLite parses the
# whole schema while preparing the FIRST statement, so EVERY statement raises —
# including ``PRAGMA journal_mode`` (it trips in apply_wal_with_fallback during
# __init__, before _init_schema) and plain ``DROP TABLE``; only
# ``PRAGMA writable_schema=ON`` + sqlite_master surgery still work. Canonical
# sessions/messages are intact; recovery rebuilds only the FTS layer.
# Malformed schema: ``sqlite_master`` itself is inconsistent (typically a DUPLICATE
# ``CREATE VIRTUAL TABLE messages_fts`` row). SQLite parses the whole schema while
# preparing the FIRST statement, so EVERY statement raises (even ``PRAGMA
# journal_mode`` during __init__); only ``PRAGMA writable_schema=ON`` +
# sqlite_master surgery still work. Canonical rows are intact; recovery rebuilds
# only the FTS layer.
_MALFORMED_SCHEMA_MARKERS = ("malformed database schema",)
_MALFORMED_DB_MARKERS = (*_MALFORMED_SCHEMA_MARKERS, "database disk image is malformed")
@@ -25,51 +23,42 @@ def is_malformed_db_error(exc: BaseException) -> bool:
)
# SQLITE_IOERR as a substring (wrapped strings still classify); shared by the
# read-only open retry and the write-path BEGIN retry.
# SQLITE_IOERR as a substring (wrapped strings still classify).
_DISK_IO_ERROR_MARKER = "disk i/o error"
# "Store BUSY, not gone" — HTTP callers map these to 503 instead of 500.
# Corruption deliberately absent: a malformed store must surface, not be
# retried into a timeout.
# "Store BUSY, not gone" — HTTP callers map these to 503 instead of 500. Corruption
# is deliberately absent: a malformed store must surface, not be retried into a timeout.
_TRANSIENT_SQLITE_MARKERS = (
_DISK_IO_ERROR_MARKER, "database is locked", "database table is locked", "busy",
)
def _is_no_more_rows(exc: sqlite3.Error) -> bool:
"""Transient engine error on contended WAL appends; the identical write succeeds
standalone, so it retries like locked/busy. Message-scoped because some builds
raise it as InterfaceError (outside DatabaseError)."""
"""Transient engine error on contended WAL appends (retries like locked/busy);
message-scoped because some builds raise it as InterfaceError."""
return "no more rows available" in str(exc).lower()
def is_transient_sqlite_error(exc: BaseException) -> bool:
""""Busy right now", not "damaged". One predicate so the read-only open
retry and the HTTP 503-vs-500 split cannot drift apart."""
""""Busy right now", not "damaged": one predicate so retry and the HTTP
503-vs-500 split cannot drift apart."""
return isinstance(exc, sqlite3.OperationalError) and any(
marker in str(exc).lower() for marker in _TRANSIENT_SQLITE_MARKERS
)
def is_malformed_schema_error(exc: BaseException) -> bool:
"""Only SQLite's explicit malformed-schema text. A generic "disk image is
malformed" (SQLITE_CORRUPT) may be any B-tree/freelist page and does not
prove canonical rows intact, so runtime repair must fail closed on it."""
"""Only SQLite's explicit malformed-schema text: a generic "disk image is
malformed" may be any B-tree page, so runtime repair must fail closed on it."""
return isinstance(exc, sqlite3.DatabaseError) and any(
marker in str(exc).lower() for marker in _MALFORMED_SCHEMA_MARKERS
)
# "Filesystem cannot accept another write" substrings (OSError, sqlite3, and
# wrapped RPC strings all match the same helper).
# "Filesystem cannot accept another write" substrings (OSError, sqlite3, wrapped RPC strings).
_DISK_FULL_MARKERS = (
"no space left on device",
"not enough space",
"database or disk is full", # SQLITE_FULL
"disk full",
"full disk",
"enospc",
"no space left on device", "not enough space", "database or disk is full", # SQLITE_FULL
"disk full", "full disk", "enospc",
)
@@ -83,67 +72,21 @@ def is_disk_full_error(exc: BaseException | str | None) -> bool:
return any(marker in lowered for marker in _DISK_FULL_MARKERS)
# Every classify_persistence_error bucket; consumers enumerate this tuple so a
# new bucket can never silently desynchronize them.
# Every classify_persistence_error bucket; consumers enumerate this tuple.
PERSISTENCE_ERROR_CAUSES = (
"locked", "compression", "compression_closed", "turn_lease", "corrupt", "replaced", "disk",
"unknown",
)
# "Database FILE structurally damaged" substrings. NOTE: "database disk image is
# "Database FILE structurally damaged" substrings. "database disk image is
# malformed" contains "disk", so this check MUST run before the disk bucket in
# classify_persistence_error or B-tree corruption reads as "free some disk space".
_DB_CORRUPTION_MARKERS = (
"malformed", # "database disk image is malformed" (SQLITE_CORRUPT)
"file is not a database", # SQLITE_NOTADB (also connection-level poisoning)
"not a database",
"database corruption",
"malformed", "file is not a database", "not a database", "database corruption",
)
def classify_persistence_error(exc_or_str) -> str:
"""Coarse cause bucket (PERSISTENCE_ERROR_CAUSES) so the user's guidance
matches: "locked" = busy, retry; "disk" = full/read-only/permissions;
"compression" = a live lease refused the write; "compression_closed" = adopt
the rotated session id; "turn_lease" = fencing, not storage; "corrupt" =
file damage (repair path, not disk space); "replaced" = stop writing."""
if exc_or_str is None:
return "unknown"
# Lease refusals contain neither "locked" nor "busy": match by type, then by
# phrase for strings that survived RPC wrapping.
if isinstance(exc_or_str, SessionTurnLeaseLostError):
return "turn_lease"
if isinstance(exc_or_str, CompressionSessionClosedError):
return "compression_closed"
if isinstance(exc_or_str, CompressionSessionBusyError):
return "compression"
if isinstance(exc_or_str, StateDbReplacedError): # incl. DeletedWalGenerationError
return "replaced"
if isinstance(exc_or_str, StateDbCorruptError):
return "corrupt"
text = str(exc_or_str).lower()
if "turn lease" in text:
return "turn_lease"
if "closed by compression" in text:
return "compression_closed"
if "being compressed" in text or "compression lease" in text:
return "compression"
if "was replaced underneath" in text:
return "replaced"
if "deleted state.db-wal" in text or "deleted state.db-shm" in text:
return "replaced"
# Corruption BEFORE the lock/disk buckets: "disk image is malformed"
# contains "disk" and some wrapped strings mention "locked" recovery.
if any(marker in text for marker in _DB_CORRUPTION_MARKERS):
return "corrupt"
if "locked" in text or "busy" in text:
return "locked"
if is_disk_full_error(exc_or_str) or "disk" in text or "readonly" in text or "read-only" in text:
return "disk"
return "unknown"
class CompressionSessionClosedError(RuntimeError):
"""A durable write targeted a parent already closed by compression."""
@@ -223,3 +166,44 @@ _STATE_DB_CORRUPT_MSG = (
"--inspect-only` or restore a snapshot. Unwritten transcripts are diverted to "
"sessions/<id>.jsonl (and the gateway pending_messages spool)."
)
_PERSISTENCE_CAUSE_BY_TYPE = (
(SessionTurnLeaseLostError, "turn_lease"),
(CompressionSessionClosedError, "compression_closed"),
(CompressionSessionBusyError, "compression"),
(StateDbReplacedError, "replaced"),
(StateDbCorruptError, "corrupt"),
)
_PERSISTENCE_CAUSE_BY_PHRASE = (
(("turn lease",), "turn_lease"),
(("closed by compression",), "compression_closed"),
(("being compressed", "compression lease"), "compression"),
(("was replaced underneath", "deleted state.db-wal", "deleted state.db-shm"), "replaced"),
(_DB_CORRUPTION_MARKERS, "corrupt"),
(("locked", "busy"), "locked"),
)
def classify_persistence_error(exc_or_str) -> str:
"""Coarse cause bucket (PERSISTENCE_ERROR_CAUSES) so the user's guidance
matches: "locked" = busy, retry; "disk" = full/read-only/permissions;
"compression" = a live lease refused the write; "compression_closed" = adopt
the rotated session id; "turn_lease" = fencing, not storage; "corrupt" =
file damage (repair path, not disk space); "replaced" = stop writing."""
if exc_or_str is None:
return "unknown"
# Lease refusals contain neither "locked" nor "busy": match by type first,
# then by phrase for strings that survived RPC wrapping. Order matters:
# StateDbReplacedError covers DeletedWalGenerationError; corruption comes
# BEFORE the lock/disk buckets ("disk image is malformed" contains "disk").
for exc_type, cause in _PERSISTENCE_CAUSE_BY_TYPE:
if isinstance(exc_or_str, exc_type):
return cause
text = str(exc_or_str).lower()
for markers, cause in _PERSISTENCE_CAUSE_BY_PHRASE:
if any(marker in text for marker in markers):
return cause
if is_disk_full_error(exc_or_str) or any(m in text for m in ("disk", "readonly", "read-only")):
return "disk"
return "unknown"

View File

@@ -14,23 +14,18 @@ from hermes_state_common import FTS_CJK_STALE_KEY, FTS_STALE_KEY, _FTS_CJK_TRIGG
logger = logging.getLogger("hermes_state")
# ── CJK-bigram FTS index (replaces the trigram index when available) ────
# Trigram needs >=3 chars per term, so 1-2 char CJK terms fell through to a
# LIKE table scan (3-6s CPU per query on multi-GB installs). ``cjk_unicode61``
# (native/fts5_cjk/, loadable) re-emits CJK runs as overlapping bigrams; FTS5
# phrase semantics then give exact substring matching down to 2 chars.
# Trigram needs >=3 chars per term, so 1-2 char CJK terms fell through to a LIKE
# table scan; ``cjk_unicode61`` (native/fts5_cjk/, loadable) re-emits CJK runs as
# overlapping bigrams. Same v23 discipline as the trigram table: external-content
# over a tool-row-excluding view, triggers gated on a DEDICATED marker pair
# (fts_cjk_rebuild_high_water / _progress). The table exists ONLY when the
# tokenizer loads; a process that cannot load it drops the cjk triggers (writes
# keep working; the index goes stale until the next optimize-storage).
#
# Same v23 discipline as the trigram table: external-content over a
# tool-row-excluding view, triggers gated on a DEDICATED marker pair
# (fts_cjk_rebuild_high_water / _progress) so a cjk-only backfill never gates
# the complete messages_fts triggers. The table exists ONLY when the tokenizer
# loads (~/.hermes/lib/libfts5_cjk.so); a process that cannot load it drops the
# cjk triggers (writes keep working; the index goes stale until the next
# optimize-storage on a capable host).
#
# Split DDL: the table/view is safe to ensure any time; triggers are created
# ONLY while the index is complete-or-marker-gated. A stale index must keep its
# Split DDL: the table/view is safe to ensure any time; triggers are created ONLY
# while the index is complete-or-marker-gated. A stale index must keep its
# triggers DROPPED — an external-content 'delete' for a rowid the index never
# held is the canonical FTS5 corruption hazard the marker gating prevents.
# held is the canonical FTS5 corruption hazard.
FTS_CJK_TABLE_SQL = """
CREATE VIEW IF NOT EXISTS messages_fts_cjk_src AS
SELECT id, role, content, tool_name, tool_calls
@@ -90,12 +85,11 @@ BEGIN
END;
"""
def fts5_cjk_so_path() -> Path:
"""Location of the cjk_unicode61 loadable extension."""
env = os.getenv("HERMES_FTS5_CJK_SO")
if env:
return Path(env).expanduser()
return get_hermes_home() / "lib" / "libfts5_cjk.so"
return Path(env).expanduser() if env else get_hermes_home() / "lib" / "libfts5_cjk.so"
def _cjk_fts_config_enabled() -> bool:
@@ -104,13 +98,10 @@ def _cjk_fts_config_enabled() -> bool:
def load_fts5_cjk_extension(conn: sqlite3.Connection) -> bool:
"""Best-effort load of the cjk_unicode61 tokenizer. False (never raises)
when the .so is absent, ``sessions.cjk_fts`` is off, or extension loading
is compiled out — callers then behave as before the cjk index existed."""
if not _cjk_fts_config_enabled():
return False
"""Best-effort load of the cjk_unicode61 tokenizer; False (never raises) when
the .so is absent, ``sessions.cjk_fts`` is off, or loading is compiled out."""
path = fts5_cjk_so_path()
if not path.exists():
if not _cjk_fts_config_enabled() or not path.exists():
return False
try:
conn.enable_load_extension(True)
@@ -129,10 +120,7 @@ class SessionFtsSetupMixin:
@staticmethod
def _is_fts5_unavailable_error(exc: sqlite3.OperationalError) -> bool:
# Builds with FTS5 but without the optional trigram tokenizer raise
# "no such tokenizer: trigram" instead of "no such module"; the loadable
# cjk_unicode61 tokenizer shows the same capability-error shape. Scoped
# to those two tokenizers so unrelated tokenizer errors aren't masked.
"""No FTS5 module, or an optional tokenizer missing (same capability-error shape)."""
err = str(exc).lower()
return ("no such module" in err and "fts5" in err) or SessionFtsSetupMixin._is_trigram_unavailable_error(exc)
@@ -141,14 +129,12 @@ class SessionFtsSetupMixin:
"""Only an optional tokenizer is missing (trigram needs SQLite >= 3.34;
cjk_unicode61 is loadable): "this one index can't be served", never "disable FTS"."""
err = str(exc).lower()
return ("no such tokenizer: trigram" in err or "no such tokenizer: cjk_unicode61" in err)
return "no such tokenizer: trigram" in err or "no such tokenizer: cjk_unicode61" in err
@staticmethod
def _db_has_legacy_inline_fts(cursor: sqlite3.Cursor) -> bool:
"""messages_fts exists in ANY pre-v23 shape. v23 is external-content over
content/tool_name/tool_calls; every legacy shape (inline single-column
v11..v22, or the v10-era external single-column) lacks tool_name, so
"stored CREATE lacks tool_name" catches both. False when absent (fresh DB)."""
"""messages_fts exists in ANY pre-v23 shape: every legacy shape lacks
tool_name, so "stored CREATE lacks tool_name" catches them all. False when absent."""
row = cursor.execute(
"SELECT sql FROM sqlite_master WHERE type = 'table' AND name = 'messages_fts'"
).fetchone()
@@ -156,7 +142,7 @@ class SessionFtsSetupMixin:
def _warn_trigram_unavailable(self, exc: sqlite3.OperationalError) -> None:
"""Log once that the trigram tokenizer is missing; base FTS5 stays enabled."""
if getattr(self, "_trigram_unavailable_warned", False):
if getattr(self, "_trigram_unavailable_warned", False): # attr is lazily created here
return
self._trigram_unavailable_warned = True
logger.info(
@@ -182,12 +168,11 @@ class SessionFtsSetupMixin:
)
def _ensure_fts_cjk_schema(self, cursor) -> None:
"""Create / repair / self-heal the CJK-bigram index (see the module
comment). Sets ``_fts_cjk_available``; never raises. Loaded + absent →
create (a populated DB gets the backfill markers and is NOT served until
optimize-storage backfills); loaded + present → ensure triggers, honour
the stale breadcrumb; NOT loaded + live triggers → drop them so INSERTs
don't fail at trigger time and leave the breadcrumb."""
"""Create / repair / self-heal the CJK-bigram index (see the module comment).
Sets ``_fts_cjk_available``; never raises. Loaded + absent → create (a
populated DB gets backfill markers and is NOT served until optimize-storage
backfills); loaded + present → ensure triggers, honour the stale breadcrumb;
NOT loaded + live triggers → drop them (INSERTs must not fail at trigger time)."""
try:
cjk_present = bool(cursor.execute(
"SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = 'messages_fts_cjk'"
@@ -200,8 +185,7 @@ class SessionFtsSetupMixin:
_FTS_CJK_TRIGGERS,
).fetchall()]
if live:
# Breadcrumb FIRST (a crash between the two statements
# is merely conservative), then drop.
# Breadcrumb FIRST (a crash between the two is merely conservative).
logger.warning(
"messages_fts_cjk triggers present but the "
"cjk_unicode61 tokenizer is unavailable (%s) — "
@@ -230,12 +214,10 @@ class SessionFtsSetupMixin:
try:
cursor.executescript(FTS_CJK_TABLE_SQL)
if not cjk_present:
# Any old stale breadcrumb refers to a table that no longer exists.
# An old stale breadcrumb refers to a table that no longer exists.
cursor.execute("DELETE FROM state_meta WHERE key = ?", (FTS_CJK_STALE_KEY,))
# Empty DB: index complete by construction (triggers cover everything),
# no markers. Populated DB: the marker pair keeps the id-gated triggers
# correct while old rows await optimize-storage; the index is NOT
# served until that backfill completes.
# Empty DB: complete by construction, no markers. Populated DB: the
# marker pair keeps the id-gated triggers correct until backfill.
if cursor.execute("SELECT COUNT(*) FROM messages WHERE role <> 'tool'").fetchone()[0] > 0:
hw = cursor.execute("SELECT COALESCE(MAX(id), 0) FROM messages").fetchone()[0]
for k, v in (
@@ -247,9 +229,7 @@ class SessionFtsSetupMixin:
(k, v),
)
if cursor.execute("SELECT 1 FROM state_meta WHERE key = ?", (FTS_CJK_STALE_KEY,)).fetchone():
# Gap of unknown extent: do NOT reinstall triggers (an
# external-content 'delete' for an unindexed rowid corrupts the
# index); the next optimize-storage rebuilds from scratch.
# Gap of unknown extent: do NOT reinstall triggers (see module comment).
self._fts_cjk_available = False
return
cursor.executescript(FTS_CJK_TRIGGER_SQL)
@@ -292,9 +272,8 @@ class SessionFtsSetupMixin:
@staticmethod
def _is_fts_write_corruption_error(exc: sqlite3.DatabaseError) -> bool:
"""Corruption SQLite identifies as FTS-scoped: SQLITE_CORRUPT_VTAB, or
(older builds) an ``fts5:`` message. A bare malformed-image error is
structural and must not trigger live FTS maintenance."""
"""Corruption SQLite identifies as FTS-scoped (SQLITE_CORRUPT_VTAB, or an
``fts5:`` message on older builds); a bare malformed image is structural."""
error_code = getattr(exc, "sqlite_errorcode", None)
if error_code is not None:
return error_code == getattr(sqlite3, "SQLITE_CORRUPT_VTAB", 267)
@@ -302,10 +281,9 @@ class SessionFtsSetupMixin:
return msg.startswith("fts5:") and "corrupt structure" in msg
def _enter_fts_fail_open(self, exc: sqlite3.DatabaseError) -> bool:
"""Detach corrupt FTS indexes so canonical writes can continue. Stale
breadcrumb + trigger drop commit atomically: once triggers are absent
the index has a gap of unknown extent, so no process may reinstall them
without rebuilding every row."""
"""Detach corrupt FTS indexes so canonical writes can continue. Breadcrumb +
trigger drop commit atomically: once triggers are absent the index has a
gap of unknown extent, so nobody may reinstall them without a full rebuild."""
if not self._fts_enabled or not self._is_fts_write_corruption_error(exc):
return False
self._raise_if_db_corrupt()
@@ -355,21 +333,18 @@ class SessionFtsSetupMixin:
return True
# ── Chunked FTS rebuild engine (v23 opt-in optimize) ──
# One blocking rebuild held the write lock ~16 minutes on a 25 GB DB, so the
# backfill runs in small chunks, each its own short transaction (resumable
# from fts_rebuild_progress; concurrent runners claim chunks by CAS).
# THROTTLING: a greedy loop owned the lock ~85% of the time and starved
# other processes' writers; 500-row chunks plus a pause of max(MIN_PAUSE,
# chunk cost x DUTY_FACTOR) cap our duty cycle unconditionally (works
# cross-process, unlike any same-process activity stamp).
# One blocking rebuild held the write lock ~16 min on a 25 GB DB, so the
# backfill runs in small chunks (each its own short transaction, resumable from
# fts_rebuild_progress, claimed by CAS). A greedy loop starved other writers:
# a pause of max(MIN_PAUSE, chunk cost x DUTY_FACTOR) caps the duty cycle
# cross-process, unlike any same-process activity stamp.
_FTS_REBUILD_CHUNK_ROWS = 500
_FTS_REBUILD_DUTY_FACTOR = 4.0 # sleep >= 4x chunk cost (≤20% duty)
_FTS_REBUILD_MIN_PAUSE = 0.2 # seconds — floor between chunks
# Demoted v22 FTS shadow tables awaiting teardown: DROP of a multi-GB vtable
# blocks for minutes, so the v23 migration demotes the vtable definitions
# out of sqlite_master and renames the orphaned shadow tables (now plain
# tables) to fts_v22_trash_*; the worker empties them in chunks, then drops.
# blocks for minutes, so the v23 migration renames the orphaned shadow tables
# to fts_v22_trash_*; the worker empties them in chunks, then drops.
_FTS_TRASH_PREFIX = "fts_v22_trash_"
def _has_fts_trash(self, conn) -> bool:
@@ -380,6 +355,6 @@ class SessionFtsSetupMixin:
(self._FTS_TRASH_PREFIX.replace("_", "\\_") + "%",),
).fetchone())
# FTS5 tables merged on optimize; trigram may be disabled and cjk exists only
# with the loadable tokenizer, so each is probed before touching (optimize_fts).
# FTS5 tables merged on optimize; each is probed before touching (trigram may
# be disabled, cjk exists only with the loadable tokenizer).
_FTS_TABLES = ("messages_fts", "messages_fts_trigram", "messages_fts_cjk")

View File

@@ -15,9 +15,7 @@ except ImportError: # pragma: no cover - stripped/scaffold installs only
psutil = None # type: ignore[assignment]
# Field evidence: pytest fixture rows landed in the production state.db and a
# pytest-spawned child flipped the journal mode under the live WAL writer,
# destroying committed transcripts; any HERMES_HOME escape (fixture ordering, a
# child spawned without it, a shell exporting the real home) fell through silently.
# pytest-spawned child flipped the journal mode under the live WAL writer.
#: Env twin of ``_STATE_DB_GUARD_BYPASS`` for child processes (a module global
#: cannot cross a process boundary, and ancestry arms the guard there).
@@ -26,29 +24,23 @@ _STATE_DB_GUARD_BYPASS_ENV = "HERMES_STATE_DB_GUARD_BYPASS"
def _real_platform_state_root() -> Optional[Path]:
"""The REAL platform-default Hermes root. Avoids ``Path.home()`` /
``hermes_constants``: tests monkeypatch Path.home to a tempdir while this
module is imported lazily, which would misidentify the hermetic home as
production or miss the real one. ``expanduser`` reads HOME/passwd, which the
conftest never rewrites."""
``hermes_constants`` (tests monkeypatch Path.home to a tempdir); ``expanduser``
reads HOME/passwd, which the conftest never rewrites."""
try:
home = Path(os.path.expanduser("~"))
if sys.platform == "win32":
base = os.environ.get("LOCALAPPDATA", "").strip()
root = (
Path(base) / "hermes"
if base
else Path(os.path.expanduser("~")) / "AppData" / "Local" / "hermes"
)
root = Path(base) / "hermes" if base else home / "AppData" / "Local" / "hermes"
else:
root = Path(os.path.expanduser("~")) / ".hermes"
root = home / ".hermes"
return root.resolve()
except Exception:
return None
#: Exported by the hermetic conftest alongside the HERMES_HOME redirect (value:
#: the isolation root). Unlike PYTEST_* (scrubbed by tests that rebuild a child
#: env) it is OURS and inherits by default, so a child carrying it that resolves
#: a production DB is by definition an isolation escape.
#: Exported by the hermetic conftest alongside the HERMES_HOME redirect. Unlike
#: PYTEST_* it is OURS and inherits by default, so a child carrying it that
#: resolves a production DB is by definition an isolation escape.
_TEST_ISOLATION_MARKER_ENV = "HERMES_TEST_ISOLATION"
@@ -70,18 +62,15 @@ _PYTEST_ANCESTOR: Optional[bool] = None
def _process_looks_like_pytest(proc: Any) -> bool:
"""True when *proc*'s command line is a pytest invocation (``pytest ...`` or
``python -m pytest``). Unreadable cmdline => not pytest: guessing the other
way would refuse production opens for unrelated reasons."""
"""True when *proc*'s command line is a pytest invocation. Unreadable cmdline
=> not pytest: guessing the other way would refuse production opens."""
try:
cmdline = proc.cmdline() or []
except Exception:
return False
for arg in cmdline:
try:
# Split on both separators on every host: os.path.basename is
# POSIX-only under Linux and would leave a Windows-style path
# intact, making the matcher's answer depend on the platform.
# Split on both separators on every host so the answer is platform-independent.
name = str(arg).strip('"').strip("'").replace("\\", "/").rsplit("/", 1)[-1].lower()
except Exception:
continue
@@ -91,10 +80,9 @@ def _process_looks_like_pytest(proc: Any) -> bool:
def _has_pytest_ancestor() -> bool:
"""True when an ancestor process is a pytest run. A child spawned with a
rebuilt env loses PYTEST_* and the HERMES_HOME redirect together — aiming at
production AND disarming the guard in one step; ancestry survives that.
Fails open without psutil / on walk errors (never block real user runs)."""
"""True when an ancestor process is a pytest run: a child spawned with a
rebuilt env loses PYTEST_* and the HERMES_HOME redirect together, ancestry
survives that. Fails open without psutil / on walk errors."""
global _PYTEST_ANCESTOR
if _PYTEST_ANCESTOR is not None:
return _PYTEST_ANCESTOR
@@ -109,15 +97,13 @@ def _has_pytest_ancestor() -> bool:
def _in_test_context() -> bool:
"""Test run by environment or ancestry. Env first (two dict lookups); the
memoised ancestry walk runs at most once per real ``hermes`` invocation."""
"""Test run by environment or ancestry (memoised; env checked first)."""
return _running_under_pytest() or _has_pytest_ancestor()
def _is_production_state_db(resolved: Path, root: Path) -> bool:
"""*resolved* is ``<root>/state.db`` or ``<root>/profiles/<name>/state.db``.
Deeper scratch paths (repo worktrees under ~/.hermes/hermes-agent/...) are
deliberately NOT matched so hermetic tests cannot false-positive."""
"""*resolved* is ``<root>/state.db`` or ``<root>/profiles/<name>/state.db``;
deeper scratch paths (repo worktrees) are deliberately NOT matched."""
if resolved.parent == root:
return True
try:
@@ -128,16 +114,15 @@ def _is_production_state_db(resolved: Path, root: Path) -> bool:
# Last SessionDB() init error, per-process; surfaced by /resume-style slash
# commands so users know WHY. Only SessionDB.__init__ writes it (kanban_db
# failures are reported via their own callers, by design).
# commands so users know WHY. Only SessionDB.__init__ writes it.
_last_init_error: Optional[str] = None
_last_init_error_lock = threading.Lock()
def _set_last_init_error(msg: Optional[str]) -> None:
"""Record (or clear with None) the most recent state.db init failure.
__init__ only SETs on failure and never clears on success: a concurrent
successful open would erase the cause another thread's /resume is about to format."""
"""Record (or clear with None) the most recent init failure. __init__ never
clears on success: a concurrent open would erase the cause another thread's
/resume is about to format."""
global _last_init_error
with _last_init_error_lock:
_last_init_error = msg

View File

@@ -18,33 +18,28 @@ if TYPE_CHECKING: # pragma: no cover
# caplog tests pin the "hermes_state" logger name.
logger = logging.getLogger("hermes_state")
# Ceiling on read-only connections ALIVE at once against one database FILE
# (idle pooled + checked out, summed over every SessionDB on that file). One
# constant for both the pool maxsize and the permit count: a LifoQueue only caps
# how many are *returned*; with open-on-miss, N readers hitting an empty pool
# all open and peak at N, and EMFILE is a peak-instant condition. So a
# connection holds a permit for its whole lifetime (_get_read_conn ->
# _close_read_conn); once permits are gone reads degrade to the locked writer
# connection — slower, but not a process-wide wedge the supervisor can't see.
# Ceiling on read-only connections ALIVE at once against one database FILE (idle
# pooled + checked out, over every SessionDB on that file). One constant for both
# the pool maxsize and the permit count: a LifoQueue only caps how many are
# *returned*, and EMFILE is a peak-instant condition, so a connection holds a
# permit for its whole lifetime; once permits are gone reads degrade to the
# locked writer connection — slower, but not a wedge the supervisor can't see.
_READ_POOL_MAX = 8
# Ceiling on read-only connections ALIVE in this PROCESS across every state.db
# (a multiplexed gateway opens one per profile, so a per-file cap still scales
# with profile count). Three profiles' worth; past it readers degrade to the
# writer connection for the same reason as _READ_POOL_MAX.
# Ceiling ALIVE in this PROCESS across every state.db (a multiplexed gateway
# opens one per profile); three profiles' worth, then readers degrade likewise.
_READ_POOL_PROCESS_MAX = 24
# Warn past this many SessionDB handles on one file in one process. Diagnostic
# only: writer connections cannot be rationed the way read connections can.
# Warn past this many SessionDB handles on one file in one process (diagnostic:
# writer connections cannot be rationed the way read connections can).
_HANDLES_PER_PATH_WARN = 4
# Descriptors kept in reserve for everything that is NOT this module (httpx
# sockets, terminal pipes, log files): SQLite's share is only part of the fd
# table, and the EMFILE it pushes over surfaces elsewhere (terminal_tool).
# sockets, terminal pipes, log files): the EMFILE SQLite pushes over surfaces elsewhere.
_FD_HEADROOM_RESERVE = 64
# The fd count is a directory listing; cache it briefly so a read burst isn't a
# syscall per query. Staleness lets through at most the ceiling's worth of opens.
# syscall per query (staleness lets through at most the ceiling's worth of opens).
_FD_USAGE_CACHE_SECONDS = 0.25
_process_read_permits = threading.BoundedSemaphore(_READ_POOL_PROCESS_MAX)
@@ -70,8 +65,7 @@ def _proc_fd_targets(pid: int) -> Iterator[str]:
def _open_fd_count() -> Optional[int]:
"""Open descriptors in THIS process; None when unmeasurable (Windows: no fd
dir and no RLIMIT_NOFILE, correctly inert — its limit is thousands); -1 when
the probe itself hit EMFILE/ENFILE (that IS the answer: no headroom)."""
dir, correctly inert); -1 when the probe itself hit EMFILE/ENFILE (no headroom)."""
for fd_dir in ("/proc/self/fd", "/dev/fd"):
try:
return len(os.listdir(fd_dir))
@@ -97,10 +91,9 @@ def _fd_soft_limit() -> Optional[int]:
def _fd_headroom_ok() -> bool:
"""Can the process spare a descriptor for a new read connection?
Fails OPEN when unmeasurable (refusing every read there would be a
self-inflicted convoy); fails CLOSED only on evidence (measured shortfall,
or a probe that couldn't get a descriptor itself)."""
"""Can the process spare a descriptor for a new read connection? Fails OPEN
when unmeasurable (refusing every read would be a self-inflicted convoy);
fails CLOSED only on evidence (measured shortfall or a starved probe)."""
soft = _fd_soft_limit()
if soft is None:
return True
@@ -127,11 +120,10 @@ def _reclaim_idle_read_conn_anywhere() -> bool:
class _PathReadBudget:
"""Read-connection permits for ONE database file, shared process-wide:
per-instance semaphores let N SessionDBs on one file peak at N x (1 + MAX)
and walk into EMFILE. An idle pooled connection keeps its permit, so a
permit miss first reclaims an IDLE connection from a peer on the same path
(idle descriptors are transferable, in-use ones are not)."""
"""Read-connection permits for ONE database file, shared process-wide
(per-instance semaphores let N SessionDBs peak at N x (1 + MAX)). An idle
pooled connection keeps its permit, so a permit miss first reclaims an IDLE
connection from a peer on the same path."""
def __init__(self) -> None:
self.permits = threading.BoundedSemaphore(_READ_POOL_MAX)
@@ -148,22 +140,19 @@ class _PathReadBudget:
if warn:
self._duplicate_handles_warned = True
if warn:
# Writer connections cannot be capped (a SessionDB without one cannot
# write); the only bound is not opening redundant handles. Make the
# next duplicate visible before it becomes an incident.
# Writer connections cannot be capped; the only bound is not opening
# redundant handles, so make the duplicate visible before it's an incident.
logger.warning(
"%d live SessionDB handles on %s in this process; each holds "
"its own writer connection (read connections are capped at %d "
"for the file). A long-lived process should share one handle per path.",
handles,
db.db_path,
_READ_POOL_MAX,
handles, db.db_path, _READ_POOL_MAX,
)
def acquire(self, requester: "SessionDB") -> bool:
"""Take a permit for a new read connection, or refuse (caller then reads
via the locked writer connection — slower, never an error). Gates,
broadest first: fd headroom, process-wide ceiling, this file's ceiling."""
"""Take a permit for a new read connection, or refuse (caller degrades to the
locked writer connection). Gates, broadest first: fd headroom, process
ceiling, this file's ceiling."""
if not _fd_headroom_ok():
global _read_open_denied_fd_headroom
with _read_budgets_lock:
@@ -182,8 +171,7 @@ class _PathReadBudget:
_process_read_permits.release()
def _acquire_process_permit(self) -> bool:
# Another thread may take a freed permit first; that is a legitimate
# loss, and the caller degrades to the writer lock rather than looping.
# Another thread may take a freed permit first: legitimate loss, no looping.
return _process_read_permits.acquire(blocking=False) or (
_reclaim_idle_read_conn_anywhere() and _process_read_permits.acquire(blocking=False)
)
@@ -201,8 +189,8 @@ class _PathReadBudget:
return any(member._evict_one_idle_read_conn() for member in members)
# canonical db path -> permits for that file. Weak values: the budget lives as
# long as some SessionDB on the path holds it, so tmp_path churn can't grow this.
# canonical db path -> permits for that file. Weak values: the budget lives only
# while some SessionDB on the path holds it, so tmp_path churn can't grow this.
_read_budgets: "weakref.WeakValueDictionary[str, _PathReadBudget]" = (weakref.WeakValueDictionary())
_read_budgets_lock = threading.Lock()

File diff suppressed because it is too large Load Diff