refactor(hermes_state): restore WHY comments dropped by round-2 sub-branches

Comment/docstring-only (AST-identical): surrogate-scrub rationale, persisted
marker stripping invariant, generation counter upgrade semantics, CJK marker
empty-vs-populated rule, WAL 0-page ordering precondition, repair backup
live-connection case, telegram topic delete precondition, mixed-mode
corruption definition, and similar.
This commit is contained in:
Teknium
2026-09-02 16:48:30 -07:00
parent 88cb5c965c
commit eb8d628c97
7 changed files with 54 additions and 18 deletions

View File

@@ -1937,6 +1937,10 @@ class SessionDB(
if not cjk_present:
# Any old stale breadcrumb refers to a table that no longer exists.
cursor.execute("DELETE FROM state_meta WHERE key = ?", (FTS_CJK_STALE_KEY,))
# Empty DB: index complete by construction (triggers cover everything),
# no markers. Populated DB: the marker pair keeps the id-gated triggers
# correct while old rows await optimize-storage; the index is NOT
# served until that backfill completes.
if cursor.execute("SELECT COUNT(*) FROM messages WHERE role <> 'tool'").fetchone()[0] > 0:
hw = cursor.execute("SELECT COALESCE(MAX(id), 0) FROM messages").fetchone()[0]
for k, v in (
@@ -2195,7 +2199,8 @@ class SessionDB(
recorded_app = int(self._db_file_application_id or 0)
if recorded_app:
disk_app = _read_sqlite_application_id(self.db_path)
# Header 0 = WAL not yet checkpointed, not a replace.
# Header 0 = WAL not yet checkpointed, not a replace; any real
# replacement (a copied Hermes DB minted its own id) is nonzero.
if disk_app and disk_app != recorded_app:
return True
return False
@@ -2303,6 +2308,9 @@ class SessionDB(
conn = self._conn
setconfig = getattr(conn, "setconfig", None)
if flag is None or conn is None or setconfig is None:
# <3.12 has no setconfig: the residual close checkpoint is tolerable — it
# can only carry pre-quarantine committed frames; this handle accepts no
# further writes.
return
try:
setconfig(flag, True)

View File

@@ -138,7 +138,8 @@ class SessionMaintenanceMixin:
Disable the gate only for sources owned by state.db itself.
SELECT, live-lease validation and UPDATE run in one ``BEGIN IMMEDIATE``
transaction; active turn leases / compression locks spare the row.
transaction; active turn leases / compression locks spare the row, and
expired guards are removed so their former owner is fenced.
"""
from hermes_state import SessionCompressionInProgressError, SessionTurnLeaseLostError
srcs = tuple(s for s in sources if s)

View File

@@ -129,6 +129,12 @@ class SessionMessagesMixin:
scrubbed from text so persistence never fails. Paired with :meth:`_decode_content`.
"""
if isinstance(content, str):
# Lone UTF-16 surrogates arrive in tool results scraped from the web. The
# upstream sanitizer only cleans the api_messages copy and the recovery
# sanitizer only runs after the API call raises (it no longer does), so the
# canonical history keeps them and this write is where they land. Left raw,
# sqlite3 raises UnicodeEncodeError, the flush is abandoned, and the session
# silently stops persisting for the rest of its life.
return _sanitize_surrogates(content)
if content is None or isinstance(content, (bytes, int, float)):
return content
@@ -432,7 +438,10 @@ class SessionMessagesMixin:
display_metadata: Optional[Dict[str, Any]] = None,
) -> bool:
"""Stamp presentation metadata on this turn's freshly persisted row; the model
still receives ``role``/``content`` unchanged."""
still receives ``role``/``content`` unchanged. Gateway/CLI synthetic inputs call
this immediately after their serial turn has flushed (the target is resolved as
newest-active-row-by-content), preserving producer provenance without
classifying by content at render time."""
from hermes_state import _scrub_surrogates
if not session_id or not content or not display_kind:
return False
@@ -1154,6 +1163,9 @@ class SessionMessagesMixin:
if row["role"] in {"user", "assistant"} and isinstance(content, str):
content = sanitize_context(content).strip()
msg = {"role": row["role"], "content": content}
# Underscore-prefixed like ``_row_id``: every transport strips it before the
# wire, and compression's assembly copies deliberately strip it so rotated
# child handoffs still flush (see _fresh_compaction_message_copy).
msg[_DB_PERSISTED_MARKER_KEY] = True
if include_row_ids and row["id"] is not None:
msg["_row_id"] = row["id"]
@@ -1525,8 +1537,9 @@ class SessionMessagesMixin:
return cursor.fetchone()[0]
def has_platform_message_id(self, session_id: str, platform_message_id: str) -> bool:
"""True when a message with *platform_message_id* exists (partial index lookup;
the gateway's transient-failure dedupe guard)."""
"""True when a message with *platform_message_id* exists (uses the
idx_messages_platform_msg_id partial index; the gateway's transient-failure
dedupe guard)."""
return self._read_one(
"SELECT 1 FROM messages WHERE session_id = ? AND platform_message_id = ? LIMIT 1",
(session_id, platform_message_id),
@@ -1560,7 +1573,8 @@ class SessionMessagesMixin:
def is_explicit_fork_child(self, session_id: str) -> bool:
"""Public read-only view of :meth:`_is_explicit_fork_child_row`; a missing row
is not a fork."""
is not a fork. ``agent/prompt_cache_scope.py`` uses it to keep a declared
conversation key from crossing the fork boundary."""
session = self.get_session(session_id)
return bool(session and self._is_explicit_fork_child_row(session))
@@ -1573,7 +1587,11 @@ class SessionMessagesMixin:
``conversation_generations`` (advanced inside each boundary's txn), not an
aggregate over session rows: deletes/prunes would let an aggregate re-emit a
retired pair. Rows are never garbage-collected, by design (dropping one would
re-issue generation 1 — the ABA this counter prevents).
re-issue generation 1 — the ABA this counter prevents). Wall-clock-free, so a
backwards NTP correction cannot reorder it. DBs upgraded mid-conversation start
at no generation and take their first from the next boundary written; a
conversation that reset before the upgrade shares its predecessor's scope once
(costs a warm prompt-cache bucket, never crosses an identity).
"""
if not session_key or not source:
return None

View File

@@ -51,6 +51,7 @@ _FINGERPRINT_SAMPLE_BYTES = 65536
# malformed-SCHEMA DB still accepts writes, so without the mask any live write
# re-keys the ledger and the repair budget resets to 1 forever. The page-1
# sqlite_master b-tree — what repair identity depends on — sits after byte 100.
# (WAL mode routes commits to the -wal sidecar; masking is harmless there.)
_FINGERPRINT_VOLATILE_HEADER_RANGES = ((24, 28), (92, 96))
# Free-space headroom for the pre-repair forensic backup (a full raw copy of
# the damaged DB plus sidecars; a repair loop on a large state.db is a disk
@@ -612,7 +613,8 @@ def _backup_db_file(db_path: Path) -> "Tuple[Optional[Path], Optional[str]]":
because the forensic bundle is the recovery path when every strategy fails.
Refuses while a connection to this DB is live in the process: reading the
file would ``close()`` a descriptor and cancel that connection's POSIX
advisory locks (see ``hermes_cli.sqlite_safe_read``).
advisory locks (see ``hermes_cli.sqlite_safe_read``) — a real case: one
SessionDB can enter repair while the gateway holds others.
Dedupe: if the newest existing backup is byte-identical to the current
recovery image (``_backup_content_identity`` — NOT mtime, NOT
@@ -1135,7 +1137,8 @@ def _restore_journal_mode_after_repair(db_path: Path, before_mode: Optional[str]
open-time WAL-reset gate never sees a flip made inside repair). Routed
through :func:`apply_wal_with_fallback`, not a direct pragma, so it
inherits the vulnerable-SQLite WAL-reset gate (on a vulnerable runtime the
gate deliberately keeps DELETE), the macOS-NFS silent-refusal handling,
gate deliberately keeps DELETE and the resulting journal_mode-changed
WARNING is expected there), the macOS-NFS silent-refusal handling,
and the WAL companions. ``before_mode`` (None if unprobeable) is only for
the log comparison; the target comes from ``database.journal_mode``.
Best-effort: the repair already succeeded, so failures log at WARNING.
@@ -1169,7 +1172,8 @@ def _repair_state_db_schema_locked(db_path: Path, *, backup: bool, report: Dict[
Caller must hold the cross-process repair lock for *db_path*. Strategies
run on a SCRATCH COPY; the result is copied back through SQLite's
transactional backup API only once proven to open cleanly, so a failed
repair cannot modify or lose committed canonical data.
repair cannot modify or lose committed canonical data. (A WAL checkpoint of
already-committed frames on guard release is not a repair mutation.)
WHY not in place: Strategy 2 ends in ``VACUUM``, which rebuilds the file
from the schema SQLite can still parse. When the damage IS in the schema

View File

@@ -448,8 +448,9 @@ class SessionSearchMixin:
def _seed_fts_rebuild_markers(self, conn, *, force: bool = False) -> int:
"""Write ``fts_rebuild_high_water`` / ``fts_rebuild_progress`` for a
full backfill; returns the high-water id. Without ``force`` and with
high_water already set, only repairs a missing progress key. Caller
holds the write transaction."""
high_water already set, only repairs a missing progress key (see
``_reseed_missing_progress`` for why a missing key must not read as
"done"). Caller holds the write transaction."""
existing_hw = _meta_row(conn, "fts_rebuild_high_water")
if existing_hw is not None and not force:
self._reseed_missing_progress(conn)
@@ -535,6 +536,8 @@ class SessionSearchMixin:
]
for sh in shadows:
conn.execute(f"ALTER TABLE {sh} RENAME TO fts_v22_trash_{sh}")
# Claim the backfill BEFORE the empty v23 tables exist so a crash before
# schema ensure resumes instead of stamping an empty index.
hw = self._seed_fts_rebuild_markers(conn, force=True)
_delete_meta(conn, "fts_optimize_available")
return hw

View File

@@ -289,8 +289,8 @@ class SessionTelegramTopicsMixin:
) -> int:
"""Remove the binding row for one (chat, thread) pair.
Called when the Bot API confirms a topic was deleted externally;
otherwise ``gateway.run._recover_telegram_topic_thread_id`` keeps
Called when the Bot API confirms a topic was deleted externally
(``Thread not found`` after the same-thread retry failed); otherwise ``gateway.run._recover_telegram_topic_thread_id`` keeps
redirecting inbound messages to the dead topic. If this removes the
chat's *last* binding, ``telegram_dm_topic_mode`` is flipped to
``enabled = 0`` in the same transaction, or a user who disabled topics
@@ -319,7 +319,8 @@ class SessionTelegramTopicsMixin:
return
if not deleted["count"]:
return
# Last binding gone → disable topic mode, same transaction.
# Last binding gone → disable topic mode in the same transaction, so
# there is no read-after-prune race.
try:
remaining = conn.execute(
"""

View File

@@ -309,7 +309,8 @@ def apply_wal_with_fallback(
# Decide BEFORE the flip whether it would overwrite a mode somebody chose:
# the probe and page_count are only readable while the file is untouched.
# A 0-page DB has no prior choice, so brand-new databases stay quiet.
# A 0-page DB has no prior choice, and every caller reaches this before
# creating schema, so brand-new databases stay quiet.
_upgrading_existing_db = (
current_mode is not None and current_mode != "wal" and _database_has_content(conn)
)
@@ -345,8 +346,8 @@ def apply_wal_with_fallback(
raise # unrelated OperationalError — don't silently swallow
# ``disk i/o error`` is ambiguous: deterministic WAL-incompatibility on
# ZFS / APFS-CoW, or a one-shot transient EIO. Treating a transient EIO
# as a permanent downgrade signal produced mixed-mode corruption, so
# retry the pragma: transient EIO clears and we return "wal";
# as a permanent downgrade signal produced mixed-mode corruption (process
# A downgrades to DELETE while siblings set WAL), so retry the pragma: transient EIO clears and we return "wal";
# deterministic cases keep failing into the guarded DELETE fallback.
if "disk i/o error" in msg:
for _ in range(2):