Files
hermes-agent/tests/hermes_state/test_session_db_read_path_split.py
teknium1 d10bb2ab6f test: make tests/ mirror the source tree; drop issue numbers from filenames
`scripts/run_tests.sh tests/<dir>/` is how a change gets its regression
coverage run, so a test filed under the wrong directory is a test nobody
runs when that code changes. Two kinds of drift had accumulated.

Parallel directories for one source package, folded into the mirror:
  tests/acp        -> tests/acp_adapter   (its __init__/conftest move with it)
  tests/cli        -> tests/hermes_cli    (prompt_toolkit fixture merged into
                                           hermes_cli/conftest.py)
  tests/run_agent  -> tests/agent         (backoff fixture becomes
                                           agent/conftest.py)
  tests/relay      -> tests/gateway/relay
  tests/state      -> tests/hermes_state

246 loose files at tests/ root, routed by the package they import/patch:
hermes_cli, hermes_state, agent, gateway, tools, plugins, tui_gateway, cron.
Installer and desktop-update script tests go to tests/scripts/{install,
desktop_update}/. 43 tests of root-level modules (batch_runner, utils,
hermes_constants, packaging) stay at the root.

Filenames drop their issue numbers (95 files: test_89315_x.py -> test_x.py);
the number stays in the module docstring where it has context.

Collisions: test_cli_skin_integration.py existed in both tests/ and tests/cli
with different subsets — merged into one (10 tests, all kept);
run_agent/test_pre_compress_memory_context.py -> agent/..._handoff.py;
tests/test_account_usage.py -> agent/test_account_usage_fetch.py;
tests/test_web_server.py -> hermes_cli/test_web_server_ws_ping.py.
Deleted: test_minisweagent_path.py (empty since PR #2804),
test_model_picker_scroll.py (tested a private copy of the logic, imported
nothing), test_process_loop_event_loop_warning.py (asserted asyncio behaviour,
imported nothing from Hermes).

Repo-root path arithmetic (Path(__file__).parents[N], dirname chains) is
bumped for the 202 files that changed depth and verified by evaluating every
such expression against the new location. classify_changes' desktop-updater
lane prefix, tests-os.yml's ignore glob and every in-tree path comment follow
the moves. tests/test_tests_tree_layout.py keeps the tree from drifting back.
2026-09-13 09:18:02 -07:00

210 lines
7.4 KiB
Python

"""Tests for the SessionDB read-path split (pooled read-only connections).
The gateway shares ONE SessionDB across every agent, so recall/browse reads
used to queue behind writer flushes on self._lock — a measured production
convoy (a 0.2s FTS query stretched to 112s while 6-8 concurrent turns
flushed tool results). These tests pin the new contract: reads run on a
read-only connection borrowed from a bounded pool under WAL, never touch
self._lock, and fall back to the legacy locked path when WAL or the read
connection is missing.
"""
import threading
import pytest
from hermes_state import SessionDB
@pytest.fixture()
def db(tmp_path):
d = SessionDB(db_path=tmp_path / "state.db")
d.create_session(session_id="s1", source="cli", model="m")
d.append_message("s1", role="user", content="hello graphiti world")
d.append_message("s1", role="assistant", content="the neo4j daemon is healthy")
yield d
d.close()
@pytest.mark.requires_wal
def test_read_conn_is_per_thread(db):
conns = {}
def grab(key):
conns[key] = db._get_read_conn()
t1 = threading.Thread(target=grab, args=(1,))
t2 = threading.Thread(target=grab, args=(2,))
t1.start(); t2.start(); t1.join(); t2.join()
assert conns[1] is not None and conns[2] is not None
assert conns[1] is not conns[2]
@pytest.mark.requires_wal
def test_read_conn_reused_via_pool(db):
"""Reuse is now the pool's job, not a per-thread memo.
The old contract (``_get_read_conn()`` returns the same object twice on one
thread) was the leak: that memo pinned one unclosable connection per
(SessionDB x thread) forever. ``_get_read_conn`` now always opens a fresh
connection and reuse happens via checkout/return, so assert on that.
"""
with db._read_ctx() as first:
assert first is not None
with db._read_ctx() as second:
assert second is first, "sequential readers must reuse the pooled conn"
@pytest.mark.requires_wal
def test_reads_do_not_take_writer_lock(db):
"""Reads must complete while another thread holds self._lock."""
acquired = db._lock.acquire()
assert acquired
try:
done = {}
def reader():
done["session"] = db.get_session("s1")
done["search"] = db.search_messages("graphiti", limit=10)
done["messages"] = db.get_messages("s1")
t = threading.Thread(target=reader)
t.start()
t.join(timeout=5.0)
assert not t.is_alive(), "read path blocked on writer lock"
assert done["session"]["id"] == "s1"
assert any("graphiti" in (m.get("snippet") or "") for m in done["search"])
assert len(done["messages"]) == 2
finally:
db._lock.release()
@pytest.mark.requires_wal
def test_background_state_reads_never_touch_shared_writer(db):
"""Background pollers must not race transcript writes on ``_conn``."""
db.try_acquire_compression_lock("s1", "holder", ttl_seconds=60)
assert db.request_handoff("s1", "telegram") is True
# Open this thread's read connection before poisoning the shared writer.
assert db._get_read_conn() is not None
writer = db._conn
class PoisonWriter:
def execute(self, *_args, **_kwargs):
raise AssertionError("read touched shared writer connection")
db._conn = PoisonWriter()
try:
assert db.get_compression_lock_holder("s1") == "holder"
db.clear_session_activity_labels("s1")
assert db.get_handoff_state("s1") == {
"state": "pending",
"platform": "telegram",
"error": None,
}
assert [row["id"] for row in db.list_pending_handoffs()] == ["s1"]
finally:
db._conn = writer
def test_read_your_writes(db):
"""A fresh committed write must be visible to the read connection."""
db.append_message("s1", role="user", content="zanzibar checkpoint")
rows = db.search_messages("zanzibar", limit=5)
assert rows, "committed write invisible to read connection"
def test_non_wal_uses_locked_path(db):
db._wal_active = False
assert db._get_read_conn() is None
# And queries still work via the legacy path.
assert db.get_session("s1")["id"] == "s1"
@pytest.mark.requires_wal
def test_read_conn_open_failure_marks_thread(db, monkeypatch, tmp_path):
"""A failed read-conn open must not retry per query; fallback still works."""
import sqlite3 as _sqlite3
calls = {"n": 0}
real_connect = _sqlite3.connect
def failing_connect(*a, **k):
if a and isinstance(a[0], str) and a[0].startswith("file:") and "mode=ro" in a[0]:
calls["n"] += 1
raise _sqlite3.OperationalError("simulated open failure")
return real_connect(*a, **k)
fresh = SessionDB(db_path=tmp_path / "state2.db")
try:
fresh.create_session(session_id="x", source="cli", model="m")
monkeypatch.setattr("hermes_state.sqlite3.connect", failing_connect)
assert fresh.get_session("x")["id"] == "x"
assert fresh.get_session("x")["id"] == "x"
assert calls["n"] == 1, "open failure should be remembered per thread"
finally:
fresh.close()
@pytest.mark.requires_wal
def test_anchored_view_and_around_use_read_path(db):
msgs = db.get_messages("s1")
anchor = msgs[0]["id"]
acquired = db._lock.acquire()
try:
done = {}
def reader():
done["around"] = db.get_messages_around("s1", anchor, window=2)
done["view"] = db.get_anchored_view("s1", anchor, window=2, bookend=1)
t = threading.Thread(target=reader)
t.start(); t.join(timeout=5.0)
assert not t.is_alive(), "anchored reads blocked on writer lock"
assert done["around"]["window"]
assert done["view"]["window"]
finally:
db._lock.release()
@pytest.mark.requires_wal
def test_session_resume_reads_do_not_take_writer_lock(db):
"""session.resume's three read paths must not convoy behind writer flushes.
get_messages_as_conversation / get_resume_conversations /
get_ancestor_display_prefix are the hottest reads in the file — every
resume across the gateway, CLI, and ACP adapter goes through one of
them — so they must use the same per-thread read-only connection as
get_messages, not the legacy self._lock path.
"""
db.create_session(session_id="parent1", source="cli", model="m")
db.append_message("parent1", role="user", content="parent turn")
db.append_message("parent1", role="assistant", content="parent reply")
db.create_session(session_id="child1", source="cli", model="m", parent_session_id="parent1")
db.append_message("child1", role="user", content="child turn")
db.append_message("child1", role="assistant", content="child reply")
acquired = db._lock.acquire()
try:
done = {}
def reader():
done["conversation"] = db.get_messages_as_conversation("s1")
done["resume"] = db.get_resume_conversations("child1")
done["ancestor_prefix"] = db.get_ancestor_display_prefix("child1")
t = threading.Thread(target=reader)
t.start(); t.join(timeout=5.0)
assert not t.is_alive(), "session resume reads blocked on writer lock"
assert len(done["conversation"]) == 2
model_history, display_history = done["resume"]
assert len(model_history) == 2
assert len(display_history) == 4
assert len(done["ancestor_prefix"]) == 2
finally:
db._lock.release()