hermes-agent/tests/test_hermes_state_wal_fallback.py
Connor Black f50d80e8eb fix(state): detect silent WAL→DELETE fallback on macOS NFS / SMB (no false "wal")
apply_wal_with_fallback() only detected WAL-incompatible filesystems via a
RAISED OperationalError matched against _WAL_INCOMPAT_MARKERS. But macOS NFS,
SMB/CIFS, and overlay filesystems (e.g. AgentFS's NFS-backed mount) refuse the
WAL switch WITHOUT raising: `PRAGMA journal_mode=WAL` returns the still-effective
mode ('delete') and no exception. The code then ran `return "wal"`
unconditionally, so it:
  1. returned a false "wal" while the DB was actually in DELETE mode, and
  2. never called _log_wal_fallback_once, so the operator got ZERO signal that
     concurrency had silently degraded (reader-blocks-writer).

state.db and kanban.db share this path, so a session/kanban board DB on a
network or overlay filesystem ran in DELETE with no diagnostic.

Fix: read the row `PRAGMA journal_mode=WAL` returns and verify it is actually
'wal' instead of assuming success; on a silent no-op, emit the existing
fallback WARNING and return the true mode. The raise-based path is unchanged.

Reproduced on a real AgentFS NFS overlay (PRAGMA journal_mode=WAL returned
('delete',) with no OperationalError). Adds a regression test for the
silent-no-op shape; the 16 existing WAL-fallback tests are unchanged.
2026-07-29 18:13:09 -07:00

291 lines
12 KiB
Python

"""Tests for the WAL→DELETE journal-mode fallback on NFS / SMB / FUSE.
When ``PRAGMA journal_mode=WAL`` raises ``OperationalError("locking protocol")``
(SQLITE_PROTOCOL — typical on NFS/SMB), Hermes must fall back to
``journal_mode=DELETE`` so ``state.db`` / ``kanban.db`` remain usable.
Without this fallback, users on NFS-mounted ``HERMES_HOME`` silently lose
``/resume``, ``/title``, ``/history``, ``/branch``, session search, and the
kanban dispatcher — because ``SessionDB()`` init propagates the error and
every caller swallows it, leaving ``_session_db = None``.
See: https://www.sqlite.org/wal.html — "WAL does not work over a network
filesystem".
"""
import sqlite3
from unittest.mock import patch
import pytest
import hermes_state
from hermes_state import (
SessionDB,
apply_wal_with_fallback,
format_session_db_unavailable,
get_last_init_error,
)
# ``sqlite3.Connection.execute`` is a C-level slot and can't be monkeypatched
# directly (``'sqlite3.Connection' object attribute 'execute' is read-only``).
# A factory-built subclass lets us intercept journal_mode=WAL per-test with
# its own mutable counter, avoiding the xdist-parallel class-state race.
def _make_blocking_factory(reason: str, attempt_counter: list):
"""Return a sqlite3.Connection subclass that raises on PRAGMA journal_mode=WAL."""
class _WalBlockingConnection(sqlite3.Connection):
def execute(self, sql, *args, **kwargs): # type: ignore[override]
if "journal_mode=wal" in sql.lower().replace(" ", ""):
attempt_counter[0] += 1
raise sqlite3.OperationalError(reason)
return super().execute(sql, *args, **kwargs)
return _WalBlockingConnection
def _open_blocking(path, reason="locking protocol", **kwargs):
"""Open a connection whose WAL pragma raises ``reason``.
Returns ``(conn, attempt_counter_list)`` so callers can assert how many
times WAL was attempted.
"""
attempts = [0]
factory = _make_blocking_factory(reason, attempts)
return sqlite3.connect(str(path), factory=factory, **kwargs), attempts
def _make_silent_noop_factory(returned_mode: str = "delete"):
"""Return a ``sqlite3.Connection`` subclass whose ``PRAGMA journal_mode=WAL``
silently NO-OPs: it returns the still-effective journal mode (e.g.
``delete``) WITHOUT raising — the way macOS NFS / SMB / the AgentFS NFS
overlay behave. NFS that *raises* SQLITE_PROTOCOL is covered by
:func:`_make_blocking_factory`; this is the other, quieter failure shape.
"""
class _WalSilentNoOpConnection(sqlite3.Connection):
def execute(self, sql, *args, **kwargs): # type: ignore[override]
if "journal_mode=wal" in sql.lower().replace(" ", ""):
# Refuse the WAL switch but DON'T raise; report the mode that is
# actually still in force, exactly as the NFS client does.
return super().execute(
f"PRAGMA journal_mode={returned_mode}", *args, **kwargs
)
return super().execute(sql, *args, **kwargs)
return _WalSilentNoOpConnection
@pytest.fixture(autouse=True)
def _reset_last_init_error():
"""Reset the module-global last-error before and after each test."""
hermes_state._set_last_init_error(None)
yield
hermes_state._set_last_init_error(None)
@pytest.fixture(autouse=True)
def _reset_wal_fallback_warned_paths():
"""Reset the WAL-fallback warned-paths set so dedup doesn't leak between tests."""
hermes_state._wal_fallback_warned_paths.clear()
yield
hermes_state._wal_fallback_warned_paths.clear()
@pytest.fixture(autouse=True)
def _assume_fixed_sqlite(monkeypatch):
"""NFS-fallback tests assume a SQLite build without the WAL-reset bug."""
monkeypatch.setattr(
hermes_state, "is_sqlite_wal_reset_vulnerable", lambda version_info=None: False
)
hermes_state._wal_reset_bug_warned_paths.clear()
yield
hermes_state._wal_reset_bug_warned_paths.clear()
class TestApplyWalWithFallback:
def test_succeeds_on_local_fs(self, tmp_path):
"""Happy path: WAL works on a normal filesystem."""
conn = sqlite3.connect(str(tmp_path / "ok.db"), isolation_level=None)
mode = apply_wal_with_fallback(conn)
assert mode == "wal"
cur = conn.execute("PRAGMA journal_mode")
assert cur.fetchone()[0].lower() == "wal"
conn.close()
def test_falls_back_when_wal_silently_refused(self, tmp_path, caplog):
"""macOS NFS / SMB / AgentFS-NFS can REFUSE the WAL switch WITHOUT
raising: ``PRAGMA journal_mode=WAL`` returns the still-effective mode
('delete') and no exception. The marker-exception path never fires, so
the function must detect the silent no-op from the PRAGMA's RETURN
value — otherwise it returns a false 'wal' and skips the WARNING.
Reproduced on a real AgentFS NFS overlay, where
``PRAGMA journal_mode=WAL`` returned ('delete',) with no
OperationalError.
"""
factory = _make_silent_noop_factory("delete")
conn = sqlite3.connect(
str(tmp_path / "macnfs.db"), factory=factory, isolation_level=None
)
with caplog.at_level("WARNING", logger="hermes_state"):
mode = apply_wal_with_fallback(conn, db_label="kanban.db")
assert mode == "delete", "must report the true mode, not a false 'wal'"
assert (
conn.execute("PRAGMA journal_mode").fetchone()[0].lower() == "delete"
)
warnings = [r for r in caplog.records if r.levelname == "WARNING"]
assert len(warnings) == 1, "silent no-op must still emit exactly one WARNING"
assert "kanban.db" in warnings[0].getMessage()
conn.close()
def test_reraises_on_disk_io_error(self, tmp_path):
"""Transient EIO from ``PRAGMA journal_mode=WAL`` must NOT silently
downgrade to DELETE.
Regression for "Bug D": treating transient EIO as a permanent
WAL-incompat marker produced the mixed-journal-mode-across-processes
corruption pattern (process A downgrades to DELETE, sibling
processes successfully set WAL, SQLite corrupts the file because
the two locking protocols are documented as incompatible). EIO is
usually transient (page-cache pressure, lock contention, brief
storage hiccups); the right behavior is to re-raise so the caller
can retry, not to walk the DB into a permanently downgraded state.
"""
conn, _ = _open_blocking(
tmp_path / "flaky.db", reason="disk I/O error", isolation_level=None
)
with pytest.raises(sqlite3.OperationalError, match="disk I/O error"):
apply_wal_with_fallback(conn)
conn.close()
def test_reraises_unrelated_operational_error(self, tmp_path):
"""Non-WAL-compat errors must NOT be silently swallowed by the fallback."""
conn, _ = _open_blocking(
tmp_path / "other.db",
reason="no such table: nope",
isolation_level=None,
)
with pytest.raises(sqlite3.OperationalError, match="no such table"):
apply_wal_with_fallback(conn)
conn.close()
def test_warning_fires_independently_per_db_label(self, tmp_path, caplog):
"""Different db_labels each get their own one warning (not globally dedup'd)."""
with caplog.at_level("WARNING", logger="hermes_state"):
conn1, _ = _open_blocking(tmp_path / "a.db", isolation_level=None)
apply_wal_with_fallback(conn1, db_label="state.db")
conn1.close()
conn2, _ = _open_blocking(tmp_path / "b.db", isolation_level=None)
apply_wal_with_fallback(conn2, db_label="kanban.db")
conn2.close()
warnings = [r for r in caplog.records if r.levelname == "WARNING"]
labels_warned = {
lbl for r in warnings for lbl in ("state.db", "kanban.db")
if lbl in r.getMessage()
}
assert labels_warned == {"state.db", "kanban.db"}, (
f"Each db_label should warn once; got {labels_warned}"
)
class TestGetLastInitError:
def test_captures_cause_on_failed_init(self, tmp_path):
"""When SessionDB() raises, the cause is preserved for slash commands.
Simulates a filesystem where BOTH WAL and DELETE journal modes fail —
e.g. a read-only mount where no ``PRAGMA journal_mode=X`` works. The
fallback tries DELETE and also gets rejected; the exception bubbles
out of ``SessionDB.__init__`` and the cause is captured.
"""
target = tmp_path / "broken.db"
real_connect = sqlite3.connect
class _BothPragmasFailConnection(sqlite3.Connection):
def execute(self, sql, *args, **kwargs): # type: ignore[override]
if "journal_mode" in sql.lower():
raise sqlite3.OperationalError(
"locking protocol: read-only filesystem"
)
return super().execute(sql, *args, **kwargs)
def gated_connect(*args, **kwargs):
# connect_tracked passes a tracking-augmented factory; drop it and
# substitute the double, which connect_tracked will re-augment.
kwargs.pop("factory", None)
return real_connect(str(target), factory=_BothPragmasFailConnection, **kwargs)
with patch("hermes_state.sqlite3.connect", side_effect=gated_connect):
with pytest.raises(sqlite3.OperationalError):
SessionDB(db_path=target)
cause = get_last_init_error()
assert cause is not None
assert "OperationalError" in cause
assert "locking protocol" in cause
class TestFormatSessionDbUnavailable:
def test_bare_message_when_no_cause(self):
"""No init error recorded → generic message."""
hermes_state._set_last_init_error(None)
assert format_session_db_unavailable() == "Session database not available."
def test_adds_nfs_hint_for_locking_protocol(self):
"""Locking-protocol cause gets an NFS/SMB pointer for the user."""
hermes_state._set_last_init_error("OperationalError: locking protocol")
msg = format_session_db_unavailable()
assert "locking protocol" in msg
assert "NFS/SMB" in msg
assert "sqlite.org/wal.html" in msg
def test_custom_prefix(self):
"""Callers can customize the prefix for context-specific messages."""
hermes_state._set_last_init_error("OperationalError: locking protocol")
msg = format_session_db_unavailable(prefix="Cannot /resume")
assert msg.startswith("Cannot /resume:")
class TestSessionDbUsesWalFallback:
def test_sessiondb_works_when_wal_unavailable(self, tmp_path):
"""E2E: SessionDB initializes and performs a write on a WAL-blocked FS."""
target = tmp_path / "nfs_style.db"
real_connect = sqlite3.connect
attempts = [0]
factory = _make_blocking_factory("locking protocol", attempts)
def gated_connect(*args, **kwargs):
# connect_tracked passes a tracking-augmented factory; drop it and
# substitute the double, which connect_tracked re-applies to the
# returned instance.
kwargs.pop("factory", None)
return real_connect(str(target), factory=factory, **kwargs)
with patch("hermes_state.sqlite3.connect", side_effect=gated_connect):
db = SessionDB(db_path=target)
try:
# WAL was attempted and rejected — fallback kicked in
assert attempts[0] >= 1, (
"WAL pragma was never executed — check the patch target"
)
# SessionDB is usable end-to-end: create a session, read it back
db.create_session(session_id="s1", source="cli", model="test")
sess = db.get_session("s1")
assert sess is not None
assert sess["source"] == "cli"
# No init error was recorded since init succeeded via the fallback
assert get_last_init_error() is None
finally:
db.close()