mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-23 16:36:23 +00:00
The trigram MATCH branch in search_messages() had the same OperationalError-only catch that #66420 fixed on the main FTS5 branch: a corrupt messages_fts_trigram shadow table raises the malformed / 'fts5: corrupt structure record' class (sqlite3.DatabaseError, parent of OperationalError), which propagated straight out of search_messages and crashed CJK session/history search for read-only sessions. Route that class through the shared one-shot _try_runtime_fts_rebuild() and retry the trigram query (catch moved outside self._lock so rebuild_fts() can re-acquire it, mirroring the main branch). If the rebuild is refused (guard consumed / FTS disabled / different error) or the retry fails, fall through to the existing LIKE substring fallback — which reads only the canonical messages table — instead of raising, so CJK search degrades gracefully rather than crashing. Adds two regression tests: trigram search self-heals in place after shadow-table corruption (answers from the rebuilt trigram index, not the LIKE fallback), and degrades to LIKE without raising when the one-shot rebuild was already consumed. Follow-up to #66420; refs #66296 #66724
218 lines
8.8 KiB
Python
218 lines
8.8 KiB
Python
"""Runtime FTS-corruption self-heal on the SessionDB write path (#65637 class).
|
|
|
|
A corrupted FTS5 shadow table (``messages_fts_data``) makes every message
|
|
write raise ``sqlite3.DatabaseError: database disk image is malformed``
|
|
through the FTS sync triggers, while the canonical ``messages`` rows stay
|
|
intact. Before this fix the gateway swallowed the failure at debug level and
|
|
the in-memory session advanced while disk silently fell behind — surfacing
|
|
later as "Persisted transcript lagged live cached history" amnesia.
|
|
|
|
The fix: ``_execute_write`` detects the malformed-image class, performs a
|
|
one-shot in-place FTS rebuild (FTS5 ``'rebuild'`` command — index rewritten
|
|
from canonical rows, no messages touched), and retries the failed write.
|
|
"""
|
|
|
|
import sqlite3
|
|
|
|
import pytest
|
|
|
|
from hermes_state import SessionDB
|
|
|
|
|
|
@pytest.fixture
|
|
def db(tmp_path):
|
|
d = SessionDB(db_path=tmp_path / "state.db")
|
|
yield d
|
|
try:
|
|
d.close()
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def _corrupt_fts(db_path):
|
|
raw = sqlite3.connect(str(db_path))
|
|
raw.execute(
|
|
"UPDATE messages_fts_data SET block = X'DEADBEEFDEADBEEFDEADBEEFDEADBEEF'"
|
|
)
|
|
raw.commit()
|
|
raw.close()
|
|
|
|
|
|
def _corrupt_trigram_fts(db_path):
|
|
raw = sqlite3.connect(str(db_path))
|
|
raw.execute(
|
|
"UPDATE messages_fts_trigram_data "
|
|
"SET block = X'DEADBEEFDEADBEEFDEADBEEFDEADBEEF'"
|
|
)
|
|
raw.commit()
|
|
raw.close()
|
|
|
|
|
|
def _message_contents(db_path):
|
|
raw = sqlite3.connect(str(db_path))
|
|
rows = raw.execute("SELECT content FROM messages ORDER BY id").fetchall()
|
|
raw.close()
|
|
return [r[0] for r in rows]
|
|
|
|
|
|
class TestRuntimeFtsRebuild:
|
|
def test_corruption_error_classification_covers_both_sqlite_messages(self):
|
|
"""SQLite's message for a corrupt FTS index varies by version: older
|
|
builds raise the generic malformed-image error, newer builds raise an
|
|
FTS5-specific one. Both must trigger the self-heal."""
|
|
assert SessionDB._is_fts_write_corruption_error(
|
|
sqlite3.DatabaseError("database disk image is malformed")
|
|
)
|
|
assert SessionDB._is_fts_write_corruption_error(
|
|
sqlite3.DatabaseError(
|
|
'fts5: corrupt structure record for table "messages_fts"'
|
|
)
|
|
)
|
|
assert not SessionDB._is_fts_write_corruption_error(
|
|
sqlite3.DatabaseError("no such table: nothing_fts_related")
|
|
)
|
|
|
|
def test_append_self_heals_after_fts_corruption(self, db, tmp_path):
|
|
if not db._fts_enabled:
|
|
pytest.skip("FTS5 unavailable in this build")
|
|
db.create_session("s1", source="test")
|
|
db.append_message("s1", "user", "hello world")
|
|
|
|
_corrupt_fts(tmp_path / "state.db")
|
|
|
|
# Before the fix this raised DatabaseError and the row was lost.
|
|
msg_id = db.append_message("s1", "user", "healed append")
|
|
assert msg_id is not None
|
|
assert _message_contents(tmp_path / "state.db") == [
|
|
"hello world",
|
|
"healed append",
|
|
]
|
|
|
|
def test_search_works_after_self_heal(self, db, tmp_path):
|
|
if not db._fts_enabled:
|
|
pytest.skip("FTS5 unavailable in this build")
|
|
db.create_session("s1", source="test")
|
|
db.append_message("s1", "user", "before corruption")
|
|
_corrupt_fts(tmp_path / "state.db")
|
|
db.append_message("s1", "user", "searchable needle text")
|
|
|
|
raw = sqlite3.connect(str(tmp_path / "state.db"))
|
|
hits = raw.execute(
|
|
"SELECT rowid FROM messages_fts WHERE messages_fts MATCH 'needle'"
|
|
).fetchall()
|
|
raw.close()
|
|
assert len(hits) == 1
|
|
|
|
def test_search_messages_self_heals_after_fts_corruption(self, db, tmp_path):
|
|
"""A read-only session that only SEARCHES (no write after corruption)
|
|
must self-heal too. The MATCH read raises the corruption class
|
|
(DatabaseError / 'fts5: corrupt structure record'), NOT the
|
|
OperationalError that search_messages caught — so before this fix the
|
|
search crashed until a write or restart rebuilt the index.
|
|
"""
|
|
if not db._fts_enabled:
|
|
pytest.skip("FTS5 unavailable in this build")
|
|
db.create_session("s1", source="test")
|
|
db.append_message("s1", "user", "a searchable needle here")
|
|
|
|
_corrupt_fts(tmp_path / "state.db")
|
|
# Injected via a raw connection, so no write on THIS instance has
|
|
# consumed the one-shot rebuild yet.
|
|
assert db._fts_runtime_rebuild_attempted is False
|
|
|
|
results = db.search_messages("needle")
|
|
|
|
assert db._fts_runtime_rebuild_attempted is True # the search rebuilt it
|
|
assert results # non-empty: the rebuilt index matched the query
|
|
assert any("needle" in (r.get("snippet") or "") for r in results)
|
|
|
|
def test_trigram_search_self_heals_after_fts_corruption(self, db, tmp_path):
|
|
"""The CJK/trigram MATCH branch has the same read-corruption exposure
|
|
as the main FTS5 branch: it caught only OperationalError (query
|
|
syntax), so a corrupt trigram shadow table raised DatabaseError
|
|
straight out of search_messages. It must self-heal via the shared
|
|
one-shot rebuild and answer from the rebuilt trigram index.
|
|
"""
|
|
if not db._fts_enabled:
|
|
pytest.skip("FTS5 unavailable in this build")
|
|
if not db._trigram_available:
|
|
pytest.skip("trigram tokenizer unavailable in this build")
|
|
db.create_session("s1", source="test")
|
|
db.append_message("s1", "user", "关于大别山项目的进展报告")
|
|
|
|
_corrupt_trigram_fts(tmp_path / "state.db")
|
|
assert db._fts_runtime_rebuild_attempted is False
|
|
|
|
# >=3 CJK chars per token → routed to the trigram branch.
|
|
results = db.search_messages("大别山项目")
|
|
|
|
assert db._fts_runtime_rebuild_attempted is True # search rebuilt it
|
|
assert results
|
|
# The rebuilt trigram index answered (trigram snippets use >>> <<<),
|
|
# i.e. we did not silently degrade to the LIKE fallback.
|
|
assert any(">>>" in (r.get("snippet") or "") for r in results)
|
|
|
|
def test_trigram_search_falls_back_to_like_when_rebuild_consumed(
|
|
self, db, tmp_path
|
|
):
|
|
"""When the one-shot rebuild was already consumed, a corrupt trigram
|
|
index must NOT crash search_messages — it degrades to the LIKE
|
|
substring fallback, which reads only the canonical messages table.
|
|
"""
|
|
if not db._fts_enabled:
|
|
pytest.skip("FTS5 unavailable in this build")
|
|
if not db._trigram_available:
|
|
pytest.skip("trigram tokenizer unavailable in this build")
|
|
db.create_session("s1", source="test")
|
|
db.append_message("s1", "user", "关于大别山项目的进展报告")
|
|
|
|
# Consume the one-shot guard, then corrupt again.
|
|
_corrupt_trigram_fts(tmp_path / "state.db")
|
|
db.append_message("s1", "user", "seed to trigger write-path heal")
|
|
assert db._fts_runtime_rebuild_attempted is True
|
|
_corrupt_trigram_fts(tmp_path / "state.db")
|
|
|
|
# Before the fix this raised sqlite3.DatabaseError.
|
|
results = db.search_messages("大别山项目")
|
|
assert results # LIKE fallback found the canonical row
|
|
assert any("大别山项目" in (r.get("snippet") or "") for r in results)
|
|
|
|
def test_rebuild_is_one_shot_per_instance(self, db, tmp_path):
|
|
if not db._fts_enabled:
|
|
pytest.skip("FTS5 unavailable in this build")
|
|
db.create_session("s1", source="test")
|
|
db.append_message("s1", "user", "seed")
|
|
_corrupt_fts(tmp_path / "state.db")
|
|
db.append_message("s1", "user", "first heal") # consumes the one shot
|
|
assert db._fts_runtime_rebuild_attempted is True
|
|
|
|
# Corrupt again: the guard must NOT loop — the write now propagates.
|
|
_corrupt_fts(tmp_path / "state.db")
|
|
with pytest.raises(sqlite3.DatabaseError):
|
|
db.append_message("s1", "user", "second corruption")
|
|
|
|
def test_non_fts_errors_still_propagate(self, db):
|
|
db.create_session("s1", source="test")
|
|
|
|
def _bad(conn):
|
|
raise sqlite3.IntegrityError("NOT NULL constraint failed: x.y")
|
|
|
|
with pytest.raises(sqlite3.IntegrityError):
|
|
db._execute_write(_bad)
|
|
# The guard must not have been consumed by an unrelated error class.
|
|
assert db._fts_runtime_rebuild_attempted is False
|
|
|
|
def test_lock_retry_path_unchanged(self, db):
|
|
"""A locked error still follows the jitter-retry path, untouched by
|
|
the DatabaseError handler (OperationalError is caught first)."""
|
|
calls = {"n": 0}
|
|
|
|
def _flaky(conn):
|
|
calls["n"] += 1
|
|
if calls["n"] < 3:
|
|
raise sqlite3.OperationalError("database is locked")
|
|
return "ok"
|
|
|
|
assert db._execute_write(_flaky) == "ok"
|
|
assert calls["n"] == 3
|
|
assert db._fts_runtime_rebuild_attempted is False
|