fix(gateway): add disk I/O error to WAL-fallback markers for ZFS (#55305)

This commit is contained in:
Sahil-SS9 2026-06-30 02:07:17 +01:00 committed by Teknium
parent 91351b7b77
commit 7dbf6c2589

View file

@ -247,9 +247,13 @@ DEFAULT_DB_PATH = get_hermes_home() / "state.db"
# propagate that, every feature backed by state.db / kanban.db breaks
# silently — /resume, /title, /history, /branch, kanban dispatcher, etc.
#
# ZFS is a separate case: its COW + mmap semantics can corrupt the WAL
# shared-memory (-shm) file under concurrent connection bursts, presenting
# as ``disk I/O error`` rather than ``locking protocol``.
#
# Instead, fall back to ``journal_mode=DELETE`` (the pre-WAL default) which
# works on NFS. Concurrency drops — concurrent readers are blocked during
# a write — but the feature works.
# works on NFS and ZFS. Concurrency drops — concurrent readers are blocked
# during a write — but the feature works.
#
# Separately, SQLite's WAL-reset bug can corrupt multi-process WAL databases
# on unfixed library builds (issue #69784). See:
@ -262,6 +266,7 @@ DEFAULT_DB_PATH = get_hermes_home() / "state.db"
_WAL_INCOMPAT_MARKERS = (
"locking protocol", # SQLITE_PROTOCOL on NFS/SMB
"not authorized", # Some FUSE mounts block WAL pragma outright
"disk i/o error", # ZFS SHM corruption under concurrent connections
)
# Last SessionDB() init error, per-process. Surfaced in /resume and
@ -386,7 +391,7 @@ def format_session_db_unavailable(prefix: str = "Session database not available"
return f"{prefix}."
hint = ""
if any(marker in cause.lower() for marker in _WAL_INCOMPAT_MARKERS):
hint = " (state.db may be on NFS/SMB/FUSE — see https://www.sqlite.org/wal.html)"
hint = " (state.db may be on NFS/SMB/FUSE/ZFS — see https://www.sqlite.org/wal.html)"
return f"{prefix}: {cause}{hint}."
@ -541,10 +546,10 @@ def apply_wal_with_fallback(
Returns the journal mode actually set (``"wal"`` or ``"delete"``).
On WAL-incompatible filesystems (NFS, SMB, some FUSE), SQLite raises
``OperationalError("locking protocol")`` when setting WAL. We fall
back to DELETE mode the pre-WAL default, which works on NFS and
log one WARNING explaining why.
On WAL-incompatible filesystems (NFS, SMB, some FUSE, ZFS), SQLite raises
``OperationalError("locking protocol")`` or ``OperationalError("disk I/O error")``
when setting WAL. We fall back to DELETE mode the pre-WAL default, which
works on NFS and ZFS and log one WARNING explaining why.
On SQLite builds that still contain the WAL-reset corruption bug
(issue #69784), refuse to enable WAL on fresh / non-WAL databases
@ -719,7 +724,7 @@ def _log_wal_fallback_once(db_label: str, exc: Exception) -> None:
logger.warning(
"%s: WAL journal_mode unsupported on this filesystem (%s) — "
"falling back to journal_mode=DELETE (slower rollback-journal "
"mode; reduces concurrency but works on NFS/SMB/FUSE). See "
"mode; reduces concurrency but works on NFS/SMB/FUSE/ZFS). See "
"https://www.sqlite.org/wal.html for details. This warning "
"fires once per process per database.",
db_label,