fix(state): bound WAL growth and checkpoint after VACUUM
`sessions optimize` could consume several GB of disk instead of freeing any, filling the host to 100% on exactly the large databases it exists to shrink. Two causes, both in the WAL lifecycle: 1. No `journal_size_limit`. SQLite defaults to -1 (unlimited), so after a checkpoint the WAL is reused in place and never truncated — `state.db-wal` permanently keeps the high-water mark of the largest transaction ever run. `hermes_cli/kanban_db.py` already bounds its WAL with `wal_autocheckpoint=100`; the session store, by far the larger database, had no equivalent. 2. `vacuum()` checkpoints BEFORE `VACUUM` but not after. VACUUM rewrites every page through the WAL, so the pre-checkpoint does nothing about the slack VACUUM itself creates. Measured on a 3.0 GB state.db: `hermes sessions optimize` reported "3143.9 MB -> 3155.1 MB (reclaimed -11.2 MB)" while leaving a 3.07 GB state.db-wal behind. Free space fell from 6.9 GB to 772 MB (100% full) and stayed there. A manual `PRAGMA wal_checkpoint(TRUNCATE)` recovered the full 3.07 GB, confirming it was slack, not data. Fix: set `journal_size_limit` (64 MiB) when enabling WAL, and truncate the WAL again after VACUUM. Both are best-effort and never raise — a failure costs disk slack and must not stop the DB from opening. Tests assert the contract (limit is a finite positive bound; VACUUM does not leave an oversized WAL) rather than pinning the byte count, which is a tunable. They skip where WAL is unavailable — including hosts where Hermes falls back to journal_mode=DELETE due to the SQLite 3.50.4 WAL-reset bug. Verified: 462 passed / 3 skipped in tests/test_hermes_state.py, and _apply_wal_size_limit flips a real WAL database from -1 to 67108864. Tested on Linux (aarch64, Python 3.11).
This commit is contained in:
@@ -543,6 +543,11 @@ _WAL_INCOMPAT_MARKERS = (
|
||||
"disk i/o error", # ZFS SHM corruption under concurrent connections
|
||||
)
|
||||
|
||||
# Upper bound for the write-ahead log. SQLite defaults to -1 (unlimited),
|
||||
# which lets state.db-wal keep the high-water mark of the largest-ever
|
||||
# transaction forever. See _apply_wal_size_limit().
|
||||
_WAL_SIZE_LIMIT_BYTES = 64 * 1024 * 1024 # 64 MiB
|
||||
|
||||
# Last SessionDB() init error, per-process. Surfaced in /resume and
|
||||
# related slash-command error strings so users know WHY the DB is
|
||||
# unavailable instead of getting a bare "Session database not available."
|
||||
@@ -741,6 +746,45 @@ def _on_disk_journal_mode(conn: sqlite3.Connection) -> Optional[str]:
|
||||
return str(mode).strip().lower() if mode is not None else None
|
||||
|
||||
|
||||
def _apply_wal_size_limit(conn: sqlite3.Connection) -> None:
|
||||
"""Bound the WAL so it returns space to the OS after big transactions.
|
||||
|
||||
SQLite's default ``journal_size_limit`` is -1 (unlimited): after a
|
||||
checkpoint the WAL file is *reused in place* and never truncated, so
|
||||
``state.db-wal`` permanently retains the high-water mark of the largest
|
||||
transaction ever run against it.
|
||||
|
||||
A single bulk operation is enough to strand gigabytes. Observed on a
|
||||
3.0 GB ``state.db``: ``hermes sessions optimize`` (FTS merge + VACUUM)
|
||||
rewrites every page through the WAL, leaving a **3.07 GB**
|
||||
``state.db-wal`` sitting next to the database indefinitely — the host
|
||||
went from 6.9 GB free to 772 MB (100% full) and stayed there, because
|
||||
nothing shrinks the WAL back down. An explicit
|
||||
``PRAGMA wal_checkpoint(TRUNCATE)`` reclaimed the full 3.07 GB, which
|
||||
confirms the space was pure slack rather than live data.
|
||||
|
||||
That also makes the maintenance command self-defeating on exactly the
|
||||
databases that need it most: the larger the DB, the larger the WAL it
|
||||
strands, so ``optimize`` can consume more disk than it frees.
|
||||
|
||||
``journal_size_limit`` makes SQLite truncate the WAL back to the limit
|
||||
at each checkpoint. 64 MiB is comfortably above normal transaction
|
||||
sizes (so steady-state commits never pay a truncate) while capping the
|
||||
stranded slack at a bounded, predictable figure.
|
||||
|
||||
``hermes_cli/kanban_db.py`` already bounds its WAL growth with
|
||||
``wal_autocheckpoint=100``; the session store — by far the larger
|
||||
database — had no equivalent.
|
||||
|
||||
Best-effort: never raises. A failure here only costs disk slack, and
|
||||
must not prevent the database from opening.
|
||||
"""
|
||||
try:
|
||||
conn.execute(f"PRAGMA journal_size_limit={_WAL_SIZE_LIMIT_BYTES}")
|
||||
except sqlite3.OperationalError as exc: # pragma: no cover - defensive
|
||||
logger.debug("journal_size_limit not applied: %s", exc)
|
||||
|
||||
|
||||
def _apply_macos_checkpoint_barrier(conn: sqlite3.Connection) -> None:
|
||||
"""Enable ``PRAGMA checkpoint_fullfsync`` on macOS (no-op elsewhere).
|
||||
|
||||
@@ -947,6 +991,7 @@ def apply_wal_with_fallback(
|
||||
# Skipping the set-pragma prevents WAL-init from unlinking files other connections hold open.
|
||||
current_mode = _on_disk_journal_mode(conn)
|
||||
if current_mode == "wal":
|
||||
_apply_wal_size_limit(conn)
|
||||
_apply_macos_checkpoint_barrier(conn)
|
||||
_enforce_macos_synchronous_full(conn)
|
||||
return "wal"
|
||||
@@ -987,6 +1032,7 @@ def apply_wal_with_fallback(
|
||||
row = conn.execute("PRAGMA journal_mode=WAL").fetchone()
|
||||
mode = str(row[0]).strip().lower() if row and row[0] is not None else ""
|
||||
if mode == "wal":
|
||||
_apply_wal_size_limit(conn)
|
||||
_apply_macos_checkpoint_barrier(conn)
|
||||
_enforce_macos_synchronous_full(conn)
|
||||
return "wal"
|
||||
@@ -1035,6 +1081,7 @@ def apply_wal_with_fallback(
|
||||
else ""
|
||||
)
|
||||
if mode == "wal":
|
||||
_apply_wal_size_limit(conn)
|
||||
_apply_macos_checkpoint_barrier(conn)
|
||||
_enforce_macos_synchronous_full(conn)
|
||||
return "wal"
|
||||
@@ -1114,6 +1161,7 @@ def _apply_delete_for_wal_reset_bug(
|
||||
# Do not TRUNCATE / journal_mode=DELETE while other processes may
|
||||
# still hold this WAL DB open — same safety rule as the NFS path.
|
||||
_log_wal_reset_bug_once(db_label, kept_wal=True)
|
||||
_apply_wal_size_limit(conn)
|
||||
_apply_macos_checkpoint_barrier(conn)
|
||||
_enforce_macos_synchronous_full(conn)
|
||||
return "wal"
|
||||
@@ -11089,6 +11137,17 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
|
||||
except Exception as exc:
|
||||
logger.debug("WAL checkpoint (PASSIVE) before VACUUM failed: %s", exc)
|
||||
self._conn.execute("VACUUM")
|
||||
# ...and again afterwards. VACUUM rewrites every page THROUGH the
|
||||
# WAL, so the pre-VACUUM checkpoint above does nothing for the
|
||||
# slack VACUUM itself creates: on a 3.0 GB database it left a
|
||||
# 3.07 GB state.db-wal behind, so `sessions optimize` reported
|
||||
# "reclaimed -11.2 MB" while actually consuming 3 GB of disk and
|
||||
# filling the host to 100%. Truncating here is what makes the
|
||||
# command a net win instead of a net loss on large databases.
|
||||
try:
|
||||
self._conn.execute("PRAGMA wal_checkpoint(TRUNCATE)")
|
||||
except Exception as exc:
|
||||
logger.debug("WAL checkpoint (TRUNCATE) after VACUUM failed: %s", exc)
|
||||
return optimized
|
||||
|
||||
def maybe_auto_prune_and_vacuum(
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
import sqlite3
|
||||
import time
|
||||
import json
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
import pytest
|
||||
@@ -2288,6 +2289,45 @@ class TestVacuum:
|
||||
assert vacuum_calls == [True, True]
|
||||
assert db.get_meta("last_vacuum") is not None
|
||||
|
||||
def test_wal_size_limit_is_bounded(self, db):
|
||||
"""journal_size_limit must be a finite bound, not SQLite's -1 default.
|
||||
|
||||
Contract, not a snapshot: assert the limit is positive (so the WAL is
|
||||
truncated back at checkpoints) rather than pinning the exact byte
|
||||
count, which is a tunable.
|
||||
"""
|
||||
mode = db._conn.execute("PRAGMA journal_mode").fetchone()[0]
|
||||
if str(mode).lower() != "wal":
|
||||
pytest.skip("WAL unavailable on this filesystem")
|
||||
limit = db._conn.execute("PRAGMA journal_size_limit").fetchone()[0]
|
||||
assert limit > 0, "unbounded WAL: state.db-wal never returns disk to the OS"
|
||||
|
||||
def test_vacuum_leaves_wal_truncated(self, db, tmp_path):
|
||||
"""VACUUM must not strand a giant WAL beside the database.
|
||||
|
||||
VACUUM rewrites every page through the write-ahead log. Without a
|
||||
checkpoint *after* it, a 3 GB database leaves a 3 GB state.db-wal
|
||||
behind — `sessions optimize` then consumes far more disk than it
|
||||
frees, which is the opposite of its purpose.
|
||||
"""
|
||||
mode = db._conn.execute("PRAGMA journal_mode").fetchone()[0]
|
||||
if str(mode).lower() != "wal":
|
||||
pytest.skip("WAL unavailable on this filesystem")
|
||||
|
||||
db.create_session(session_id="s1", source="cli")
|
||||
for i in range(500):
|
||||
db.append_message(
|
||||
session_id="s1", role="user", content=f"padding message {i} " * 20
|
||||
)
|
||||
db.vacuum()
|
||||
|
||||
wal = Path(str(db.db_path) + "-wal")
|
||||
if wal.exists():
|
||||
limit = db._conn.execute("PRAGMA journal_size_limit").fetchone()[0]
|
||||
assert wal.stat().st_size <= max(limit, 0) or wal.stat().st_size == 0, (
|
||||
f"WAL left at {wal.stat().st_size} bytes after VACUUM"
|
||||
)
|
||||
|
||||
|
||||
class TestOptimizeFts:
|
||||
def test_optimize_returns_index_count(self, db):
|
||||
|
||||
Reference in New Issue
Block a user