feat(kanban): add hermes kanban repair CLI verb

Adds kanban_db.repair_db() — a structured, non-raising wrapper around
the same narrow repair policy as the connect-time guard: probe with
PRAGMA integrity_check under the board's cross-process init flock;
quarantine the corrupt bytes FIRST via the content-addressed backup;
REINDEX only when every integrity message is index-scoped; re-check;
report ok / repaired / corrupt / missing. Locked/busy OperationalError
still propagates raw (a locked healthy DB is not corruption and gets
no quarantine), and a repair invalidates the per-process healthy-path
cache so the next connect() re-probes.

The CLI verb reports status human-readably (or --json), exits 0 for
ok/repaired/missing and 1 when the DB is still corrupt (non-index
corruption stays fail-closed with manual-recovery guidance). It
dispatches BEFORE kanban_command's auto-init: init_db() raises
KanbanDbCorruptError on a corrupt board, which previously would have
made a repair verb unreachable on exactly the boards that need it.

CLI tests drive the real argparse surface (build_parser +
kanban_command) against real corrupted SQLite fixtures.
This commit is contained in:
Teknium
2026-07-21 05:59:51 -07:00
parent 49828a3fd6
commit 60cfa11136
3 changed files with 332 additions and 0 deletions
+95
View File
@@ -880,6 +880,25 @@ def build_parser(parent_subparsers: argparse._SubParsersAction) -> argparse.Argu
p_gc.add_argument("--log-retention-days", type=int, default=30,
help="Delete worker log files older than N days (default: 30)")
# --- repair ---
p_repair = sub.add_parser(
"repair",
help="Check kanban.db integrity and auto-repair index-only corruption",
description=(
"Runs PRAGMA integrity_check on the board's DB and reports the "
"result. When the failure consists only of index-scoped errors "
"('wrong # of entries in index <name>' / 'row N missing from "
"index <name>'), the corrupt file is quarantined to a "
".corrupt.<hash>.bak sibling first and the damaged indexes are "
"rebuilt with REINDEX — the same narrow auto-repair the "
"connect-time guard applies. Any other corruption class is "
"reported and left untouched (fail-closed). Exits 0 when the DB "
"is healthy or was repaired, non-zero when it is still corrupt."
),
)
p_repair.add_argument("--json", action="store_true",
help="Emit the repair report as JSON")
kanban_parser.set_defaults(_kanban_parser=kanban_parser)
return kanban_parser
@@ -950,6 +969,12 @@ def kanban_command(args: argparse.Namespace) -> int:
# schema creation; `create` / `list` / every other command would
# error out on a fresh install.
with board_scope:
# `repair` must dispatch BEFORE the auto-init below: on a corrupt DB
# init_db() itself raises KanbanDbCorruptError, which would turn
# every `hermes kanban repair` into "could not initialize database"
# without ever reaching the repair path.
if action == "repair":
return _cmd_repair(args)
try:
kb.init_db()
except Exception as exc:
@@ -2886,6 +2911,76 @@ def _cmd_gc(args: argparse.Namespace) -> int:
return 0
def _cmd_repair(args: argparse.Namespace) -> int:
"""Check DB integrity and apply the narrow index-REINDEX auto-repair.
Dispatched BEFORE the auto ``kb.init_db()`` in :func:`kanban_command`
(init itself refuses corrupt DBs), so this is reachable on exactly the
boards that need it. Exit codes: 0 = healthy / repaired / no DB file,
1 = still corrupt (non-index corruption, or REINDEX did not produce a
clean re-check).
"""
try:
report = kb.repair_db()
except Exception as exc: # locked/busy probe, unexpected I/O
print(f"kanban repair: {exc}", file=sys.stderr)
return 1
if getattr(args, "json", False):
print(json.dumps({
"status": report.status,
"db_path": str(report.db_path),
"messages": report.messages,
"post_repair_messages": report.post_repair_messages,
"backup_path": (
str(report.backup_path) if report.backup_path else None
),
"reindexed": report.reindexed,
}, indent=2))
return 0 if report.status in {"ok", "repaired", "missing"} else 1
if report.status == "missing":
print(f"No kanban DB at {report.db_path} — nothing to repair.")
return 0
if report.status == "ok":
print(f"{report.db_path}: integrity_check ok — no repair needed.")
return 0
if report.status == "repaired":
print(f"{report.db_path}: repaired.")
print(f" reindexed: {', '.join(report.reindexed)}")
if report.backup_path:
print(f" pre-repair backup: {report.backup_path}")
print(" integrity_check now ok.")
return 0
# still corrupt
print(f"{report.db_path}: CORRUPT.", file=sys.stderr)
for line in (report.messages or [])[:10]:
print(f" {line}", file=sys.stderr)
if report.reindexed:
print(
f" REINDEX ({', '.join(report.reindexed)}) attempted but "
f"integrity_check is still failing:",
file=sys.stderr,
)
for line in (report.post_repair_messages or [])[:10]:
print(f" {line}", file=sys.stderr)
else:
print(
" Not an index-only failure — automatic REINDEX repair does "
"not apply (fail-closed).",
file=sys.stderr,
)
if report.backup_path:
print(f" corrupt copy quarantined at: {report.backup_path}",
file=sys.stderr)
print(
" Recover manually (e.g. `sqlite3 kanban.db \".recover\"` into a "
"fresh file) or move the file aside to start a new board.",
file=sys.stderr,
)
return 1
# ---------------------------------------------------------------------------
# Slash-command entry point (used by /kanban from CLI and gateway)
# ---------------------------------------------------------------------------