fix(state): report closed stale-open count from auto-maintenance, document the sweep (#54189)

Follow-up on top of the salvaged #94095 commit:
- maybe_auto_prune_and_vacuum() now returns 'closed' (stale open state-owned
  sessions marked ended) alongside 'pruned', so entrypoints can report the
  reconciliation without parsing logs.
- Docstring explains the two-window lifecycle (close now, delete after a
  further retention window).
- Regression test: cron/kanban/subagent rows with ended_at NULL are closed on
  pass 1 and deleted on pass 2; a telegram row is never touched.
- website/docs sessions.md documents the automatic stale-open sweep.
This commit is contained in:
Teknium
2026-09-01 21:26:52 -07:00
parent 0aa84bb3fd
commit 9bc249c7e5
3 changed files with 72 additions and 2 deletions
+19 -2
View File
@@ -16273,16 +16273,33 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
(``.json`` / ``.jsonl`` / ``request_dump_*``) for pruned sessions
are removed as part of the same sweep (issue #3015).
Stale-open reconciliation (#54189): several state-owned producers
(cron, kanban workers, subagents, one-shot CLI runs) never set
``ended_at`` when their process dies, and ``prune_sessions`` only
deletes ended rows — so retention was a no-op exactly where growth
concentrates. After pruning, this pass closes open rows from
:attr:`_AUTO_PRUNE_STALE_OPEN_SOURCES` whose activity is older than
``retention_days`` (``end_reason='startup_orphan_reap'``). Closed rows
stay resumable and are aged from their close, so they get one more
full retention window before a later pass deletes them. Messaging
and UI sources are never touched here.
Never raises. On any failure, logs a warning and returns a dict
with ``"error"`` set.
Returns a dict with keys:
- ``"skipped"`` (bool) — true if within min_interval_hours of last run
- ``"pruned"`` (int) — number of sessions deleted
- ``"closed"`` (int) — stale open state-owned sessions marked ended
- ``"vacuumed"`` (bool) — true if VACUUM ran
- ``"error"`` (str, optional) — present only on failure
"""
result: Dict[str, Any] = {"skipped": False, "pruned": 0, "vacuumed": False}
result: Dict[str, Any] = {
"skipped": False,
"pruned": 0,
"closed": 0,
"vacuumed": False,
}
maintenance_lock = _try_acquire_auto_maintenance_lock(self.db_path)
if maintenance_lock is None:
result["skipped"] = True
@@ -16320,7 +16337,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
# dashboard/TUI gateway heartbeats used by startup recovery.
respect_gateway_heartbeats=False,
)
result["closed"] = len(closed)
# Only VACUUM if we actually freed rows, and no more often than
# once every min_vacuum_interval_days -- a large prune (e.g. the
# first one to cross retention_days on a DB with tens of
@@ -536,6 +536,46 @@ class TestSweepOrphanedSessions:
def test_returns_empty_on_empty_db(self, db):
assert db.sweep_orphaned_sessions(max_idle_seconds=IDLE_S) == []
def test_auto_prune_reports_closed_count_and_deletes_after_second_window(
self, db
):
"""#54189 end-to-end: leaky producers (cron/kanban/subagent) never set
``ended_at``; pass 1 closes them (reported via ``closed``), pass 2 —
after a further retention window — deletes them, and a messaging row
is never touched by either pass."""
stale = time.time() - 200 * 86400
for sid, source in (
("cron-0", "cron"),
("kanban-1", "kanban"),
("subagent-2", "subagent"),
("telegram-3", "telegram"),
):
_make_session(db, sid, source=source, started_at=stale, message_at=stale)
_set_last_activity(db, sid, stale)
first = db.maybe_auto_prune_and_vacuum(
retention_days=90, min_interval_hours=0, vacuum=False
)
assert first["closed"] == 3
assert first["pruned"] == 0
for sid in ("cron-0", "kanban-1", "subagent-2"):
assert db.get_session(sid)["end_reason"] == "startup_orphan_reap"
assert db.get_session("telegram-3")["ended_at"] is None
# Simulate the next maintenance pass after another retention window.
db._conn.execute(
"UPDATE sessions SET ended_at = ended_at - 91 * 86400 "
"WHERE end_reason = 'startup_orphan_reap'"
)
db._conn.commit()
second = db.maybe_auto_prune_and_vacuum(
retention_days=90, min_interval_hours=0, vacuum=False
)
assert second["closed"] == 0
assert second["pruned"] == 3
remaining = [r["id"] for r in db._conn.execute("SELECT id FROM sessions")]
assert remaining == ["telegram-3"]
def test_zero_ttl_is_noop(self, db):
stale = time.time() - 8 * 3600
_make_session(db, "stale-tui", source="tui", started_at=stale, message_at=stale)
+13
View File
@@ -888,6 +888,19 @@ Active sessions are never auto-pruned, regardless of age. Ended sessions are
aged from their latest message, so a long-lived conversation used recently is
not deleted merely because it began before the retention window.
**Stale open sessions from automation.** Some producers — cron jobs, kanban
workers, subagents, one-shot CLI runs — can die without ever marking their
session ended, and pruning only deletes *ended* rows. To keep those from
accumulating forever, each auto-prune pass also *closes* open sessions from
those state-owned sources (`cli`, `cron`, `kanban`, `acp`, `api_server`,
`subagent`, `tool`) whose last activity is older than `retention_days`
(`end_reason: startup_orphan_reap`). Closing is non-destructive — the
session stays resumable — and the row is aged from its close, so it is only
deleted by a *later* pass after a further full retention window. Messaging
platform sessions (Telegram, Discord, …), TUI/desktop sessions, pinned
sessions, and sessions with a live turn or compression in progress are
never closed by this sweep.
### Oversized-Transcript Guards
Two limits stop a runaway transcript from being loaded into memory all at once