fix(state): report closed stale-open count from auto-maintenance, document the sweep (#54189)
Follow-up on top of the salvaged #94095 commit: - maybe_auto_prune_and_vacuum() now returns 'closed' (stale open state-owned sessions marked ended) alongside 'pruned', so entrypoints can report the reconciliation without parsing logs. - Docstring explains the two-window lifecycle (close now, delete after a further retention window). - Regression test: cron/kanban/subagent rows with ended_at NULL are closed on pass 1 and deleted on pass 2; a telegram row is never touched. - website/docs sessions.md documents the automatic stale-open sweep.
This commit is contained in:
+19
-2
@@ -16273,16 +16273,33 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
|
||||
(``.json`` / ``.jsonl`` / ``request_dump_*``) for pruned sessions
|
||||
are removed as part of the same sweep (issue #3015).
|
||||
|
||||
Stale-open reconciliation (#54189): several state-owned producers
|
||||
(cron, kanban workers, subagents, one-shot CLI runs) never set
|
||||
``ended_at`` when their process dies, and ``prune_sessions`` only
|
||||
deletes ended rows — so retention was a no-op exactly where growth
|
||||
concentrates. After pruning, this pass closes open rows from
|
||||
:attr:`_AUTO_PRUNE_STALE_OPEN_SOURCES` whose activity is older than
|
||||
``retention_days`` (``end_reason='startup_orphan_reap'``). Closed rows
|
||||
stay resumable and are aged from their close, so they get one more
|
||||
full retention window before a later pass deletes them. Messaging
|
||||
and UI sources are never touched here.
|
||||
|
||||
Never raises. On any failure, logs a warning and returns a dict
|
||||
with ``"error"`` set.
|
||||
|
||||
Returns a dict with keys:
|
||||
- ``"skipped"`` (bool) — true if within min_interval_hours of last run
|
||||
- ``"pruned"`` (int) — number of sessions deleted
|
||||
- ``"closed"`` (int) — stale open state-owned sessions marked ended
|
||||
- ``"vacuumed"`` (bool) — true if VACUUM ran
|
||||
- ``"error"`` (str, optional) — present only on failure
|
||||
"""
|
||||
result: Dict[str, Any] = {"skipped": False, "pruned": 0, "vacuumed": False}
|
||||
result: Dict[str, Any] = {
|
||||
"skipped": False,
|
||||
"pruned": 0,
|
||||
"closed": 0,
|
||||
"vacuumed": False,
|
||||
}
|
||||
maintenance_lock = _try_acquire_auto_maintenance_lock(self.db_path)
|
||||
if maintenance_lock is None:
|
||||
result["skipped"] = True
|
||||
@@ -16320,7 +16337,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
|
||||
# dashboard/TUI gateway heartbeats used by startup recovery.
|
||||
respect_gateway_heartbeats=False,
|
||||
)
|
||||
|
||||
result["closed"] = len(closed)
|
||||
# Only VACUUM if we actually freed rows, and no more often than
|
||||
# once every min_vacuum_interval_days -- a large prune (e.g. the
|
||||
# first one to cross retention_days on a DB with tens of
|
||||
|
||||
@@ -536,6 +536,46 @@ class TestSweepOrphanedSessions:
|
||||
def test_returns_empty_on_empty_db(self, db):
|
||||
assert db.sweep_orphaned_sessions(max_idle_seconds=IDLE_S) == []
|
||||
|
||||
def test_auto_prune_reports_closed_count_and_deletes_after_second_window(
|
||||
self, db
|
||||
):
|
||||
"""#54189 end-to-end: leaky producers (cron/kanban/subagent) never set
|
||||
``ended_at``; pass 1 closes them (reported via ``closed``), pass 2 —
|
||||
after a further retention window — deletes them, and a messaging row
|
||||
is never touched by either pass."""
|
||||
stale = time.time() - 200 * 86400
|
||||
for sid, source in (
|
||||
("cron-0", "cron"),
|
||||
("kanban-1", "kanban"),
|
||||
("subagent-2", "subagent"),
|
||||
("telegram-3", "telegram"),
|
||||
):
|
||||
_make_session(db, sid, source=source, started_at=stale, message_at=stale)
|
||||
_set_last_activity(db, sid, stale)
|
||||
|
||||
first = db.maybe_auto_prune_and_vacuum(
|
||||
retention_days=90, min_interval_hours=0, vacuum=False
|
||||
)
|
||||
assert first["closed"] == 3
|
||||
assert first["pruned"] == 0
|
||||
for sid in ("cron-0", "kanban-1", "subagent-2"):
|
||||
assert db.get_session(sid)["end_reason"] == "startup_orphan_reap"
|
||||
assert db.get_session("telegram-3")["ended_at"] is None
|
||||
|
||||
# Simulate the next maintenance pass after another retention window.
|
||||
db._conn.execute(
|
||||
"UPDATE sessions SET ended_at = ended_at - 91 * 86400 "
|
||||
"WHERE end_reason = 'startup_orphan_reap'"
|
||||
)
|
||||
db._conn.commit()
|
||||
second = db.maybe_auto_prune_and_vacuum(
|
||||
retention_days=90, min_interval_hours=0, vacuum=False
|
||||
)
|
||||
assert second["closed"] == 0
|
||||
assert second["pruned"] == 3
|
||||
remaining = [r["id"] for r in db._conn.execute("SELECT id FROM sessions")]
|
||||
assert remaining == ["telegram-3"]
|
||||
|
||||
def test_zero_ttl_is_noop(self, db):
|
||||
stale = time.time() - 8 * 3600
|
||||
_make_session(db, "stale-tui", source="tui", started_at=stale, message_at=stale)
|
||||
|
||||
@@ -888,6 +888,19 @@ Active sessions are never auto-pruned, regardless of age. Ended sessions are
|
||||
aged from their latest message, so a long-lived conversation used recently is
|
||||
not deleted merely because it began before the retention window.
|
||||
|
||||
**Stale open sessions from automation.** Some producers — cron jobs, kanban
|
||||
workers, subagents, one-shot CLI runs — can die without ever marking their
|
||||
session ended, and pruning only deletes *ended* rows. To keep those from
|
||||
accumulating forever, each auto-prune pass also *closes* open sessions from
|
||||
those state-owned sources (`cli`, `cron`, `kanban`, `acp`, `api_server`,
|
||||
`subagent`, `tool`) whose last activity is older than `retention_days`
|
||||
(`end_reason: startup_orphan_reap`). Closing is non-destructive — the
|
||||
session stays resumable — and the row is aged from its close, so it is only
|
||||
deleted by a *later* pass after a further full retention window. Messaging
|
||||
platform sessions (Telegram, Discord, …), TUI/desktop sessions, pinned
|
||||
sessions, and sessions with a live turn or compression in progress are
|
||||
never closed by this sweep.
|
||||
|
||||
### Oversized-Transcript Guards
|
||||
|
||||
Two limits stop a runaway transcript from being loaded into memory all at once
|
||||
|
||||
Reference in New Issue
Block a user