fix(state): close leaked SessionDB connections on exception paths (#83226)

SessionDB could leave native SQLite handles open when construction failed
partway through schema/pragma/FTS/repair/lock/interrupt handling. Other
short-lived callers (MCP reads/polling, session search, reactions, trace
upload, insights, shutdown recovery) opened temporary SessionDB handles
without a complete ownership boundary. API-server profile caches and
RetainDB shutdown had similar late-close races. Under sustained load this
exhausted file descriptors (EMFILE).

- Close partially initialized SessionDB connections on every constructor
  exception path via a finally block guarded by an initialization-complete
  flag.
- Close temporary/cross-profile SessionDB handles in finally blocks across
  CLI, MCP, search, trace, reactions, insights, and recovery paths.
- Add API-server per-profile cache ownership and disconnect cleanup.
- Make RetainDB writer-queue shutdown exception-safe: track connections per
  thread, close on worker exit, reject new enqueues after shutdown starts,
  and sweep any connections left by short-lived threads.
- Add regression coverage for constructor failures, worker-thread readers,
  API disconnect failures, shutdown recovery, RetainDB late enqueue, and
  foreign-loop async clients.

Salvage notes: the original PR's per-thread WAL-reader ownership changes
were superseded by main's read-connection pool (permits + checkout/return);
its cron timeout-abandon fix is credited separately to #72822's earlier
identical fix.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
joaomarcos
2026-08-10 23:12:01 -03:00
committed by Teknium
parent 0d91ab8889
commit 39e480c051
25 changed files with 1035 additions and 138 deletions
+45 -25
View File
@@ -79,6 +79,22 @@ def _get_session_db():
return None
def _load_session_messages(session_id: str):
"""Read one session and close the temporary database handle."""
db = _get_session_db()
if db is None:
return None, "Session database unavailable"
try:
return db.get_messages(session_id), None
except Exception as e:
return None, f"Failed to read messages: {e}"
finally:
try:
db.close()
except Exception:
logger.debug("Failed to close MCP SessionDB", exc_info=True)
def _load_sessions_index() -> dict:
"""Load the gateway session routing index.
@@ -448,6 +464,18 @@ class EventBridge:
self._new_event.set()
def _establish_baseline(self) -> None:
db = _get_session_db()
if not db:
return
try:
self._establish_baseline_with_db(db)
finally:
try:
db.close()
except Exception:
logger.debug("Failed to close MCP baseline SessionDB", exc_info=True)
def _establish_baseline_with_db(self, db) -> None:
"""Record the latest per-session message timestamp and the current
state.db mtime WITHOUT emitting events, so startup does not replay
history (#13414).
@@ -457,9 +485,6 @@ class EventBridge:
last_seen=0.0 in _poll_once, so a brand-new conversation's first
message is still delivered on its state.db-change tick.
"""
db = _get_session_db()
if not db:
return
try:
from hermes_constants import get_hermes_home
db_file = get_hermes_home() / "state.db"
@@ -486,7 +511,6 @@ class EventBridge:
latest = max(all_ts)
if latest > 0.0:
self._last_poll_timestamps[session_key] = latest
def _poll_loop(self):
"""Background loop: poll SessionDB for new messages."""
db = _get_session_db()
@@ -494,12 +518,18 @@ class EventBridge:
logger.warning("EventBridge: SessionDB unavailable, event polling disabled")
return
while self._running:
try:
while self._running:
try:
self._poll_once(db)
except Exception as e:
logger.debug("EventBridge poll error: %s", e)
time.sleep(POLL_INTERVAL)
finally:
try:
self._poll_once(db)
except Exception as e:
logger.debug("EventBridge poll error: %s", e)
time.sleep(POLL_INTERVAL)
db.close()
except Exception:
logger.debug("Failed to close MCP polling SessionDB", exc_info=True)
def _poll_once(self, db):
"""Check for new messages across all sessions.
@@ -722,14 +752,9 @@ def create_mcp_server(event_bridge: Optional[EventBridge] = None) -> "FastMCP":
if not session_id:
return json.dumps({"error": "No session ID for this conversation"})
db = _get_session_db()
if not db:
return json.dumps({"error": "Session database unavailable"})
try:
all_messages = db.get_messages(session_id)
except Exception as e:
return json.dumps({"error": f"Failed to read messages: {e}"})
all_messages, error = _load_session_messages(session_id)
if error:
return json.dumps({"error": error})
filtered = []
for msg in all_messages:
@@ -778,14 +803,9 @@ def create_mcp_server(event_bridge: Optional[EventBridge] = None) -> "FastMCP":
if not session_id:
return json.dumps({"error": "No session ID for this conversation"})
db = _get_session_db()
if not db:
return json.dumps({"error": "Session database unavailable"})
try:
all_messages = db.get_messages(session_id)
except Exception as e:
return json.dumps({"error": f"Failed to read messages: {e}"})
all_messages, error = _load_session_messages(session_id)
if error:
return json.dumps({"error": error})
# Find the target message
target_msg = None