Files
hermes-agent/gateway/delivery_ledger.py
T
Teknium 3c7069bdcb refactor(gateway): dead-code removal, helper unification, defensive-layer collapse and rationale-preserving comment compaction across 40 modules
authz_mixin, browser_control_broker, delivery, delivery_ledger, display_config, drain_control,
hosted_room_links/peer/policy_checkpoint, hosted_rooms, platform_registry, relay/__init__,
relay/ws_transport, run.py and slash_commands.py (comments), session_context, session_state,
streaming_tts_consumer, turn_lease and small modules.

- HostedRoomPolicyCheckpoint._apply_event -> per-kind handler table
- WebSocketRelayTransport._handle_frame -> frame-handler table
- GatewayAuthorizationMixin: unified adapter setting/flag/extra readers
- dead symbols removed (verified zero references): RoomLinkProbe/select_room_link,
  relay_bot_username, is_restart_loop_tripped, debug_rows, DeadTargetRegistry.all_dead,
  BrowserControlBroker.detach_owner/_prune_tickets, StreamingTTSConsumer.started/_enqueue_done/
  _iter_stream_chunks/_next_stream_chunk, RecoverableHandleCache.status_for, _auth_env,
  _copy_default_catalog, _parse_timestamp_prefix, _present_* helpers, _send_result_error_kind,
  _truthy_env, SessionFieldView/TurnLeaseTokenView dunder shims, and their orphaned tests.
- lost WHY/invariant text from the earlier compaction restored compactly (541 hunks audited)
2026-09-02 13:30:50 -07:00

488 lines
20 KiB
Python

"""Durable delivery-obligation ledger for gateway final responses.
A final agent response generated but not yet confirmed-delivered is the one
artifact the gateway can lose without a trace: the turn already burned its
tokens, the text exists only in a Python local, and a crash / restart between
finalize and platform ACK drops it silently.
Each outbound final response gets a small durable row in the shared ``state.db``
(same conventions as ``tools.async_delegation`` — WAL, owner pid + process-start
liveness, bounded retention). The gateway writes three checkpoints:
record_obligation() state='pending' before any send attempt
mark_attempting() state='attempting' immediately before the await
mark_delivered() / state='delivered' only on SendResult.success
mark_failed() state='failed' on a definitive rejection
On startup ``sweep_recoverable()`` claims rows whose owner is dead and hands
them back for redelivery. After an adapter reconnects without a restart,
``sweep_failed_for_runtime()`` may claim only the same live process's
allowlisted transient failures. Crash semantics are explicit about ambiguity
(never silently resend an ambiguous send):
- ``pending`` — send never started: redeliver plainly, no dup risk.
- ``attempting`` — crashed mid-await: the platform MAY already have it.
Redelivered WITH a visible recovered-reply marker (honest at-least-once).
- ``failed`` — definitively rejected once; restart is a natural retry
boundary. Also carries the marker.
- ``delivered`` — nothing to do; retention prunes.
Poison rows cannot spin: attempts are capped and stale rows expire, both
transitioning to ``abandoned`` (kept briefly, then pruned).
Everything here is best-effort: ledger failures must never block or delay an
actual send. Callers wrap every call in try/except.
"""
from __future__ import annotations
import hashlib
import logging
import os
import sqlite3
import threading
import time
from contextlib import contextmanager
from typing import Any, Dict, Iterator, List, Optional
from hermes_constants import get_hermes_home
logger = logging.getLogger(__name__)
_DB_LOCK = threading.Lock()
# Redelivery policy knobs (deliberately not config — the ledger is gated by
# ``gateway.delivery_ledger`` and these only matter in the rare recovery path).
MAX_ATTEMPTS = 3
STALE_AFTER_SECONDS = 24 * 60 * 60
_RETENTION_SECONDS = 7 * 24 * 60 * 60
_MAX_ROWS = 500
# Visible prefix for redeliveries that might duplicate an already-received
# message (crash mid-send / post-rejection retry). Honest at-least-once.
RECOVERED_MARKER = (
"♻️ Recovered reply — the gateway restarted during delivery, "
"so this may be a duplicate:\n\n"
)
# Runtime recovery uses a distinct marker: no restart occurred, but a network
# rejection's acknowledgement can still have been lost independently.
RECONNECTED_MARKER = (
"♻️ Recovered reply — the messaging platform reconnected after the original "
"delivery failed, so this may be a duplicate:\n\n"
)
# Runtime replay is fail-closed: only errors whose send contract proves they are
# transient reconnect failures. Permanent rejects (blocked bot, bad auth, missing
# chat) must not be retried merely because an adapter reconnected.
_RUNTIME_RETRYABLE_ERRORS = frozenset({"send_path_degraded"})
def _db_path():
return get_hermes_home() / "state.db"
def _connect() -> sqlite3.Connection:
path = _db_path()
path.parent.mkdir(parents=True, exist_ok=True)
conn = sqlite3.connect(path, timeout=10)
try:
_initialize_schema(conn)
except Exception:
# A PRAGMA/DDL failure after connect() must not leak the connection.
conn.close()
raise
return conn
def _initialize_schema(conn: sqlite3.Connection) -> None:
from hermes_state import apply_wal_with_fallback
apply_wal_with_fallback(conn, db_label="state.db (delivery_ledger)")
conn.execute(
"""CREATE TABLE IF NOT EXISTS delivery_obligations (
obligation_id TEXT PRIMARY KEY,
session_key TEXT NOT NULL,
platform TEXT NOT NULL,
chat_id TEXT NOT NULL,
thread_id TEXT,
content TEXT NOT NULL,
state TEXT NOT NULL,
attempts INTEGER NOT NULL DEFAULT 0,
created_at REAL NOT NULL,
updated_at REAL NOT NULL,
owner_pid INTEGER,
owner_started_at INTEGER,
last_error TEXT,
adapter_profile TEXT
)"""
)
columns = {row[1] for row in conn.execute("PRAGMA table_info(delivery_obligations)")}
if "adapter_profile" not in columns:
try:
conn.execute("ALTER TABLE delivery_obligations ADD COLUMN adapter_profile TEXT")
except sqlite3.OperationalError as exc:
# Concurrent first-use connections can both observe the old schema.
if "duplicate column" not in str(exc).lower():
raise
@contextmanager
def _transaction() -> Iterator[sqlite3.Connection]:
"""Open a connection, commit/rollback on exit, and ALWAYS close it.
``sqlite3.Connection`` as a context manager only commits/rolls back; it does
not close. ``with _connect()`` alone would leak a connection (and its WAL/SHM
fds) per call until GC — ``record_obligation`` runs on every outbound final
response, so this ledger would exhaust ``RLIMIT_NOFILE`` on a long-running gateway.
"""
conn = _connect()
try:
with conn:
yield conn
finally:
conn.close()
def _owner_stamp() -> tuple[int, Optional[int]]:
pid = os.getpid()
try:
from gateway.status import get_process_start_time
return pid, get_process_start_time(pid)
except Exception:
return pid, None
def _owner_alive(pid: Any, started_at: Any) -> bool:
"""True when the recorded owning process still exists (pid + start time)."""
if not pid:
return False
try:
pid = int(pid)
except (TypeError, ValueError):
return False
try:
from gateway.status import get_process_start_time
current_start = get_process_start_time(pid)
except Exception:
current_start = None
if current_start is None:
# Start time unreadable: alive iff the pid exists. Route through the
# cross-platform probe — ``os.kill(pid, 0)`` on Windows is NOT a no-op
# (bpo-14484: it maps to ``GenerateConsoleCtrlEvent(0, pid)`` and could
# Ctrl+C the gateway's own console group). ``_pid_exists`` keeps the
# EPERM-means-alive semantics (pid exists but is owned by another user).
try:
from gateway.status import _pid_exists
except Exception:
if os.name == "nt":
return False # never fall back to a raw sig-0 probe on Windows
try:
os.kill(pid, 0) # windows-footgun: ok — POSIX-only fallback branch
except ProcessLookupError:
return False
except PermissionError:
return True
except OSError:
return False
return True
try:
return bool(_pid_exists(pid))
except Exception:
return False
if started_at is None:
return True
try:
return int(current_start) == int(started_at)
except (TypeError, ValueError):
return True
def compute_obligation_id(session_key: str, message_ref: str, content: str) -> str:
"""Stable id: same turn + same content re-records idempotently, while distinct
threads/topics on one chat never collide (session_key carries platform, chat and
thread; ``message_ref`` is the triggering inbound message id)."""
payload = f"{session_key}|{message_ref}|{content}"
return hashlib.sha256(payload.encode("utf-8", "replace")).hexdigest()[:24]
def record_obligation(
*,
obligation_id: str,
session_key: str,
platform: str,
chat_id: str,
thread_id: Optional[str],
content: str,
adapter_profile: Optional[str] = None,
) -> None:
"""Record a final response as owed to the platform (state='pending')."""
now = time.time()
stored_profile = str(adapter_profile).strip() if adapter_profile else "default"
pid, started = _owner_stamp()
with _DB_LOCK, _transaction() as conn:
conn.execute(
"""INSERT OR REPLACE INTO delivery_obligations
(obligation_id, session_key, platform, chat_id, thread_id,
content, state, attempts, created_at, updated_at,
owner_pid, owner_started_at, adapter_profile)
VALUES (?, ?, ?, ?, ?, ?, 'pending', 0, ?, ?, ?, ?, ?)""",
(obligation_id, session_key, platform, str(chat_id),
str(thread_id) if thread_id else None, content, now, now,
pid, started, stored_profile),
)
_prune()
def mark_attempting(obligation_id: str) -> None:
_update_state(obligation_id, "attempting")
def mark_delivered(obligation_id: str) -> None:
_update_state(obligation_id, "delivered")
def mark_failed(obligation_id: str, error: str = "") -> None:
_update_state(obligation_id, "failed", error=error)
def release_runtime_claim(obligation_id: str, error: str = "") -> bool:
"""Return an unsent runtime claim to ``failed`` without spending an attempt.
Runtime recovery claims before clearing ``resume_pending`` so two reconnect
paths cannot send the same row. If the session flag cannot be cleared, no
send was attempted and the claim must not consume the redelivery budget.
Fail-closed to the exact current process instance and ``attempting`` state.
"""
pid, started = _owner_stamp()
if started is None:
return False
with _DB_LOCK, _transaction() as conn:
cursor = conn.execute(
"""UPDATE delivery_obligations
SET state='failed', attempts=CASE
WHEN attempts > 0 THEN attempts - 1 ELSE 0 END,
updated_at=?, last_error=?
WHERE obligation_id=? AND state='attempting'
AND owner_pid IS ? AND owner_started_at IS ?""",
(time.time(), error[:500] if error else None,
obligation_id, pid, started),
)
return bool(cursor.rowcount)
def _update_state(obligation_id: str, state: str, error: str = "") -> None:
with _DB_LOCK, _transaction() as conn:
conn.execute(
"""UPDATE delivery_obligations
SET state=?, updated_at=?, last_error=?
WHERE obligation_id=?""",
(state, time.time(), error[:500] if error else None, obligation_id),
)
def _exhausted(attempts: int, created_at: float, now: float) -> bool:
return attempts >= MAX_ATTEMPTS or (now - created_at) > STALE_AFTER_SECONDS
def sweep_recoverable(
now: Optional[float] = None,
*,
deliverable_platforms: Optional[set] = None,
deliverable_targets: Optional[set] = None,
) -> List[Dict[str, Any]]:
"""Claim undelivered rows owned by dead processes; return them for redelivery.
Claiming atomically re-stamps the owner to THIS process and increments
``attempts`` (the UPDATE is guarded on the previous owner stamp, so a second
gateway racing the same sweep cannot double-claim). Rows over the attempts
cap or stale cutoff transition to 'abandoned' instead of being returned.
``deliverable_platforms`` restricts claiming to platforms the caller can
actually send on this boot: ``attempts`` is the redelivery budget and must
only be spent on a real send, else a platform that failed to connect burns
one attempt per boot and hits the cap having never been sent once. Rows for
absent platforms are left untouched (the stale cutoff still bounds them).
``deliverable_targets`` further scopes multiplexed gateways by exact
``(platform, adapter_profile)`` so one connected bot cannot spend another
disconnected bot's retry budget.
"""
now = now if now is not None else time.time()
pid, started = _owner_stamp()
claimed: List[Dict[str, Any]] = []
with _DB_LOCK, _transaction() as conn:
rows = conn.execute(
"""SELECT obligation_id, session_key, platform, chat_id, thread_id,
content, state, attempts, created_at,
owner_pid, owner_started_at, adapter_profile
FROM delivery_obligations
WHERE state IN ('pending', 'attempting', 'failed')"""
).fetchall()
for (oid, session_key, platform, chat_id, thread_id, content, state,
attempts, created_at, owner_pid, owner_started_at,
adapter_profile) in rows:
if _owner_alive(owner_pid, owner_started_at):
continue # a live gateway still owns this row
if _exhausted(attempts, created_at, now):
conn.execute(
"""UPDATE delivery_obligations
SET state='abandoned', updated_at=? WHERE obligation_id=?""",
(now, oid),
)
continue
if deliverable_platforms is not None and platform not in deliverable_platforms:
continue # no adapter this boot — claiming would spend an attempt on a no-op
if deliverable_targets is not None and (platform, adapter_profile) not in deliverable_targets:
continue
cursor = conn.execute(
"""UPDATE delivery_obligations
SET owner_pid=?, owner_started_at=?, attempts=attempts+1,
updated_at=?
WHERE obligation_id=? AND (owner_pid IS ? OR owner_pid=?)""",
(pid, started, now, oid, owner_pid, owner_pid),
)
if cursor.rowcount:
claimed.append({
"obligation_id": oid,
"session_key": session_key,
"platform": platform,
"chat_id": chat_id,
"thread_id": thread_id,
"content": content,
# pending = never started, redeliver plainly; else carry marker.
"needs_marker": state != "pending",
"profile": adapter_profile,
"attempts": attempts + 1,
})
return claimed
def sweep_failed_for_runtime(
platform: str,
now: Optional[float] = None,
*,
profile: Optional[str] = None,
) -> List[Dict[str, Any]]:
"""Claim this process's reconnect-retryable failed rows for one adapter.
``profile`` scopes multiplexed gateways to the bot identity that owned the
failed send; ``None`` means the primary/default adapter. The persisted
adapter owner is independent of the routed session namespace. Unowned rows
and rows owned by another process are left for the startup/dead-owner sweep.
Startup recovery ignores rows owned by a live gateway, so a final response
rejected with ``send_path_degraded`` would stay stranded when only the
adapter reconnects. This sweep closes that gap without weakening ownership:
only rows stamped to this exact process instance, only allowlisted transient
errors, same attempts/staleness bounds, every update guarded by the prior
owner stamp and ``failed`` state. Claimed rows always carry the reconnect
marker because the failed send's acknowledgement is not safe to infer.
"""
now = now if now is not None else time.time()
pid, started = _owner_stamp()
if started is None:
# PID alone cannot distinguish this process from a stale row left after
# PID reuse. Runtime replay is optional, so fail closed; startup
# recovery remains the durable fallback.
return []
expected_profile = "default" if not profile or profile == "default" else str(profile)
claimed: List[Dict[str, Any]] = []
with _DB_LOCK, _transaction() as conn:
rows = conn.execute(
"""SELECT obligation_id, session_key, platform, chat_id, thread_id,
content, attempts, created_at, owner_pid,
owner_started_at, last_error, adapter_profile
FROM delivery_obligations
WHERE state='failed' AND platform=?""",
(platform,),
).fetchall()
for (oid, session_key, row_platform, chat_id, thread_id, content,
attempts, created_at, owner_pid, owner_started_at, last_error,
adapter_profile) in rows:
if adapter_profile != expected_profile:
continue
# Exact process-start matching prevents PID reuse from stealing work.
if owner_pid != pid or owner_started_at != started:
continue
if str(last_error or "").strip().lower() not in _RUNTIME_RETRYABLE_ERRORS:
continue
owner_guard = (oid, owner_pid, owner_started_at)
if _exhausted(attempts, created_at, now):
conn.execute(
"""UPDATE delivery_obligations
SET state='abandoned', updated_at=?
WHERE obligation_id=? AND state='failed'
AND owner_pid IS ? AND owner_started_at IS ?""",
(now, *owner_guard),
)
continue
cursor = conn.execute(
"""UPDATE delivery_obligations
SET state='attempting', attempts=attempts+1, updated_at=?
WHERE obligation_id=? AND state='failed'
AND owner_pid IS ? AND owner_started_at IS ?""",
(now, *owner_guard),
)
if cursor.rowcount:
claimed.append({
"obligation_id": oid,
"session_key": session_key,
"platform": row_platform,
"chat_id": chat_id,
"thread_id": thread_id,
"content": content,
"needs_marker": True,
"marker": RECONNECTED_MARKER,
"profile": adapter_profile,
"runtime_recovery": True,
"attempts": attempts + 1,
})
return claimed
def _prune(now: Optional[float] = None) -> None:
now = now if now is not None else time.time()
cutoff = now - _RETENTION_SECONDS
try:
with _transaction() as conn:
conn.execute(
"""DELETE FROM delivery_obligations
WHERE state IN ('delivered', 'abandoned') AND updated_at < ?""",
(cutoff,),
)
total = conn.execute("SELECT COUNT(*) FROM delivery_obligations").fetchone()[0]
excess = max(0, total - _MAX_ROWS)
if excess:
conn.execute(
"""DELETE FROM delivery_obligations WHERE obligation_id IN (
SELECT obligation_id FROM delivery_obligations
ORDER BY CASE state
WHEN 'delivered' THEN 0
WHEN 'abandoned' THEN 1
ELSE 2
END, updated_at ASC
LIMIT ?)""",
(excess,),
)
except Exception:
logger.debug("delivery ledger prune failed", exc_info=True)
def ledger_enabled(config: Optional[Dict[str, Any]] = None) -> bool:
"""Read the ``gateway.delivery_ledger`` config gate (default on)."""
try:
if config is None:
from hermes_cli.config import load_config
config = load_config()
gw = config.get("gateway") or {}
value = gw.get("delivery_ledger", True)
if isinstance(value, str):
return value.strip().lower() not in {"false", "0", "no", "off"}
return bool(value)
except Exception:
return True